From b87c41da5d09945ee0a3f83f5e0c4f441e958fc3 Mon Sep 17 00:00:00 2001 From: Jessica_Natalia Date: Tue, 18 Aug 2026 18:15:20 -0300 Subject: [PATCH] =?UTF-8?q?otimiza=C3=A7=C3=B5es=20round=2014=20(severo),?= =?UTF-8?q?=20radio=20fix?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit otimizações round 14 (severo), radio fix --- include/psprecomp/allegrex_context.hpp | 368 ++++- .../vcs/generated/generated_unit_0000.cpp | 33 +- .../vcs/generated/generated_unit_0001.cpp | 6 +- .../vcs/generated/generated_unit_0002.cpp | 110 +- .../vcs/generated/generated_unit_0003.cpp | 44 +- .../vcs/generated/generated_unit_0004.cpp | 84 +- .../vcs/generated/generated_unit_0005.cpp | 113 +- .../vcs/generated/generated_unit_0006.cpp | 165 +-- .../vcs/generated/generated_unit_0007.cpp | 235 +--- .../vcs/generated/generated_unit_0008.cpp | 159 +-- .../vcs/generated/generated_unit_0009.cpp | 365 +---- .../vcs/generated/generated_unit_0010.cpp | 47 +- .../vcs/generated/generated_unit_0011.cpp | 224 +-- .../vcs/generated/generated_unit_0012.cpp | 112 +- .../vcs/generated/generated_unit_0014.cpp | 5 +- .../vcs/generated/generated_unit_0016.cpp | 268 +--- .../vcs/generated/generated_unit_0017.cpp | 256 +--- .../vcs/generated/generated_unit_0018.cpp | 96 +- .../vcs/generated/generated_unit_0022.cpp | 254 +--- .../vcs/generated/generated_unit_0023.cpp | 104 +- .../vcs/generated/generated_unit_0024.cpp | 11 +- .../vcs/generated/generated_unit_0025.cpp | 57 +- .../vcs/generated/generated_unit_0026.cpp | 34 +- .../vcs/generated/generated_unit_0027.cpp | 49 +- .../vcs/generated/generated_unit_0030.cpp | 234 +--- .../vcs/generated/generated_unit_0031.cpp | 163 +-- .../vcs/generated/generated_unit_0033.cpp | 34 +- .../vcs/generated/generated_unit_0034.cpp | 112 +- .../vcs/generated/generated_unit_0035.cpp | 298 +--- .../vcs/generated/generated_unit_0036.cpp | 70 +- .../vcs/generated/generated_unit_0037.cpp | 11 +- .../vcs/generated/generated_unit_0038.cpp | 184 +-- .../vcs/generated/generated_unit_0039.cpp | 179 +-- .../vcs/generated/generated_unit_0040.cpp | 208 +-- .../vcs/generated/generated_unit_0041.cpp | 266 +--- .../vcs/generated/generated_unit_0042.cpp | 513 ++----- .../vcs/generated/generated_unit_0043.cpp | 1224 +++-------------- .../vcs/generated/generated_unit_0044.cpp | 215 +-- .../vcs/generated/generated_unit_0045.cpp | 9 +- .../vcs/generated/generated_unit_0047.cpp | 6 +- .../vcs/generated/generated_unit_0048.cpp | 11 +- .../vcs/generated/generated_unit_0049.cpp | 35 +- .../vcs/generated/generated_unit_0051.cpp | 10 +- .../vcs/generated/generated_unit_0054.cpp | 54 +- .../vcs/generated/generated_unit_0056.cpp | 166 +-- .../vcs/generated/generated_unit_0057.cpp | 115 +- .../vcs/generated/generated_unit_0058.cpp | 199 +-- .../vcs/generated/generated_unit_0059.cpp | 347 +---- .../vcs/generated/generated_unit_0060.cpp | 72 +- .../vcs/generated/generated_unit_0061.cpp | 41 +- .../vcs/generated/generated_unit_0062.cpp | 44 +- .../vcs/generated/generated_unit_0063.cpp | 226 +-- .../vcs/generated/generated_unit_0064.cpp | 217 +-- .../vcs/generated/generated_unit_0065.cpp | 43 +- .../vcs/generated/generated_unit_0066.cpp | 230 +--- .../vcs/generated/generated_unit_0067.cpp | 38 +- .../vcs/generated/generated_unit_0068.cpp | 58 +- .../vcs/generated/generated_unit_0069.cpp | 173 +-- .../vcs/generated/generated_unit_0070.cpp | 262 +--- .../vcs/generated/generated_unit_0071.cpp | 63 +- .../vcs/generated/generated_unit_0072.cpp | 99 +- .../vcs/generated/generated_unit_0073.cpp | 133 +- .../vcs/generated/generated_unit_0074.cpp | 507 +------ .../vcs/generated/generated_unit_0075.cpp | 125 +- .../vcs/generated/generated_unit_0076.cpp | 56 +- .../vcs/generated/generated_unit_0077.cpp | 22 +- .../vcs/generated/generated_unit_0078.cpp | 154 +-- .../vcs/generated/generated_unit_0079.cpp | 340 +---- .../vcs/generated/generated_unit_0080.cpp | 72 +- .../vcs/generated/generated_unit_0081.cpp | 39 +- .../vcs/generated/generated_unit_0082.cpp | 208 +-- .../vcs/generated/generated_unit_0083.cpp | 229 +-- .../vcs/generated/generated_unit_0084.cpp | 114 +- .../vcs/generated/generated_unit_0085.cpp | 333 +---- .../vcs/generated/generated_unit_0086.cpp | 357 +---- .../vcs/generated/generated_unit_0087.cpp | 83 +- .../vcs/generated/generated_unit_0088.cpp | 180 +-- .../vcs/generated/generated_unit_0089.cpp | 227 +-- .../vcs/generated/generated_unit_0090.cpp | 84 +- .../vcs/generated/generated_unit_0091.cpp | 71 +- .../vcs/generated/generated_unit_0092.cpp | 237 +--- .../vcs/generated/generated_unit_0093.cpp | 501 +------ .../vcs/generated/generated_unit_0095.cpp | 94 +- .../vcs/generated/generated_unit_0096.cpp | 131 +- .../vcs/generated/generated_unit_0097.cpp | 237 +--- .../vcs/generated/generated_unit_0098.cpp | 64 +- .../vcs/generated/generated_unit_0101.cpp | 433 ++---- .../vcs/generated/generated_unit_0102.cpp | 620 ++------- .../vcs/generated/generated_unit_0103.cpp | 632 ++------- .../vcs/generated/generated_unit_0104.cpp | 613 ++------- .../vcs/generated/generated_unit_0105.cpp | 530 +------ .../vcs/generated/generated_unit_0106.cpp | 152 +- .../vcs/generated/generated_unit_0107.cpp | 178 +-- .../vcs/generated/generated_unit_0108.cpp | 188 +-- .../vcs/generated/generated_unit_0109.cpp | 50 +- .../vcs/generated/generated_unit_0111.cpp | 11 +- .../vcs/generated/generated_unit_0112.cpp | 74 +- .../vcs/generated/generated_unit_0113.cpp | 243 +--- .../vcs/generated/generated_unit_0114.cpp | 302 +--- .../vcs/generated/generated_unit_0115.cpp | 22 +- .../vcs/generated/generated_unit_0116.cpp | 5 +- .../vcs/generated/generated_unit_0117.cpp | 205 +-- .../vcs/generated/generated_unit_0118.cpp | 142 +- .../vcs/generated/generated_unit_0119.cpp | 22 +- .../vcs/generated/generated_unit_0120.cpp | 18 +- .../vcs/generated/generated_unit_0121.cpp | 66 +- .../vcs/generated/generated_unit_0122.cpp | 370 +---- .../vcs/generated/generated_unit_0123.cpp | 253 +--- .../vcs/generated/generated_unit_0124.cpp | 159 +-- .../vcs/generated/generated_unit_0125.cpp | 59 +- .../vcs/generated/generated_unit_0126.cpp | 325 +---- .../vcs/generated/generated_unit_0127.cpp | 187 +-- .../vcs/generated/generated_unit_0129.cpp | 380 +---- .../vcs/generated/generated_unit_0130.cpp | 280 +--- .../vcs/generated/generated_unit_0131.cpp | 727 +--------- .../vcs/generated/generated_unit_0132.cpp | 219 +-- .../vcs/generated/generated_unit_0133.cpp | 354 +---- .../vcs/generated/generated_unit_0134.cpp | 275 +--- .../vcs/generated/generated_unit_0135.cpp | 409 ++---- .../vcs/generated/generated_unit_0136.cpp | 472 ++----- .../vcs/generated/generated_unit_0137.cpp | 116 +- .../vcs/generated/generated_unit_0138.cpp | 36 +- .../vcs/generated/generated_unit_0139.cpp | 129 +- .../vcs/generated/generated_unit_0140.cpp | 35 +- .../vcs/generated/generated_unit_0141.cpp | 82 +- .../vcs/generated/generated_unit_0142.cpp | 156 +-- .../vcs/generated/generated_unit_0143.cpp | 178 +-- .../vcs/generated/generated_unit_0144.cpp | 185 +-- .../vcs/generated/generated_unit_0145.cpp | 244 +--- .../vcs/generated/generated_unit_0146.cpp | 472 +------ .../vcs/generated/generated_unit_0147.cpp | 216 +-- .../vcs/generated/generated_unit_0148.cpp | 99 +- .../vcs/generated/generated_unit_0149.cpp | 180 +-- .../vcs/generated/generated_unit_0151.cpp | 371 +---- .../vcs/generated/generated_unit_0152.cpp | 384 +----- .../vcs/generated/generated_unit_0153.cpp | 285 +--- .../vcs/generated/generated_unit_0154.cpp | 593 ++------ .../vcs/generated/generated_unit_0155.cpp | 298 +--- .../vcs/generated/generated_unit_0156.cpp | 33 +- .../vcs/generated/generated_unit_0157.cpp | 45 +- .../vcs/generated/generated_unit_0158.cpp | 102 +- .../vcs/generated/generated_unit_0160.cpp | 137 +- .../vcs/generated/generated_unit_0162.cpp | 190 +-- .../vcs/generated/generated_unit_0163.cpp | 312 +---- .../vcs/generated/generated_unit_0164.cpp | 113 +- .../vcs/generated/generated_unit_0165.cpp | 56 +- .../vcs/generated/generated_unit_0166.cpp | 45 +- .../vcs/generated/generated_unit_0168.cpp | 334 +---- .../vcs/generated/generated_unit_0169.cpp | 95 +- .../vcs/generated/generated_unit_0172.cpp | 22 +- .../vcs/generated/generated_unit_0176.cpp | 192 +-- .../vcs/generated/generated_unit_0178.cpp | 65 +- .../vcs/generated/generated_unit_0179.cpp | 208 +-- .../vcs/generated/generated_unit_0183.cpp | 110 +- .../vcs/generated/generated_unit_0184.cpp | 98 +- .../vcs/generated/generated_unit_0185.cpp | 51 +- .../vcs/generated/generated_unit_0186.cpp | 23 +- .../vcs/generated/generated_unit_0188.cpp | 32 +- .../vcs/generated/generated_unit_0189.cpp | 147 +- .../vcs/generated/generated_unit_0190.cpp | 223 +-- .../vcs/generated/generated_unit_0191.cpp | 46 +- .../vcs/generated/generated_unit_0192.cpp | 230 +--- .../vcs/generated/generated_unit_0193.cpp | 265 +--- .../vcs/generated/generated_unit_0194.cpp | 68 +- .../vcs/generated/generated_unit_0195.cpp | 412 +----- .../vcs/generated/generated_unit_0196.cpp | 61 +- .../vcs/generated/generated_unit_0197.cpp | 161 +-- .../vcs/generated/generated_unit_0198.cpp | 47 +- .../vcs/generated/generated_unit_0199.cpp | 194 +-- .../vcs/generated/generated_unit_0200.cpp | 99 +- .../vcs/generated/generated_unit_0201.cpp | 64 +- .../vcs/generated/generated_unit_0202.cpp | 23 +- .../vcs/generated/generated_unit_0203.cpp | 62 +- .../vcs/generated/generated_unit_0204.cpp | 129 +- .../vcs/generated/generated_unit_0205.cpp | 14 +- .../vcs/generated/generated_unit_0206.cpp | 136 +- .../vcs/generated/generated_unit_0207.cpp | 323 +---- .../vcs/generated/generated_unit_0208.cpp | 134 +- .../vcs/generated/generated_unit_0209.cpp | 337 +---- .../vcs/generated/generated_unit_0210.cpp | 364 +---- .../vcs/generated/generated_unit_0219.cpp | 20 +- .../vcs/generated/generated_unit_0220.cpp | 24 +- .../vcs/generated/generated_unit_0221.cpp | 4 +- .../vcs/generated/generated_unit_0222.cpp | 62 +- .../vcs/generated/generated_unit_0223.cpp | 102 +- .../vcs/generated/generated_unit_0224.cpp | 30 +- .../vcs/generated/generated_unit_0225.cpp | 68 +- .../vcs/generated/generated_unit_0231.cpp | 10 +- .../vcs/generated/generated_unit_0232.cpp | 6 +- profiles/vcs/host/vcs_profile.cpp | 145 +- profiles/vcs/host/vcs_runtime_log.cpp | 6 +- .../vcs/host/vcs_tier2_cluster_edge43.cpp | 126 +- .../vcs/host/vcs_tier2_cluster_entity.cpp | 117 +- .../vcs/host/vcs_tier2_cluster_geometry.cpp | 310 +---- .../vcs/host/vcs_tier2_cluster_matrix.cpp | 183 +-- .../vcs/host/vcs_tier2_cluster_physics.cpp | 121 +- profiles/vcs/host/vcs_tier2_cluster_world.cpp | 38 +- profiles/vcs/scripts/build_release_ninja.bat | 41 +- .../tests/check_v827_save_thread_lifecycle.py | 6 +- .../check_v827a_internal_save_repro_gate.py | 6 +- .../vcs/tests/check_v84_aggressive_cpu.py | 6 +- .../vcs/tests/check_v85_aggressive_vfpu.py | 72 + .../vcs/tests/check_v86_radio_vfpu_ct2.py | 46 + profiles/vcs/tests/vfpu_tier2_tests.cpp | 55 +- .../vcs/tools/optimize_generated_v85_vfpu.py | 76 + profiles/vcs/tools/vcs_codegen_main.cpp | 100 +- tests/test_main.cpp | 4 +- tools/codegen_main.cpp | 136 +- 208 files changed, 5887 insertions(+), 28895 deletions(-) create mode 100644 profiles/vcs/tests/check_v85_aggressive_vfpu.py create mode 100644 profiles/vcs/tests/check_v86_radio_vfpu_ct2.py create mode 100644 profiles/vcs/tools/optimize_generated_v85_vfpu.py diff --git a/include/psprecomp/allegrex_context.hpp b/include/psprecomp/allegrex_context.hpp index 24709c6..dce4faa 100644 --- a/include/psprecomp/allegrex_context.hpp +++ b/include/psprecomp/allegrex_context.hpp @@ -316,6 +316,138 @@ struct alignas(16) AllegrexContext { eat_vfpu_prefixes(); } + template + PSPRECOMP_CONTEXT_FORCEINLINE void execute_vfpu_vec3_ct() noexcept { + static_assert(Length >= 1u && Length <= 4u); + static_assert(Operation <= 3u); + // V8.5: VCS almost always executes VFPU arithmetic with the architectural + // S/T/D prefixes already at their consumed defaults (E4/E4/0). In that + // case avoid three temporary vectors, two prefix decoders and the + // destination-prefix writer entirely. Keep a fully architectural + // fallback for the rare explicit VPFX instruction. + if (vfpu_ctrl[0] == 0xE4u && vfpu_ctrl[1] == 0xE4u && vfpu_ctrl[2] == 0u) { + constexpr std::size_t s0i = vfpu_vector_lane_index(SourceRegister, Length, 0u); + constexpr std::size_t t0i = vfpu_vector_lane_index(TargetRegister, Length, 0u); + constexpr std::size_t d0i = vfpu_vector_lane_index(DestinationRegister, Length, 0u); + const float s0 = vfpu[s0i], t0 = vfpu[t0i]; + float s1 = 0.0f, s2 = 0.0f, s3 = 0.0f; + float t1 = 0.0f, t2 = 0.0f, t3 = 0.0f; + if constexpr (Length >= 2u) { + constexpr std::size_t s1i = vfpu_vector_lane_index(SourceRegister, Length, 1u); + constexpr std::size_t t1i = vfpu_vector_lane_index(TargetRegister, Length, 1u); + s1 = vfpu[s1i]; t1 = vfpu[t1i]; + } + if constexpr (Length >= 3u) { + constexpr std::size_t s2i = vfpu_vector_lane_index(SourceRegister, Length, 2u); + constexpr std::size_t t2i = vfpu_vector_lane_index(TargetRegister, Length, 2u); + s2 = vfpu[s2i]; t2 = vfpu[t2i]; + } + if constexpr (Length >= 4u) { + constexpr std::size_t s3i = vfpu_vector_lane_index(SourceRegister, Length, 3u); + constexpr std::size_t t3i = vfpu_vector_lane_index(TargetRegister, Length, 3u); + s3 = vfpu[s3i]; t3 = vfpu[t3i]; + } + auto op = [](float a, float b) noexcept { + if constexpr (Operation == 0u) return a + b; + else if constexpr (Operation == 1u) return a - b; + else if constexpr (Operation == 2u) return a * b; + else return a / b; + }; + const float r0 = op(s0, t0); + float r1 = 0.0f, r2 = 0.0f, r3 = 0.0f; + if constexpr (Length >= 2u) r1 = op(s1, t1); + if constexpr (Length >= 3u) r2 = op(s2, t2); + if constexpr (Length >= 4u) r3 = op(s3, t3); + vfpu[d0i] = r0; + if constexpr (Length >= 2u) { + constexpr std::size_t d1i = vfpu_vector_lane_index(DestinationRegister, Length, 1u); + vfpu[d1i] = r1; + } + if constexpr (Length >= 3u) { + constexpr std::size_t d2i = vfpu_vector_lane_index(DestinationRegister, Length, 2u); + vfpu[d2i] = r2; + } + if constexpr (Length >= 4u) { + constexpr std::size_t d3i = vfpu_vector_lane_index(DestinationRegister, Length, 3u); + vfpu[d3i] = r3; + } + return; + } + float source[4]{}, target[4]{}, result[4]{}; + read_vfpu_vector_with_source_prefix_ct(source); + read_vfpu_vector_with_source_prefix_ct(target); + auto op = [](float a, float b) noexcept { + if constexpr (Operation == 0u) return a + b; + else if constexpr (Operation == 1u) return a - b; + else if constexpr (Operation == 2u) return a * b; + else return a / b; + }; + result[0] = op(source[0], target[0]); + if constexpr (Length >= 2u) result[1] = op(source[1], target[1]); + if constexpr (Length >= 3u) result[2] = op(source[2], target[2]); + if constexpr (Length >= 4u) result[3] = op(source[3], target[3]); + write_vfpu_vector_with_destination_prefix_ct(result); + } + + template + [[nodiscard]] PSPRECOMP_CONTEXT_FORCEINLINE static float vfpu_unary_lane(float value) noexcept { + if constexpr (Operation == 0u) return value; + else if constexpr (Operation == 1u) return std::fabs(value); + else if constexpr (Operation == 2u) return -value; + else if constexpr (Operation == 4u) return value <= 0.0f ? 0.0f : (value > 1.0f ? 1.0f : value); + else if constexpr (Operation == 5u) return value < -1.0f ? -1.0f : (value > 1.0f ? 1.0f : value); + else if constexpr (Operation == 16u) return 1.0f / value; + else if constexpr (Operation == 17u) return 1.0f / std::sqrt(value); + else if constexpr (Operation == 18u) return std::sin(value * 1.57079632679489661923f); + else if constexpr (Operation == 19u) return std::cos(value * 1.57079632679489661923f); + else if constexpr (Operation == 20u) return std::exp2(value); + else if constexpr (Operation == 21u) return std::log2(value); + else if constexpr (Operation == 22u) return std::fabs(std::sqrt(value)); + else if constexpr (Operation == 23u) return std::asin(value) * 0.63661977236758134308f; + else if constexpr (Operation == 24u) return -1.0f / value; + else if constexpr (Operation == 26u) return -std::sin(value * 1.57079632679489661923f); + else return 1.0f / std::exp2(value); + } + + template + PSPRECOMP_CONTEXT_FORCEINLINE void execute_vfpu_unary_ct() noexcept { + static_assert(Length >= 1u && Length <= 4u); + if (vfpu_ctrl[0] == 0xE4u && vfpu_ctrl[1] == 0xE4u && vfpu_ctrl[2] == 0u) { + constexpr std::size_t s0i = vfpu_vector_lane_index(SourceRegister, Length, 0u); + constexpr std::size_t d0i = vfpu_vector_lane_index(DestinationRegister, Length, 0u); + const float s0 = vfpu[s0i]; + float s1 = 0.0f, s2 = 0.0f, s3 = 0.0f; + if constexpr (Length >= 2u) { + constexpr std::size_t s1i = vfpu_vector_lane_index(SourceRegister, Length, 1u); s1 = vfpu[s1i]; + } + if constexpr (Length >= 3u) { + constexpr std::size_t s2i = vfpu_vector_lane_index(SourceRegister, Length, 2u); s2 = vfpu[s2i]; + } + if constexpr (Length >= 4u) { + constexpr std::size_t s3i = vfpu_vector_lane_index(SourceRegister, Length, 3u); s3 = vfpu[s3i]; + } + const float r0 = vfpu_unary_lane(s0); + float r1 = 0.0f, r2 = 0.0f, r3 = 0.0f; + if constexpr (Length >= 2u) r1 = vfpu_unary_lane(s1); + if constexpr (Length >= 3u) r2 = vfpu_unary_lane(s2); + if constexpr (Length >= 4u) r3 = vfpu_unary_lane(s3); + vfpu[d0i] = r0; + if constexpr (Length >= 2u) { constexpr std::size_t d1i = vfpu_vector_lane_index(DestinationRegister, Length, 1u); vfpu[d1i] = r1; } + if constexpr (Length >= 3u) { constexpr std::size_t d2i = vfpu_vector_lane_index(DestinationRegister, Length, 2u); vfpu[d2i] = r2; } + if constexpr (Length >= 4u) { constexpr std::size_t d3i = vfpu_vector_lane_index(DestinationRegister, Length, 3u); vfpu[d3i] = r3; } + return; + } + float source[4]{}, result[4]{}; + read_vfpu_vector_with_source_prefix_ct(source); + result[0] = vfpu_unary_lane(source[0]); + if constexpr (Length >= 2u) result[1] = vfpu_unary_lane(source[1]); + if constexpr (Length >= 3u) result[2] = vfpu_unary_lane(source[2]); + if constexpr (Length >= 4u) result[3] = vfpu_unary_lane(source[3]); + write_vfpu_vector_with_destination_prefix_ct(result); + } + void read_vfpu_vector(float *destination, std::uint32_t vector_register, std::uint32_t length) const noexcept { if (length == 1u) { destination[0] = vfpu[vfpu_scalar_index(vector_register & 0x7Fu)]; @@ -704,6 +836,29 @@ struct alignas(16) AllegrexContext { std::uint32_t TargetRegister, std::uint32_t Length> PSPRECOMP_CONTEXT_FORCEINLINE void execute_vfpu_vdot_ct() noexcept { static_assert(Length >= 1u && Length <= 4u); + if (vfpu_ctrl[0] == 0xE4u && vfpu_ctrl[1] == 0xE4u && vfpu_ctrl[2] == 0u) { + constexpr std::size_t s0i = vfpu_vector_lane_index(SourceRegister, Length, 0u); + constexpr std::size_t t0i = vfpu_vector_lane_index(TargetRegister, Length, 0u); + float sum = vfpu[s0i] * vfpu[t0i]; + if constexpr (Length >= 2u) { + constexpr std::size_t si = vfpu_vector_lane_index(SourceRegister, Length, 1u); + constexpr std::size_t ti = vfpu_vector_lane_index(TargetRegister, Length, 1u); + sum += vfpu[si] * vfpu[ti]; + } + if constexpr (Length >= 3u) { + constexpr std::size_t si = vfpu_vector_lane_index(SourceRegister, Length, 2u); + constexpr std::size_t ti = vfpu_vector_lane_index(TargetRegister, Length, 2u); + sum += vfpu[si] * vfpu[ti]; + } + if constexpr (Length >= 4u) { + constexpr std::size_t si = vfpu_vector_lane_index(SourceRegister, Length, 3u); + constexpr std::size_t ti = vfpu_vector_lane_index(TargetRegister, Length, 3u); + sum += vfpu[si] * vfpu[ti]; + } + constexpr std::size_t di = vfpu_vector_lane_index(DestinationScalarRegister, 1u, 0u); + vfpu[di] = sum; + return; + } float source[4]{}; float target[4]{}; read_vfpu_vector_ct(source); @@ -879,6 +1034,41 @@ struct alignas(16) AllegrexContext { std::uint32_t TargetRegister, std::uint32_t Length> PSPRECOMP_CONTEXT_FORCEINLINE void execute_vfpu_cross_quat_ct() noexcept { static_assert(Length >= 1u && Length <= 4u); + if (vfpu_ctrl[0] == 0xE4u && vfpu_ctrl[1] == 0xE4u && vfpu_ctrl[2] == 0u) { + if constexpr (Length == 3u || Length == 4u) { + constexpr std::size_t s0i = vfpu_vector_lane_index(SourceRegister, Length, 0u); + constexpr std::size_t s1i = vfpu_vector_lane_index(SourceRegister, Length, 1u); + constexpr std::size_t s2i = vfpu_vector_lane_index(SourceRegister, Length, 2u); + constexpr std::size_t t0i = vfpu_vector_lane_index(TargetRegister, Length, 0u); + constexpr std::size_t t1i = vfpu_vector_lane_index(TargetRegister, Length, 1u); + constexpr std::size_t t2i = vfpu_vector_lane_index(TargetRegister, Length, 2u); + const float sx = vfpu[s0i], sy = vfpu[s1i], sz = vfpu[s2i]; + const float tx = vfpu[t0i], ty = vfpu[t1i], tz = vfpu[t2i]; + float r0, r1, r2, r3 = 0.0f; + if constexpr (Length == 3u) { + r0 = sy * tz - sz * ty; + r1 = sz * tx - sx * tz; + r2 = sx * ty - sy * tx; + } else { + constexpr std::size_t s3i = vfpu_vector_lane_index(SourceRegister, Length, 3u); + constexpr std::size_t t3i = vfpu_vector_lane_index(TargetRegister, Length, 3u); + const float sw = vfpu[s3i], tw = vfpu[t3i]; + r0 = sx * tw + sy * tz - sz * ty + sw * tx; + r1 = -sx * tz + sy * tw + sz * tx + sw * ty; + r2 = sx * ty - sy * tx + sz * tw + sw * tz; + r3 = -sx * tx - sy * ty - sz * tz + sw * tw; + } + constexpr std::size_t d0i = vfpu_vector_lane_index(DestinationRegister, Length, 0u); + constexpr std::size_t d1i = vfpu_vector_lane_index(DestinationRegister, Length, 1u); + constexpr std::size_t d2i = vfpu_vector_lane_index(DestinationRegister, Length, 2u); + vfpu[d0i] = r0; vfpu[d1i] = r1; vfpu[d2i] = r2; + if constexpr (Length == 4u) { + constexpr std::size_t d3i = vfpu_vector_lane_index(DestinationRegister, Length, 3u); + vfpu[d3i] = r3; + } + return; + } + } float source[4]{}; float target[4]{}; float result[4]{}; @@ -1162,11 +1352,6 @@ struct alignas(16) AllegrexContext { PSPRECOMP_CONTEXT_FORCEINLINE void execute_vfpu_vcmp_ct() noexcept { static_assert(Length >= 1u && Length <= 4u); static_assert(Condition < 16u); - float source[4]{}; - float target[4]{}; - read_vfpu_vector_with_source_prefix_ct(source); - read_vfpu_vector_with_source_prefix_ct(target); - auto compare_lane = [](float sv, float tv) -> bool { if constexpr (Condition == 0u) return false; else if constexpr (Condition == 1u) return sv == tv; @@ -1185,6 +1370,31 @@ struct alignas(16) AllegrexContext { else if constexpr (Condition == 14u) return !std::isinf(sv); else return !(std::isnan(sv) || std::isinf(sv)); }; + if (vfpu_ctrl[0] == 0xE4u && vfpu_ctrl[1] == 0xE4u && vfpu_ctrl[2] == 0u) { + constexpr std::size_t s0i = vfpu_vector_lane_index(SourceRegister, Length, 0u); + constexpr std::size_t t0i = vfpu_vector_lane_index(TargetRegister, Length, 0u); + const bool r0 = compare_lane(vfpu[s0i], vfpu[t0i]); + bool r1 = false, r2 = false, r3 = false; + if constexpr (Length >= 2u) { constexpr std::size_t si=vfpu_vector_lane_index(SourceRegister,Length,1u), ti=vfpu_vector_lane_index(TargetRegister,Length,1u); r1=compare_lane(vfpu[si],vfpu[ti]); } + if constexpr (Length >= 3u) { constexpr std::size_t si=vfpu_vector_lane_index(SourceRegister,Length,2u), ti=vfpu_vector_lane_index(TargetRegister,Length,2u); r2=compare_lane(vfpu[si],vfpu[ti]); } + if constexpr (Length >= 4u) { constexpr std::size_t si=vfpu_vector_lane_index(SourceRegister,Length,3u), ti=vfpu_vector_lane_index(TargetRegister,Length,3u); r3=compare_lane(vfpu[si],vfpu[ti]); } + std::uint32_t lane_bits = static_cast(r0); + if constexpr (Length >= 2u) lane_bits |= static_cast(r1) << 1u; + if constexpr (Length >= 3u) lane_bits |= static_cast(r2) << 2u; + if constexpr (Length >= 4u) lane_bits |= static_cast(r3) << 3u; + bool any=r0, all=r0; + if constexpr (Length >= 2u) { any|=r1; all&=r1; } + if constexpr (Length >= 3u) { any|=r2; all&=r2; } + if constexpr (Length >= 4u) { any|=r3; all&=r3; } + constexpr std::uint32_t affected=((1u<(any)<<4u)|(static_cast(all)<<5u); + vfpu_ctrl[3]=(vfpu_ctrl[3]&~affected)|(update&affected); + return; + } + float source[4]{}; + float target[4]{}; + read_vfpu_vector_with_source_prefix_ct(source); + read_vfpu_vector_with_source_prefix_ct(target); const bool r0 = compare_lane(source[0], target[0]); const bool r1 = Length >= 2u ? compare_lane(source[1], target[1]) : false; @@ -1260,6 +1470,29 @@ struct alignas(16) AllegrexContext { PSPRECOMP_CONTEXT_FORCEINLINE void execute_vfpu_vcmov_ct() noexcept { static_assert(Length >= 1u && Length <= 4u); static_assert(ConditionIndex < 8u); + if (vfpu_ctrl[0] == 0xE4u && vfpu_ctrl[1] == 0xE4u && vfpu_ctrl[2] == 0u) { + // Capture all source lanes before any destination write: VS and VD + // may overlap, and hardware observes the pre-instruction source. + constexpr std::size_t s0i=vfpu_vector_lane_index(SourceRegister,Length,0u); + const float s0=vfpu[s0i]; + float s1=0.0f,s2=0.0f,s3=0.0f; + if constexpr (Length>=2u) { constexpr std::size_t si=vfpu_vector_lane_index(SourceRegister,Length,1u); s1=vfpu[si]; } + if constexpr (Length>=3u) { constexpr std::size_t si=vfpu_vector_lane_index(SourceRegister,Length,2u); s2=vfpu[si]; } + if constexpr (Length>=4u) { constexpr std::size_t si=vfpu_vector_lane_index(SourceRegister,Length,3u); s3=vfpu[si]; } + const std::uint32_t cc=vfpu_ctrl[3]; + auto write_lane=[&](std::uint32_t lane,float value) { vfpu[vfpu_vector_lane_index(DestinationRegister,Length,lane)]=value; }; + if constexpr (ConditionIndex < 6u) { + const bool selected=(((cc>>ConditionIndex)&1u)!=0u)==!MoveIfFalse; + if (selected) { write_lane(0u,s0); if constexpr(Length>=2u) write_lane(1u,s1); if constexpr(Length>=3u) write_lane(2u,s2); if constexpr(Length>=4u) write_lane(3u,s3); } + } else if constexpr (ConditionIndex == 6u) { + constexpr bool want=!MoveIfFalse; + if ((((cc>>0u)&1u)!=0u)==want) write_lane(0u,s0); + if constexpr(Length>=2u) if ((((cc>>1u)&1u)!=0u)==want) write_lane(1u,s1); + if constexpr(Length>=3u) if ((((cc>>2u)&1u)!=0u)==want) write_lane(2u,s2); + if constexpr(Length>=4u) if ((((cc>>3u)&1u)!=0u)==want) write_lane(3u,s3); + } + return; + } float source[4]{}; float destination[4]{}; read_vfpu_vector_with_source_prefix_ct(source); @@ -1324,6 +1557,21 @@ struct alignas(16) AllegrexContext { std::uint32_t TargetScalarRegister, std::uint32_t Length> PSPRECOMP_CONTEXT_FORCEINLINE void execute_vfpu_vscl_ct() noexcept { static_assert(Length >= 1u && Length <= 4u); + if (vfpu_ctrl[0] == 0xE4u && vfpu_ctrl[1] == 0xE4u && vfpu_ctrl[2] == 0u) { + const float scale = std::bit_cast(vfpu_scalar_bits_ct<(TargetScalarRegister & 0x7Fu)>()); + constexpr std::size_t s0i = vfpu_vector_lane_index(SourceRegister, Length, 0u); + constexpr std::size_t d0i = vfpu_vector_lane_index(DestinationRegister, Length, 0u); + const float s0 = vfpu[s0i]; + float s1 = 0.0f, s2 = 0.0f, s3 = 0.0f; + if constexpr (Length >= 2u) { constexpr std::size_t si = vfpu_vector_lane_index(SourceRegister, Length, 1u); s1 = vfpu[si]; } + if constexpr (Length >= 3u) { constexpr std::size_t si = vfpu_vector_lane_index(SourceRegister, Length, 2u); s2 = vfpu[si]; } + if constexpr (Length >= 4u) { constexpr std::size_t si = vfpu_vector_lane_index(SourceRegister, Length, 3u); s3 = vfpu[si]; } + vfpu[d0i] = s0 * scale; + if constexpr (Length >= 2u) { constexpr std::size_t di = vfpu_vector_lane_index(DestinationRegister, Length, 1u); vfpu[di] = s1 * scale; } + if constexpr (Length >= 3u) { constexpr std::size_t di = vfpu_vector_lane_index(DestinationRegister, Length, 2u); vfpu[di] = s2 * scale; } + if constexpr (Length >= 4u) { constexpr std::size_t di = vfpu_vector_lane_index(DestinationRegister, Length, 3u); vfpu[di] = s3 * scale; } + return; + } float source[4]{}; read_vfpu_vector_with_source_prefix_ct(source); @@ -1547,6 +1795,116 @@ struct alignas(16) AllegrexContext { } } + // V8.6: compile-time matrix/vector transform used by VTFM/VHTFM. The + // encoded matrix/vector registers and dimensions are literals in AOT code, + // so keep address decoding and loop trip counts visible to the host + // optimizer. The default-prefix lane is intentionally very small because + // matrix transforms dominate world/geometry math in VCS. + template + PSPRECOMP_CONTEXT_FORCEINLINE void execute_vfpu_vtfm_ct() noexcept { + static_assert(Side >= 1u && Side <= 4u); + static_assert(InputLength >= 1u && InputLength <= 4u); + float matrix[16]{}; + float target_raw[4]{}; + float target[4]{}; + float result[4]{}; + read_vfpu_matrix_ct(matrix); + read_vfpu_vector_ct(target_raw); + if constexpr (InputLength >= 1u) target[0] = target_raw[0]; + if constexpr (InputLength >= 2u) target[1] = target_raw[1]; + if constexpr (InputLength >= 3u) target[2] = target_raw[2]; + if constexpr (InputLength >= 4u) target[3] = target_raw[3]; + if constexpr ((Side - 1u) >= InputLength) target[Side - 1u] = 1.0f; + + if (vfpu_ctrl[0] == 0xE4u && vfpu_ctrl[1] == 0xE4u && vfpu_ctrl[2] == 0u) { + for (std::uint32_t row = 0u; row < Side; ++row) { + float sum = 0.0f; + for (std::uint32_t column = 0u; column < Side; ++column) + sum += matrix[row * 4u + column] * target[column]; + result[row] = sum; + } + constexpr std::size_t d0 = vfpu_vector_lane_index(DestinationRegister, Side, 0u); + vfpu[d0] = result[0]; + if constexpr (Side >= 2u) { + constexpr std::size_t d1 = vfpu_vector_lane_index(DestinationRegister, Side, 1u); + vfpu[d1] = result[1]; + } + if constexpr (Side >= 3u) { + constexpr std::size_t d2 = vfpu_vector_lane_index(DestinationRegister, Side, 2u); + vfpu[d2] = result[2]; + } + if constexpr (Side >= 4u) { + constexpr std::size_t d3 = vfpu_vector_lane_index(DestinationRegister, Side, 3u); + vfpu[d3] = result[3]; + } + return; + } + + for (std::uint32_t row = 0u; row + 1u < Side; ++row) { + float sum = 0.0f; + for (std::uint32_t column = 0u; column < Side; ++column) + sum += matrix[row * 4u + column] * target[column]; + result[row] = sum; + } + float final_row[4]{ + matrix[(Side - 1u) * 4u + 0u], matrix[(Side - 1u) * 4u + 1u], + matrix[(Side - 1u) * 4u + 2u], matrix[(Side - 1u) * 4u + 3u] + }; + apply_vfpu_source_prefix_ct<4u, 0u>(final_row); + apply_vfpu_source_prefix_ct<4u, 1u>(target); + result[Side - 1u] = final_row[0] * target[0] + final_row[1] * target[1] + + final_row[2] * target[2] + final_row[3] * target[3]; + const std::uint32_t destination_prefix = vfpu_ctrl[2]; + constexpr std::uint32_t last_lane = Side - 1u; + vfpu_ctrl[2] = ((destination_prefix & (1u << 8u)) << last_lane) | + ((destination_prefix & 3u) << (last_lane * 2u)); + write_vfpu_vector_with_destination_prefix_ct(result); + } + + // V8.6: integer-bit VFPU -> float conversion with all operands known at + // translation time. Common default prefixes avoid temporary arrays and + // prefix walkers, while the fallback keeps full VPFX semantics. + template + PSPRECOMP_CONTEXT_FORCEINLINE void execute_vfpu_vi2f_ct() noexcept { + static_assert(Length >= 1u && Length <= 4u); + static_assert(Immediate < 32u); + constexpr float scale = 1.0f / static_cast(std::uint64_t{1} << Immediate); + if (vfpu_ctrl[0] == 0xE4u && vfpu_ctrl[1] == 0xE4u && vfpu_ctrl[2] == 0u) { + constexpr std::size_t s0 = vfpu_vector_lane_index(SourceRegister, Length, 0u); + constexpr std::size_t d0 = vfpu_vector_lane_index(DestinationRegister, Length, 0u); + const float r0 = static_cast(static_cast(std::bit_cast(vfpu[s0]))) * scale; + float r1 = 0.0f, r2 = 0.0f, r3 = 0.0f; + if constexpr (Length >= 2u) { + constexpr std::size_t s1 = vfpu_vector_lane_index(SourceRegister, Length, 1u); + r1 = static_cast(static_cast(std::bit_cast(vfpu[s1]))) * scale; + } + if constexpr (Length >= 3u) { + constexpr std::size_t s2 = vfpu_vector_lane_index(SourceRegister, Length, 2u); + r2 = static_cast(static_cast(std::bit_cast(vfpu[s2]))) * scale; + } + if constexpr (Length >= 4u) { + constexpr std::size_t s3 = vfpu_vector_lane_index(SourceRegister, Length, 3u); + r3 = static_cast(static_cast(std::bit_cast(vfpu[s3]))) * scale; + } + vfpu[d0] = r0; + if constexpr (Length >= 2u) { constexpr std::size_t d1 = vfpu_vector_lane_index(DestinationRegister, Length, 1u); vfpu[d1] = r1; } + if constexpr (Length >= 3u) { constexpr std::size_t d2 = vfpu_vector_lane_index(DestinationRegister, Length, 2u); vfpu[d2] = r2; } + if constexpr (Length >= 4u) { constexpr std::size_t d3 = vfpu_vector_lane_index(DestinationRegister, Length, 3u); vfpu[d3] = r3; } + return; + } + float source[4]{}, result[4]{}; + read_vfpu_vector_ct(source); + apply_vfpu_source_prefix_ct(source); + for (std::uint32_t lane = 0u; lane < Length; ++lane) { + const auto integer = static_cast(std::bit_cast(source[lane])); + result[lane] = static_cast(integer) * scale; + } + write_vfpu_vector_with_destination_prefix_ct(result); + } + void read_vfpu_matrix(float *destination, std::uint32_t matrix_register, std::uint32_t side) const noexcept { const std::uint32_t matrix = (matrix_register >> 2u) & 7u; const std::uint32_t column = matrix_register & 3u; diff --git a/profiles/vcs/generated/generated_unit_0000.cpp b/profiles/vcs/generated/generated_unit_0000.cpp index 7d9738e..d918db5 100644 --- a/profiles/vcs/generated/generated_unit_0000.cpp +++ b/profiles/vcs/generated/generated_unit_0000.cpp @@ -1266,15 +1266,8 @@ L_088043F8: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[13])); @@ -1327,15 +1320,8 @@ L_08804498: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[13])); @@ -1605,11 +1591,7 @@ L_08804734: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1638,10 +1620,7 @@ L_08804734: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); diff --git a/profiles/vcs/generated/generated_unit_0001.cpp b/profiles/vcs/generated/generated_unit_0001.cpp index a6521ae..9b84cd0 100644 --- a/profiles/vcs/generated/generated_unit_0001.cpp +++ b/profiles/vcs/generated/generated_unit_0001.cpp @@ -2951,11 +2951,7 @@ L_08809114: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0002.cpp b/profiles/vcs/generated/generated_unit_0002.cpp index d702de6..4ac3216 100644 --- a/profiles/vcs/generated/generated_unit_0002.cpp +++ b/profiles/vcs/generated/generated_unit_0002.cpp @@ -3731,10 +3731,7 @@ L_0880D624: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16025u << 16u); @@ -6113,11 +6110,7 @@ L_0880EAE0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6949,10 +6942,7 @@ L_0880F23C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -6997,11 +6987,7 @@ L_0880F294: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7109,15 +7095,8 @@ L_0880F35C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[28] + static_cast(5964), std::bit_cast(ctx.fpr[13])); @@ -7126,15 +7105,8 @@ L_0880F35C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[28] + static_cast(5968), std::bit_cast(ctx.fpr[13])); @@ -7591,15 +7563,8 @@ L_0880F688: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[10] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[10]); { const float fs = ctx.fpr[0]; const float ft = ctx.fpr[18]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[0] = std::bit_cast(0x7FC00000u); else ctx.fpr[0] = fs * ft; } @@ -7608,15 +7573,8 @@ L_0880F688: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[10] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[2] = std::bit_cast(ctx.gpr[10]); { const float fs = ctx.fpr[2]; const float ft = ctx.fpr[17]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[2] = std::bit_cast(0x7FC00000u); else ctx.fpr[2] = fs * ft; } @@ -7629,15 +7587,8 @@ L_0880F688: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[10] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[10]); { const float fs = ctx.fpr[0]; const float ft = ctx.fpr[17]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[0] = std::bit_cast(0x7FC00000u); else ctx.fpr[0] = fs * ft; } @@ -7646,15 +7597,8 @@ L_0880F688: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[10] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[2] = std::bit_cast(ctx.gpr[10]); { const float fs = ctx.fpr[2]; const float ft = ctx.fpr[18]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[19] = std::bit_cast(0x7FC00000u); else ctx.fpr[19] = fs * ft; } @@ -8173,15 +8117,8 @@ L_0880FC18: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.fpr[14] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[19] + static_cast(0))); @@ -8192,15 +8129,8 @@ L_0880FC18: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.fpr[12] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[19] + static_cast(4))); diff --git a/profiles/vcs/generated/generated_unit_0003.cpp b/profiles/vcs/generated/generated_unit_0003.cpp index cb4196f..0fb4390 100644 --- a/profiles/vcs/generated/generated_unit_0003.cpp +++ b/profiles/vcs/generated/generated_unit_0003.cpp @@ -2233,10 +2233,7 @@ L_088107EC: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[4]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[22] < ctx.fpr[20])) ? 0x00800000u : 0u); @@ -2262,10 +2259,7 @@ L_08810810: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -2303,24 +2297,10 @@ L_08810810: { const float vfpu_constant = std::bit_cast(0x3FC90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 64u, 32u, 1u, 2u>(); + ctx.execute_vfpu_vec3_ct<64u, 32u, 0u, 1u, 1u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<64u>()); ctx.fpr[22] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16457u << 16u); @@ -4054,11 +4034,7 @@ L_08811498: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(208)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4083,11 +4059,7 @@ L_08811498: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(192)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); diff --git a/profiles/vcs/generated/generated_unit_0004.cpp b/profiles/vcs/generated/generated_unit_0004.cpp index b57106a..dae3d16 100644 --- a/profiles/vcs/generated/generated_unit_0004.cpp +++ b/profiles/vcs/generated/generated_unit_0004.cpp @@ -2347,11 +2347,7 @@ L_08814ABC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -2411,11 +2407,7 @@ L_08814AEC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(160)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -2506,11 +2498,7 @@ L_08814B68: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(192)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -3702,11 +3690,7 @@ L_08815410: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(96)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3768,11 +3752,7 @@ L_08815438: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3866,11 +3846,7 @@ L_088154DC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(160)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -5571,11 +5547,7 @@ L_08816368: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[8] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[8] + static_cast(0); @@ -5632,11 +5604,7 @@ L_08816368: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5689,11 +5657,7 @@ L_08816368: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5745,11 +5709,7 @@ L_08816368: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5801,11 +5761,7 @@ L_08816368: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5857,11 +5813,7 @@ L_08816368: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5913,11 +5865,7 @@ L_08816368: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5969,11 +5917,7 @@ L_08816368: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0005.cpp b/profiles/vcs/generated/generated_unit_0005.cpp index 59cf928..a4c4ed9 100644 --- a/profiles/vcs/generated/generated_unit_0005.cpp +++ b/profiles/vcs/generated/generated_unit_0005.cpp @@ -1093,7 +1093,7 @@ L_0881822C: ctx.gpr[8] = (aot_mem.aot_direct_load16(ctx.gpr[8] + static_cast(0))); ctx.gpr[8] = (ctx.gpr[8] & 65535u); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[8]); - ctx.execute_vfpu_vh2f(0u, 0u, 1u); + ctx.execute_vfpu_vh2f_ct<0u, 0u, 1u>(); ctx.gpr[8] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[8]); ctx.gpr[8] = (16256u << 16u); @@ -1109,7 +1109,7 @@ L_0881822C: ctx.gpr[6] = (aot_mem.aot_direct_load16(ctx.gpr[6] + static_cast(0))); ctx.gpr[6] = (ctx.gpr[6] & 65535u); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[6]); - ctx.execute_vfpu_vh2f(0u, 0u, 1u); + ctx.execute_vfpu_vh2f_ct<0u, 0u, 1u>(); ctx.gpr[6] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[17] = std::bit_cast(ctx.gpr[6]); { const float fs = ctx.fpr[13]; const float ft = ctx.fpr[15]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[15] = std::bit_cast(0x7FC00000u); else ctx.fpr[15] = fs * ft; } @@ -1122,7 +1122,7 @@ L_0881822C: ctx.gpr[6] = (aot_mem.aot_direct_load16(ctx.gpr[6] + static_cast(0))); ctx.gpr[6] = (ctx.gpr[6] & 65535u); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[6]); - ctx.execute_vfpu_vh2f(0u, 0u, 1u); + ctx.execute_vfpu_vh2f_ct<0u, 0u, 1u>(); ctx.gpr[6] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[15] = std::bit_cast(ctx.gpr[6]); { const float fs = ctx.fpr[16]; const float ft = ctx.fpr[12]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[16] = std::bit_cast(0x7FC00000u); else ctx.fpr[16] = fs * ft; } @@ -1135,7 +1135,7 @@ L_0881822C: ctx.gpr[4] = (aot_mem.aot_direct_load16(ctx.gpr[4] + static_cast(0))); ctx.gpr[4] = (ctx.gpr[4] & 65535u); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); - ctx.execute_vfpu_vh2f(0u, 0u, 1u); + ctx.execute_vfpu_vh2f_ct<0u, 0u, 1u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[13]; const float ft = ctx.fpr[12]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[12] = std::bit_cast(0x7FC00000u); else ctx.fpr[12] = fs * ft; } @@ -1220,15 +1220,8 @@ L_088183F0: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[8] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[1] = std::bit_cast(ctx.gpr[8]); { const float fs = ctx.fpr[18]; const float ft = ctx.fpr[1]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[2] = std::bit_cast(0x7FC00000u); else ctx.fpr[2] = fs * ft; } @@ -1267,15 +1260,8 @@ L_0881846C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[8] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[2] = std::bit_cast(ctx.gpr[8]); { const float fs = ctx.fpr[2]; const float ft = ctx.fpr[12]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[18] = std::bit_cast(0x7FC00000u); else ctx.fpr[18] = fs * ft; } @@ -1301,15 +1287,8 @@ L_088184B8: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[6] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[6]); ctx.fpr[12] = ctx.fpr[15] + ctx.fpr[19]; @@ -1330,15 +1309,8 @@ L_08818500: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[8] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[15] = std::bit_cast(ctx.gpr[8]); aot_mem.aot_direct_store32(ctx.gpr[7] + static_cast(200), std::bit_cast(ctx.fpr[15])); @@ -1407,7 +1379,7 @@ L_0881858C: ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[6]); { float vfpu_value[4]{}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - ctx.execute_vfpu_vf2h(64u, 0u, 2u); + ctx.execute_vfpu_vf2h_ct<64u, 0u, 2u>(); ctx.gpr[6] = (ctx.vfpu_scalar_bits_ct<64u>()); ctx.gpr[6] = (ctx.gpr[6] & 65535u); aot_mem.aot_direct_store16(ctx.gpr[11] + static_cast(0), static_cast(ctx.gpr[6])); @@ -1627,7 +1599,7 @@ L_08818734: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[2] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1645,10 +1617,7 @@ L_08818734: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -3973,11 +3942,7 @@ L_088199B8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[9] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4439,11 +4404,7 @@ L_08819D00: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5725,11 +5686,7 @@ L_0881A7FC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -5748,10 +5705,7 @@ L_0881A7FC: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -6010,33 +5964,8 @@ L_0881A9FC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0006.cpp b/profiles/vcs/generated/generated_unit_0006.cpp index 33cb224..3571e01 100644 --- a/profiles/vcs/generated/generated_unit_0006.cpp +++ b/profiles/vcs/generated/generated_unit_0006.cpp @@ -1838,10 +1838,7 @@ L_0881CA7C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -2123,10 +2120,7 @@ L_0881CC90: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -2533,10 +2527,7 @@ L_0881CF94: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -3211,19 +3202,9 @@ L_0881D520: ctx.gpr[5] = (std::bit_cast(ctx.fpr[13])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[13] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[19] + static_cast(248))); @@ -3524,33 +3505,8 @@ L_0881D760: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[19] = (ctx.gpr[29] + static_cast(336)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); @@ -3601,33 +3557,8 @@ L_0881D760: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[20] = (ctx.gpr[29] + static_cast(352)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); @@ -3723,33 +3654,8 @@ L_0881D820: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[22] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3812,33 +3718,8 @@ L_0881D820: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[22] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4223,11 +4104,7 @@ L_0881DB78: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4267,11 +4144,7 @@ L_0881DB78: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[23] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6887,11 +6760,7 @@ L_0881EF30: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0007.cpp b/profiles/vcs/generated/generated_unit_0007.cpp index 06b55c6..f8453ce 100644 --- a/profiles/vcs/generated/generated_unit_0007.cpp +++ b/profiles/vcs/generated/generated_unit_0007.cpp @@ -994,11 +994,7 @@ L_08820084: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -1869,11 +1865,7 @@ L_088206A4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1918,11 +1910,7 @@ L_088206C8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1957,10 +1945,7 @@ L_088206E4: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -1984,24 +1969,10 @@ L_0882071C: { const float vfpu_constant = std::bit_cast(0x3FC90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 64u, 32u, 1u, 2u>(); + ctx.execute_vfpu_vec3_ct<64u, 32u, 0u, 1u, 1u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<64u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[4]); { const bool branch_taken = 0u == 0u; @@ -2018,24 +1989,10 @@ L_0882074C: { const float vfpu_constant = std::bit_cast(0x3FC90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 64u, 32u, 1u, 2u>(); + ctx.execute_vfpu_vec3_ct<64u, 32u, 0u, 1u, 1u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<64u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[4]); goto L_08820774; @@ -3051,28 +3008,7 @@ L_08820EF4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3097,11 +3033,7 @@ L_08820EF4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -3248,28 +3180,7 @@ L_08820FEC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -3665,28 +3576,7 @@ L_08821360: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3768,11 +3658,7 @@ L_088213C0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -3804,11 +3690,7 @@ L_0882141C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3883,11 +3765,7 @@ L_08821480: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(96)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4222,11 +4100,7 @@ L_088216BC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[19] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); @@ -4245,10 +4119,7 @@ L_088216BC: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -4280,7 +4151,7 @@ L_088216BC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[18] = (ctx.gpr[29] + static_cast(96)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); @@ -4296,10 +4167,7 @@ L_088216BC: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16147u << 16u); @@ -4330,11 +4198,7 @@ L_088216BC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4382,10 +4246,7 @@ L_088217D4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -4459,11 +4320,7 @@ L_088217EC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4486,11 +4343,7 @@ L_088217EC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4546,10 +4399,7 @@ L_088218A8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(192)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4607,11 +4457,7 @@ L_088218A8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(160)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4635,11 +4481,7 @@ L_088218A8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(144)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5665,11 +5507,7 @@ L_088221A0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -6440,10 +6278,7 @@ L_0882272C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7894,11 +7729,7 @@ L_088232B8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); diff --git a/profiles/vcs/generated/generated_unit_0008.cpp b/profiles/vcs/generated/generated_unit_0008.cpp index 589de30..df931ae 100644 --- a/profiles/vcs/generated/generated_unit_0008.cpp +++ b/profiles/vcs/generated/generated_unit_0008.cpp @@ -1183,11 +1183,7 @@ L_088243A0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1488,7 +1484,7 @@ L_08824600: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(96)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1662,15 +1658,8 @@ L_088247AC: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[16] + static_cast(208), std::bit_cast(ctx.fpr[13])); @@ -1679,15 +1668,8 @@ L_088247AC: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[16] + static_cast(212), std::bit_cast(ctx.fpr[14])); @@ -2154,11 +2136,7 @@ L_08824B94: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2228,11 +2206,7 @@ L_08824BDC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2302,11 +2276,7 @@ L_08824C24: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3038,11 +3008,7 @@ L_08825340: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -3069,7 +3035,7 @@ L_08825340: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -3094,11 +3060,7 @@ L_08825340: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3122,11 +3084,7 @@ L_08825340: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5816,15 +5774,8 @@ L_08826E18: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); @@ -5832,15 +5783,8 @@ L_08826E18: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); ctx.fpr[12] = std::bit_cast(std::bit_cast(ctx.fpr[13]) ^ 0x80000000u); @@ -5883,28 +5827,7 @@ L_08826E18: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6847,33 +6770,8 @@ L_08827628: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7345,11 +7243,7 @@ L_08827978: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7912,11 +7806,7 @@ L_08827E20: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -7946,10 +7836,7 @@ L_08827E20: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[7] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[7]); ctx.gpr[7] = (16672u << 16u); diff --git a/profiles/vcs/generated/generated_unit_0009.cpp b/profiles/vcs/generated/generated_unit_0009.cpp index ba58bcf..c731901 100644 --- a/profiles/vcs/generated/generated_unit_0009.cpp +++ b/profiles/vcs/generated/generated_unit_0009.cpp @@ -1139,7 +1139,7 @@ L_08828458: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -1176,7 +1176,7 @@ L_08828458: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -1226,7 +1226,7 @@ L_08828458: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1471,11 +1471,7 @@ L_08828714: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1490,10 +1486,7 @@ L_08828714: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (ctx.gpr[16] + static_cast(16)); @@ -1524,10 +1517,7 @@ L_08828714: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -2144,10 +2134,7 @@ L_08828F08: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[6] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[6]); { const std::uint32_t vfpu_address = ctx.gpr[23] + static_cast(0); @@ -2160,10 +2147,7 @@ L_08828F08: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[6] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[22] = std::bit_cast(ctx.gpr[6]); ctx.gpr[6] = (0u | 1u); @@ -2464,7 +2448,7 @@ L_088291A0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[19] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); @@ -2497,10 +2481,7 @@ L_088291A0: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -2620,11 +2601,7 @@ L_08829380: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2773,11 +2750,7 @@ L_088294B0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2836,10 +2809,7 @@ L_08829500: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -3041,11 +3011,7 @@ L_0882966C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(144)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3061,10 +3027,7 @@ L_0882966C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (17096u << 16u); @@ -3232,11 +3195,7 @@ L_08829800: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3251,10 +3210,7 @@ L_08829800: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16928u << 16u); @@ -3297,10 +3253,7 @@ L_08829850: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -4394,33 +4347,8 @@ L_0882A3D4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4473,33 +4401,8 @@ L_0882A3D4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4558,11 +4461,7 @@ L_0882A45C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[7] = (ctx.gpr[29] + static_cast(208)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); @@ -4658,11 +4557,7 @@ L_0882A4E4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[7] = (ctx.gpr[29] + static_cast(288)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); @@ -4889,11 +4784,7 @@ L_0882A6BC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[17] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); @@ -4917,11 +4808,7 @@ L_0882A6BC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[18] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); @@ -4945,7 +4832,7 @@ L_0882A6BC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[19] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); @@ -4978,10 +4865,7 @@ L_0882A6BC: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -5024,28 +4908,7 @@ L_0882A6BC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5081,7 +4944,7 @@ L_0882A6BC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5113,10 +4976,7 @@ L_0882A6BC: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -5142,7 +5002,7 @@ L_0882A6BC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5174,10 +5034,7 @@ L_0882A6BC: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -5771,11 +5628,7 @@ L_0882AA5C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(96)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5817,11 +5670,7 @@ L_0882AA5C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[9] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[9] + static_cast(0); @@ -6110,11 +5959,7 @@ L_0882AD78: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6696,33 +6541,8 @@ L_0882B210: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[17] = (ctx.gpr[29] + static_cast(256)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); @@ -6776,33 +6596,8 @@ L_0882B210: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[19] = (ctx.gpr[29] + static_cast(272)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); @@ -6849,11 +6644,7 @@ L_0882B288: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[7] = (ctx.gpr[29] + static_cast(144)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); @@ -7117,19 +6908,9 @@ L_0882B4BC: ctx.gpr[5] = (std::bit_cast(ctx.fpr[13])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -8195,28 +7976,7 @@ L_0882BD30: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(464)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8240,11 +8000,7 @@ L_0882BD30: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(448)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8361,28 +8117,7 @@ L_0882BDD4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(512)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8408,11 +8143,7 @@ L_0882BDD4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(496)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); diff --git a/profiles/vcs/generated/generated_unit_0010.cpp b/profiles/vcs/generated/generated_unit_0010.cpp index ae9cd4a..e1b66f1 100644 --- a/profiles/vcs/generated/generated_unit_0010.cpp +++ b/profiles/vcs/generated/generated_unit_0010.cpp @@ -3062,19 +3062,9 @@ L_0882CD78: ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[20])); @@ -3274,10 +3264,7 @@ L_0882CED8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); ctx.gpr[18] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); @@ -3390,19 +3377,9 @@ L_0882CFC4: ctx.gpr[5] = (std::bit_cast(ctx.fpr[14])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[17] = (ctx.gpr[16] + static_cast(320)); @@ -3558,10 +3535,7 @@ L_0882D134: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (15820u << 16u); @@ -3591,10 +3565,7 @@ L_0882D164: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); diff --git a/profiles/vcs/generated/generated_unit_0011.cpp b/profiles/vcs/generated/generated_unit_0011.cpp index 19cba2e..11d2569 100644 --- a/profiles/vcs/generated/generated_unit_0011.cpp +++ b/profiles/vcs/generated/generated_unit_0011.cpp @@ -1359,28 +1359,7 @@ L_08830268: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1437,33 +1416,8 @@ L_08830288: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1512,10 +1466,7 @@ L_088302CC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1545,11 +1496,7 @@ L_088302E0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1579,11 +1526,7 @@ L_088302F8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1643,10 +1586,7 @@ L_0883032C: L_08830348: ctx.gpr[6] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[6]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -1865,15 +1805,8 @@ L_08830498: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -1888,15 +1821,8 @@ L_088304BC: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -1911,19 +1837,9 @@ L_088304E0: { const float vfpu_constant = std::bit_cast(0x3FC90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 64u, 32u, 1u, 2u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -1938,24 +1854,10 @@ L_08830508: { const float vfpu_constant = std::bit_cast(0x3FC90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 64u, 32u, 1u, 2u>(); + ctx.execute_vfpu_vec3_ct<64u, 32u, 0u, 1u, 1u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<64u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -1982,10 +1884,7 @@ L_0883053C: L_08830548: ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 17u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); @@ -2003,19 +1902,9 @@ L_08830574: ctx.gpr[5] = (std::bit_cast(ctx.fpr[13])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -2093,10 +1982,7 @@ L_0883060C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -2120,10 +2006,7 @@ L_08830634: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -2880,11 +2763,7 @@ L_08830B98: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -2936,11 +2815,7 @@ L_08830B98: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2964,11 +2839,7 @@ L_08830B98: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3007,11 +2878,7 @@ L_08830C0C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -3063,11 +2930,7 @@ L_08830C0C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3091,11 +2954,7 @@ L_08830C0C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7234,11 +7093,7 @@ L_08832A4C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7362,10 +7217,7 @@ L_08832B18: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (15820u << 16u); @@ -8660,11 +8512,7 @@ L_08833554: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[30] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8703,11 +8551,7 @@ L_08833554: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[22] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0012.cpp b/profiles/vcs/generated/generated_unit_0012.cpp index ab72d83..1732481 100644 --- a/profiles/vcs/generated/generated_unit_0012.cpp +++ b/profiles/vcs/generated/generated_unit_0012.cpp @@ -1121,10 +1121,7 @@ L_088343C0: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16384u << 16u); @@ -1676,7 +1673,7 @@ L_08834830: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(208)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5268,10 +5265,7 @@ L_08836854: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (ctx.gpr[18] + static_cast(32)); @@ -5285,10 +5279,7 @@ L_08836854: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[13] <= ctx.fpr[12])) ? 0x00800000u : 0u); @@ -5623,19 +5614,9 @@ L_08836BA4: ctx.gpr[5] = (std::bit_cast(ctx.fpr[13])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[30] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[30])); @@ -5753,10 +5734,7 @@ L_08836C84: aot_mem.aot_direct_store32_block(vfpu_address, vfpu_words); } ctx.gpr[4] = (std::bit_cast(ctx.fpr[24])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -5792,20 +5770,14 @@ L_08836CD8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[22] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; aot_mem.aot_direct_store32_block(vfpu_address, vfpu_words); } ctx.gpr[4] = (std::bit_cast(ctx.fpr[24])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[22] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -5931,10 +5903,7 @@ L_08836DB4: aot_mem.aot_direct_store32_block(vfpu_address, vfpu_words); } ctx.gpr[4] = (std::bit_cast(ctx.fpr[30])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[22] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -5970,20 +5939,14 @@ L_08836E08: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[23] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; aot_mem.aot_direct_store32_block(vfpu_address, vfpu_words); } ctx.gpr[4] = (std::bit_cast(ctx.fpr[30])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[23] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -6735,10 +6698,7 @@ L_088373D4: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (ctx.gpr[21] + static_cast(48)); @@ -6819,7 +6779,7 @@ L_088374B0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[22] = (ctx.gpr[29] + static_cast(288)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[22] + static_cast(0); @@ -6843,11 +6803,7 @@ L_088374B0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[23] = (ctx.gpr[29] + static_cast(272)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[23] + static_cast(0); @@ -6882,11 +6838,7 @@ L_0883750C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[23] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7342,11 +7294,7 @@ L_08837890: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(496)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7370,11 +7318,7 @@ L_08837890: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(480)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7547,11 +7491,7 @@ L_08837998: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(544)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7588,11 +7528,7 @@ L_08837998: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7669,11 +7605,7 @@ L_08837A90: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0014.cpp b/profiles/vcs/generated/generated_unit_0014.cpp index 81640a4..ad08205 100644 --- a/profiles/vcs/generated/generated_unit_0014.cpp +++ b/profiles/vcs/generated/generated_unit_0014.cpp @@ -3444,10 +3444,7 @@ L_0883D2A8: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[12] <= ctx.fpr[22])) ? 0x00800000u : 0u); diff --git a/profiles/vcs/generated/generated_unit_0016.cpp b/profiles/vcs/generated/generated_unit_0016.cpp index e79546a..7661021 100644 --- a/profiles/vcs/generated/generated_unit_0016.cpp +++ b/profiles/vcs/generated/generated_unit_0016.cpp @@ -2405,10 +2405,7 @@ L_08844B84: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16384u << 16u); @@ -2646,10 +2643,7 @@ L_08844D5C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16384u << 16u); @@ -3434,10 +3428,7 @@ L_0884534C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16384u << 16u); @@ -4007,11 +3998,7 @@ L_088457A8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(144)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4052,11 +4039,7 @@ L_088457A8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -4127,11 +4110,7 @@ L_08845828: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(224)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4172,11 +4151,7 @@ L_08845828: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(208)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -5079,11 +5054,7 @@ L_08845F14: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5101,10 +5072,7 @@ L_08845F14: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -5265,11 +5233,7 @@ L_08846064: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5325,11 +5289,7 @@ L_088460AC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5361,10 +5321,7 @@ L_088460AC: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7608,28 +7565,7 @@ L_08847310: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<8u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 4u, 3u); - ctx.read_vfpu_vector_ct<8u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 4u, 8u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7671,28 +7607,7 @@ L_08847310: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<8u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 4u, 3u); - ctx.read_vfpu_vector_ct<8u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 4u, 8u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7788,28 +7703,7 @@ L_08847390: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8254,28 +8148,7 @@ L_08847664: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(208)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8337,11 +8210,7 @@ L_088476C0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8373,10 +8242,7 @@ L_088476C0: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.fpr[13] = ctx.fpr[12] + ctx.fpr[24]; @@ -8415,11 +8281,7 @@ L_088476C0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9189,11 +9051,7 @@ L_08847C68: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -9406,28 +9264,7 @@ L_08847EA0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[18] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); @@ -9452,15 +9289,8 @@ L_08847EA0: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); @@ -9468,15 +9298,8 @@ L_08847EA0: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); ctx.fpr[12] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[29] + static_cast(64))); @@ -9546,33 +9369,8 @@ L_08847F5C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[22] = (ctx.gpr[29] + static_cast(176)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[22] + static_cast(0); @@ -9618,11 +9416,7 @@ L_08847F5C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(836), ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(96)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -9653,11 +9447,7 @@ L_08847F5C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0017.cpp b/profiles/vcs/generated/generated_unit_0017.cpp index 0ef83a6..bd0d3c3 100644 --- a/profiles/vcs/generated/generated_unit_0017.cpp +++ b/profiles/vcs/generated/generated_unit_0017.cpp @@ -860,11 +860,7 @@ L_08848000: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[23] = (ctx.gpr[29] + static_cast(192)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[23] + static_cast(0); @@ -955,11 +951,7 @@ L_088480A4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(336)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1057,11 +1049,7 @@ L_0884816C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1077,10 +1065,7 @@ L_0884816C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16000u << 16u); @@ -1123,11 +1108,7 @@ L_088481DC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1169,11 +1150,7 @@ L_088481DC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1222,11 +1199,7 @@ L_08848264: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[19] = (ctx.gpr[29] + static_cast(256)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); @@ -1245,10 +1218,7 @@ L_08848264: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -1385,10 +1355,7 @@ L_08848368: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (15779u << 16u); @@ -1987,11 +1954,7 @@ L_08848858: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2006,10 +1969,7 @@ L_08848858: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[12] < ctx.fpr[24])) ? 0x00800000u : 0u); @@ -2132,11 +2092,7 @@ L_08848934: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3053,15 +3009,8 @@ L_088490B4: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); @@ -3069,15 +3018,8 @@ L_088490B4: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); ctx.fpr[12] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[29] + static_cast(52))); @@ -3115,7 +3057,7 @@ L_088490B4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3159,11 +3101,7 @@ L_088490B4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -3637,10 +3575,7 @@ L_088494E0: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[26] = std::bit_cast(ctx.gpr[4]); ctx.gpr[31] = (0x08849500u); @@ -3659,10 +3594,7 @@ L_08849500: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[26] <= ctx.fpr[12])) ? 0x00800000u : 0u); @@ -3691,10 +3623,7 @@ L_08849530: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (15820u << 16u); @@ -5891,11 +5820,7 @@ L_0884A530: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6292,15 +6217,8 @@ L_0884A8E0: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(600), std::bit_cast(ctx.fpr[12])); @@ -6309,15 +6227,8 @@ L_0884A8E0: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(596), std::bit_cast(ctx.fpr[12])); @@ -6605,10 +6516,7 @@ L_0884AB14: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -6639,7 +6547,7 @@ L_0884AB14: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[30] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6670,10 +6578,7 @@ L_0884AB14: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -6744,11 +6649,7 @@ L_0884AB94: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(144)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -6818,11 +6719,7 @@ L_0884AB94: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7235,10 +7132,7 @@ L_0884AF24: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7269,7 +7163,7 @@ L_0884AF24: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[30] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7300,10 +7194,7 @@ L_0884AF24: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7374,11 +7265,7 @@ L_0884AFA4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(240)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -7448,11 +7335,7 @@ L_0884AFA4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7827,19 +7710,9 @@ L_0884B2EC: ctx.gpr[5] = (std::bit_cast(ctx.fpr[14])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); ctx.fpr[15] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[18] + static_cast(1712))); @@ -8376,11 +8249,7 @@ L_0884B7B0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[23] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8781,11 +8650,7 @@ L_0884BAB4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8819,10 +8684,7 @@ L_0884BB28: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (15692u << 16u); @@ -8911,28 +8773,7 @@ L_0884BB68: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[18] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); @@ -8970,11 +8811,7 @@ L_0884BB68: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[16] = (ctx.gpr[29] + static_cast(160)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); @@ -9003,10 +8840,7 @@ L_0884BC14: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); ctx.gpr[17] = (ctx.gpr[29] + static_cast(192)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); diff --git a/profiles/vcs/generated/generated_unit_0018.cpp b/profiles/vcs/generated/generated_unit_0018.cpp index dac555a..e8ad4dc 100644 --- a/profiles/vcs/generated/generated_unit_0018.cpp +++ b/profiles/vcs/generated/generated_unit_0018.cpp @@ -3165,11 +3165,7 @@ L_0884D18C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 4u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 4u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 4u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3200,11 +3196,7 @@ L_0884D1A4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 4u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 4u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 4u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3280,10 +3272,7 @@ L_0884D210: } goto L_0884D228; L_0884D228: - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<4u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<4u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<4u, 4u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<3u, 0u, 4u, 4u>(); goto L_0884D230; L_0884D230: @@ -3619,11 +3608,7 @@ L_0884D4C4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 4u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 4u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 4u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3653,7 +3638,7 @@ L_0884D4F0: goto L_0884D500; } L_0884D500: - ctx.execute_vfpu_matrix_init(0u, 4u, 3u); + ctx.execute_vfpu_matrix_init_ct<0u, 4u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3690,10 +3675,7 @@ L_0884D500: } goto L_0884D52C; L_0884D52C: - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<4u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<4u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<4u, 4u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<3u, 0u, 4u, 4u>(); goto L_0884D534; L_0884D534: @@ -4377,11 +4359,7 @@ L_0884DA74: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 4u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 4u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 4u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4412,11 +4390,7 @@ L_0884DA8C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 4u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 4u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 4u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4568,10 +4542,7 @@ L_0884DBB4: } goto L_0884DBCC; L_0884DBCC: - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<4u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<4u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<4u, 4u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<3u, 0u, 4u, 4u>(); goto L_0884DBD4; L_0884DBD4: @@ -4930,11 +4901,7 @@ L_0884DF44: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 4u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 4u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 4u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5049,10 +5016,7 @@ L_0884E00C: } goto L_0884E024; L_0884E024: - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<4u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<4u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<4u, 4u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<3u, 0u, 4u, 4u>(); goto L_0884E02C; L_0884E02C: @@ -5103,11 +5067,7 @@ L_0884E064: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -5428,11 +5388,7 @@ L_0884E348: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 4u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 4u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 4u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5562,7 +5518,7 @@ L_0884E464: goto L_0884E474; } L_0884E474: - ctx.execute_vfpu_matrix_init(0u, 4u, 3u); + ctx.execute_vfpu_matrix_init_ct<0u, 4u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5599,10 +5555,7 @@ L_0884E474: } goto L_0884E4A0; L_0884E4A0: - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<4u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<4u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<4u, 4u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<3u, 0u, 4u, 4u>(); goto L_0884E4A8; L_0884E4A8: @@ -5896,11 +5849,7 @@ L_0884E760: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 4u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 4u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 4u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5995,7 +5944,7 @@ L_0884E814: goto L_0884E828; } L_0884E828: - ctx.execute_vfpu_matrix_init(0u, 4u, 3u); + ctx.execute_vfpu_matrix_init_ct<0u, 4u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[30] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6032,10 +5981,7 @@ L_0884E828: } goto L_0884E854; L_0884E854: - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<4u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<4u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<4u, 4u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<3u, 0u, 4u, 4u>(); goto L_0884E85C; L_0884E85C: @@ -6079,11 +6025,7 @@ L_0884E87C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); diff --git a/profiles/vcs/generated/generated_unit_0022.cpp b/profiles/vcs/generated/generated_unit_0022.cpp index 34c93ca..2a81d19 100644 --- a/profiles/vcs/generated/generated_unit_0022.cpp +++ b/profiles/vcs/generated/generated_unit_0022.cpp @@ -8804,15 +8804,8 @@ L_0885F854: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); @@ -8820,15 +8813,8 @@ L_0885F854: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (16256u << 16u); @@ -8855,15 +8841,8 @@ L_0885F8C8: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); @@ -8871,15 +8850,8 @@ L_0885F8C8: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[5]); aot_mem.aot_direct_store32(ctx.gpr[4] + static_cast(0), std::bit_cast(ctx.fpr[13])); @@ -8906,15 +8878,8 @@ L_0885F93C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); @@ -8922,15 +8887,8 @@ L_0885F93C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[5]); aot_mem.aot_direct_store32(ctx.gpr[4] + static_cast(0), std::bit_cast(ctx.fpr[13])); @@ -9008,15 +8966,8 @@ L_0885FA28: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[15] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); @@ -9024,15 +8975,8 @@ L_0885FA28: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[16] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (std::bit_cast(ctx.fpr[13])); @@ -9040,15 +8984,8 @@ L_0885FA28: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (std::bit_cast(ctx.fpr[13])); @@ -9056,15 +8993,8 @@ L_0885FA28: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[17] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (std::bit_cast(ctx.fpr[14])); @@ -9072,15 +9002,8 @@ L_0885FA28: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (std::bit_cast(ctx.fpr[14])); @@ -9088,15 +9011,8 @@ L_0885FA28: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[18] = std::bit_cast(ctx.gpr[5]); { const float fs = ctx.fpr[16]; const float ft = ctx.fpr[18]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[14] = std::bit_cast(0x7FC00000u); else ctx.fpr[14] = fs * ft; } @@ -9143,15 +9059,8 @@ L_0885FB5C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); @@ -9159,15 +9068,8 @@ L_0885FB5C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[5]); ctx.fpr[12] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[4] + static_cast(4))); @@ -9222,15 +9124,8 @@ L_0885FC3C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); @@ -9238,15 +9133,8 @@ L_0885FC3C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[5]); ctx.fpr[12] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[4] + static_cast(0))); @@ -9301,15 +9189,8 @@ L_0885FD1C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); @@ -9317,15 +9198,8 @@ L_0885FD1C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[5]); ctx.fpr[12] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[4] + static_cast(0))); @@ -9529,32 +9403,8 @@ L_0885FE50: ctx.write_vfpu_vector_ct<34u, 4u>(vfpu_value); } { float vfpu_value[4]{}; ctx.write_vfpu_vector_with_destination_prefix_ct<35u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<32u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<4u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<3u, 3u>(vfpu_result); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<3u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<3u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<3u, 32u, 4u, 3u, 3u>(); + ctx.execute_vfpu_unary_ct<3u, 3u, 3u, 2u>(); aot_mem.aot_direct_store32(ctx.gpr[4] + static_cast(64), ctx.vfpu_scalar_bits_ct<99u>()); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -9617,32 +9467,8 @@ L_0885FE8C: ctx.gpr[2] = (ctx.gpr[5] | 0u); { float vfpu_value[4]{}; ctx.write_vfpu_vector_with_destination_prefix_ct<35u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<32u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<4u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<3u, 3u>(vfpu_result); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<3u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<3u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<3u, 32u, 4u, 3u, 3u>(); + ctx.execute_vfpu_unary_ct<3u, 3u, 3u, 2u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0023.cpp b/profiles/vcs/generated/generated_unit_0023.cpp index 4511640..d13209a 100644 --- a/profiles/vcs/generated/generated_unit_0023.cpp +++ b/profiles/vcs/generated/generated_unit_0023.cpp @@ -1098,11 +1098,7 @@ L_08860094: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<3u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<3u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<8u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<3u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<3u, 3u, 8u, 3u, 0u>(); jump_target = ctx.gpr[31]; { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<3u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(48); @@ -1118,15 +1114,8 @@ L_088600C4: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[15] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); @@ -1134,15 +1123,8 @@ L_088600C4: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[16] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (std::bit_cast(ctx.fpr[13])); @@ -1150,15 +1132,8 @@ L_088600C4: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (std::bit_cast(ctx.fpr[13])); @@ -1166,15 +1141,8 @@ L_088600C4: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[17] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (std::bit_cast(ctx.fpr[14])); @@ -1182,15 +1150,8 @@ L_088600C4: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (std::bit_cast(ctx.fpr[14])); @@ -1198,15 +1159,8 @@ L_088600C4: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[18] = std::bit_cast(ctx.gpr[5]); { const float fs = ctx.fpr[16]; const float ft = ctx.fpr[18]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[14] = std::bit_cast(0x7FC00000u); else ctx.fpr[14] = fs * ft; } @@ -1339,10 +1293,7 @@ L_08860314: ctx.execute_vfpu_vdot_ct<8u, 0u, 0u, 3u>(); ctx.execute_vfpu_vdot_ct<40u, 1u, 1u, 3u>(); ctx.execute_vfpu_vdot_ct<72u, 2u, 2u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<8u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<9u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<9u, 8u, 3u, 17u>(); ctx.execute_vfpu_vscl_ct<0u, 0u, 9u, 3u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 41u, 3u>(); ctx.execute_vfpu_vscl_ct<2u, 2u, 73u, 3u>(); @@ -4725,10 +4676,7 @@ L_08861D78: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (15395u << 16u); @@ -4755,10 +4703,7 @@ L_08861DAC: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (15523u << 16u); @@ -5230,11 +5175,7 @@ L_08862148: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5262,10 +5203,7 @@ L_08862148: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.fpr[13] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[18] + static_cast(36))); @@ -7740,11 +7678,7 @@ L_088632CC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); diff --git a/profiles/vcs/generated/generated_unit_0024.cpp b/profiles/vcs/generated/generated_unit_0024.cpp index be11b72..25f7d35 100644 --- a/profiles/vcs/generated/generated_unit_0024.cpp +++ b/profiles/vcs/generated/generated_unit_0024.cpp @@ -2221,11 +2221,7 @@ L_08864A14: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[8] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2253,10 +2249,7 @@ L_08864A14: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[10] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[10]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[28] <= ctx.fpr[12])) ? 0x00800000u : 0u); diff --git a/profiles/vcs/generated/generated_unit_0025.cpp b/profiles/vcs/generated/generated_unit_0025.cpp index d6f96ce..1fed9c3 100644 --- a/profiles/vcs/generated/generated_unit_0025.cpp +++ b/profiles/vcs/generated/generated_unit_0025.cpp @@ -6459,11 +6459,7 @@ L_0886AAB0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6746,15 +6742,8 @@ L_0886ACC0: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (15692u << 16u); @@ -7072,28 +7061,7 @@ L_0886AF6C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<8u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 4u, 3u); - ctx.read_vfpu_vector_ct<8u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 4u, 8u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[30] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7132,11 +7100,7 @@ L_0886AF6C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[22] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7159,11 +7123,7 @@ L_0886AF6C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7179,10 +7139,7 @@ L_0886AF6C: L_0886AFD0: ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); diff --git a/profiles/vcs/generated/generated_unit_0026.cpp b/profiles/vcs/generated/generated_unit_0026.cpp index 663556d..af1f6b9 100644 --- a/profiles/vcs/generated/generated_unit_0026.cpp +++ b/profiles/vcs/generated/generated_unit_0026.cpp @@ -1542,10 +1542,7 @@ L_0886C6E4: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[22] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (static_cast(static_cast(static_cast(aot_mem.aot_direct_load16(ctx.gpr[19] + static_cast(86)))))); @@ -1881,11 +1878,7 @@ L_0886C8F4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[7] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); @@ -1909,11 +1902,7 @@ L_0886C8F4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1953,11 +1942,7 @@ L_0886C8F4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1980,11 +1965,7 @@ L_0886C8F4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2419,10 +2400,7 @@ L_0886CD4C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); ctx.gpr[7] = (ctx.gpr[29] + static_cast(320)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); diff --git a/profiles/vcs/generated/generated_unit_0027.cpp b/profiles/vcs/generated/generated_unit_0027.cpp index 625913e..3d150af 100644 --- a/profiles/vcs/generated/generated_unit_0027.cpp +++ b/profiles/vcs/generated/generated_unit_0027.cpp @@ -1227,10 +1227,7 @@ L_08870130: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4041,11 +4038,7 @@ L_0887156C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4060,10 +4053,7 @@ L_0887156C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); { const bool branch_taken = 0u == 0u; @@ -6640,19 +6630,9 @@ L_08872B8C: ctx.gpr[5] = (std::bit_cast(ctx.fpr[13])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -6854,11 +6834,7 @@ L_08872D0C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6992,10 +6968,7 @@ L_08872E30: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7536,11 +7509,7 @@ L_0887322C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0030.cpp b/profiles/vcs/generated/generated_unit_0030.cpp index 9584338..4958548 100644 --- a/profiles/vcs/generated/generated_unit_0030.cpp +++ b/profiles/vcs/generated/generated_unit_0030.cpp @@ -3068,33 +3068,8 @@ L_0887CED0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3116,10 +3091,7 @@ L_0887CEF8: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[6] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[6]); aot_mem.aot_direct_store32(ctx.gpr[5] + static_cast(0), std::bit_cast(ctx.fpr[12])); @@ -3546,28 +3518,7 @@ L_0887D190: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -3591,11 +3542,7 @@ L_0887D190: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3760,11 +3707,7 @@ L_0887D2D8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3807,10 +3750,7 @@ L_0887D2D8: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); aot_mem.aot_direct_store32(ctx.gpr[6] + static_cast(68), std::bit_cast(ctx.fpr[12])); @@ -3917,11 +3857,7 @@ L_0887D388: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4673,11 +4609,7 @@ L_0887D964: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -4715,11 +4647,7 @@ L_0887D964: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(160)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -4736,10 +4664,7 @@ L_0887D964: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[5]); ctx.fpr[24] = ctx.fpr[28] + ctx.fpr[24]; @@ -4775,19 +4700,9 @@ L_0887DA00: { const float vfpu_constant = std::bit_cast(0x3FC90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 64u, 32u, 1u, 2u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[28] = std::bit_cast(ctx.gpr[5]); ctx.fpr[12] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[17] + static_cast(288))); @@ -4796,24 +4711,10 @@ L_0887DA00: { const float vfpu_constant = std::bit_cast(0x3FC90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 64u, 32u, 1u, 2u>(); + ctx.execute_vfpu_vec3_ct<64u, 32u, 0u, 1u, 1u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<64u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (std::bit_cast(ctx.fpr[28])); @@ -4821,17 +4722,9 @@ L_0887DA00: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); ctx.execute_vfpu_vrot_ct<1u, 64u, 2u, 4u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<33u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] / vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 33u, 1u, 1u, 3u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[15] = std::bit_cast(ctx.gpr[5]); { const float fs = ctx.fpr[15]; const float ft = ctx.fpr[14]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[14] = std::bit_cast(0x7FC00000u); else ctx.fpr[14] = fs * ft; } @@ -4984,11 +4877,7 @@ L_0887DAE0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5046,17 +4935,9 @@ L_0887DB94: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); ctx.execute_vfpu_vrot_ct<1u, 64u, 2u, 4u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<33u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] / vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 33u, 1u, 1u, 3u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[12]; const float ft = ctx.fpr[24]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[12] = std::bit_cast(0x7FC00000u); else ctx.fpr[12] = fs * ft; } @@ -5111,11 +4992,7 @@ L_0887DB94: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[22] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5536,11 +5413,7 @@ L_0887DF00: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[30] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5619,10 +5492,7 @@ L_0887DF54: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -5767,24 +5637,10 @@ L_0887E04C: { const float vfpu_constant = std::bit_cast(0x3FC90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 64u, 32u, 1u, 2u>(); + ctx.execute_vfpu_vec3_ct<64u, 32u, 0u, 1u, 1u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<64u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); @@ -5792,17 +5648,9 @@ L_0887E04C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); ctx.execute_vfpu_vrot_ct<1u, 64u, 2u, 4u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<33u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] / vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 33u, 1u, 1u, 3u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[20]; const float ft = ctx.fpr[12]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[12] = std::bit_cast(0x7FC00000u); else ctx.fpr[12] = fs * ft; } @@ -5848,11 +5696,7 @@ L_0887E04C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(288)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -5893,11 +5737,7 @@ L_0887E04C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(224)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5955,11 +5795,7 @@ L_0887E04C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(208)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); diff --git a/profiles/vcs/generated/generated_unit_0031.cpp b/profiles/vcs/generated/generated_unit_0031.cpp index 7009320..8b5011f 100644 --- a/profiles/vcs/generated/generated_unit_0031.cpp +++ b/profiles/vcs/generated/generated_unit_0031.cpp @@ -1195,11 +1195,7 @@ L_088801A4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<2u, 0u, 1u, 3u, 2u>(); ctx.gpr[7] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<2u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); @@ -1236,11 +1232,7 @@ L_088801A4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1317,15 +1309,8 @@ L_08880298: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -1340,15 +1325,8 @@ L_088802BC: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -1433,11 +1411,7 @@ L_08880344: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[7] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); @@ -1470,10 +1444,7 @@ L_08880344: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -1504,7 +1475,7 @@ L_08880344: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1540,7 +1511,7 @@ L_08880344: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[10] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1739,28 +1710,7 @@ L_08880500: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(160)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -1790,10 +1740,7 @@ L_08880500: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3597,10 +3544,7 @@ L_088812F8: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -3673,7 +3617,7 @@ L_0888134C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(224)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3923,28 +3867,7 @@ L_08881560: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -3974,10 +3897,7 @@ L_08881560: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5264,11 +5184,7 @@ L_08882188: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5781,11 +5697,7 @@ L_08882598: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6485,11 +6397,7 @@ L_08882A50: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6537,11 +6445,7 @@ L_08882A7C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6708,10 +6612,7 @@ L_08882B68: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16880u << 16u); @@ -7357,10 +7258,7 @@ L_08882F90: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[20]; const float ft = ctx.fpr[22]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[13] = std::bit_cast(0x7FC00000u); else ctx.fpr[13] = fs * ft; } @@ -7540,10 +7438,7 @@ L_088830AC: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[12] < ctx.fpr[20])) ? 0x00800000u : 0u); @@ -7814,11 +7709,7 @@ L_08883278: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -9271,11 +9162,7 @@ L_08883E54: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0033.cpp b/profiles/vcs/generated/generated_unit_0033.cpp index 3eeb285..60596b6 100644 --- a/profiles/vcs/generated/generated_unit_0033.cpp +++ b/profiles/vcs/generated/generated_unit_0033.cpp @@ -4236,11 +4236,7 @@ L_0888926C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4637,11 +4633,7 @@ L_088894B8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6032,11 +6024,7 @@ L_08889DEC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6051,10 +6039,7 @@ L_08889DEC: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[8] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[8]); ctx.gpr[8] = (aot_mem.aot_direct_load32(ctx.gpr[16] + static_cast(1528))); @@ -6311,11 +6296,7 @@ L_08889FD4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6331,10 +6312,7 @@ L_08889FD4: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (aot_mem.aot_direct_load32(ctx.gpr[17] + static_cast(1488))); diff --git a/profiles/vcs/generated/generated_unit_0034.cpp b/profiles/vcs/generated/generated_unit_0034.cpp index 99378be..cc572e5 100644 --- a/profiles/vcs/generated/generated_unit_0034.cpp +++ b/profiles/vcs/generated/generated_unit_0034.cpp @@ -1941,11 +1941,7 @@ L_0888C784: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2812,15 +2808,8 @@ L_0888CD6C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (14979u << 16u); @@ -3078,15 +3067,8 @@ L_0888D0B0: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); @@ -3094,15 +3076,8 @@ L_0888D0B0: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[22] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (0u | 6u); @@ -4560,7 +4535,7 @@ L_0888DF50: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4578,10 +4553,7 @@ L_0888DF50: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -4616,24 +4588,10 @@ L_0888E034: { const float vfpu_constant = std::bit_cast(0x3FC90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 64u, 32u, 1u, 2u>(); + ctx.execute_vfpu_vec3_ct<64u, 32u, 0u, 1u, 1u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<64u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[28] + static_cast(8388), std::bit_cast(ctx.fpr[12])); @@ -4659,17 +4617,9 @@ L_0888E07C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - ctx.execute_vfpu_vrot(1u, 64u, 2u, 4u); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<33u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] / vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_vrot_ct<1u, 64u, 2u, 4u>(); + ctx.execute_vfpu_vec3_ct<0u, 33u, 1u, 1u, 3u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (aot_mem.aot_direct_load32(ctx.gpr[28] + static_cast(-8744))); @@ -5917,11 +5867,7 @@ L_0888EBAC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5979,11 +5925,7 @@ L_0888EC30: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6080,11 +6022,7 @@ L_0888EC90: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6150,11 +6088,7 @@ L_0888ECF0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6222,11 +6156,7 @@ L_0888ED30: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0035.cpp b/profiles/vcs/generated/generated_unit_0035.cpp index 453a4de..05a6fa4 100644 --- a/profiles/vcs/generated/generated_unit_0035.cpp +++ b/profiles/vcs/generated/generated_unit_0035.cpp @@ -2652,27 +2652,12 @@ L_088910D0: aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(384), std::bit_cast(ctx.fpr[12])); ctx.gpr[4] = (ctx.gpr[29] + static_cast(384)); ctx.set_vfpu_scalar_bits_ct<0u>(aot_mem.aot_direct_load32(ctx.gpr[4] + static_cast(0))); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 0u, 1u, 2u>(); ctx.vfpu_ctrl[1u] = 0x000010E5u; - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] / vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<32u, 0u, 0u, 1u, 0u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 32u, 1u, 3u>(); ctx.execute_vfpu_vocp_ct<32u, 0u, 1u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 2u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 2u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 2u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<1u, 0u, 2u, 22u>(); aot_mem.aot_direct_store32(ctx.gpr[21] + static_cast(0), ctx.vfpu_scalar_bits_ct<1u>()); aot_mem.aot_direct_store32(ctx.gpr[20] + static_cast(0), ctx.vfpu_scalar_bits_ct<33u>()); ctx.fpr[12] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[16] + static_cast(100))); @@ -2681,27 +2666,12 @@ L_088910D0: aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(388), std::bit_cast(ctx.fpr[12])); ctx.gpr[4] = (ctx.gpr[29] + static_cast(388)); ctx.set_vfpu_scalar_bits_ct<0u>(aot_mem.aot_direct_load32(ctx.gpr[4] + static_cast(0))); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 0u, 1u, 2u>(); ctx.vfpu_ctrl[1u] = 0x000010E5u; - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] / vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<32u, 0u, 0u, 1u, 0u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 32u, 1u, 3u>(); ctx.execute_vfpu_vocp_ct<32u, 0u, 1u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 2u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 2u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 2u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<1u, 0u, 2u, 22u>(); aot_mem.aot_direct_store32(ctx.gpr[19] + static_cast(0), ctx.vfpu_scalar_bits_ct<1u>()); aot_mem.aot_direct_store32(ctx.gpr[18] + static_cast(0), ctx.vfpu_scalar_bits_ct<33u>()); ctx.gpr[5] = (2233u << 16u); @@ -2763,27 +2733,12 @@ L_088911CC: aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(392), std::bit_cast(ctx.fpr[12])); ctx.gpr[4] = (ctx.gpr[29] + static_cast(392)); ctx.set_vfpu_scalar_bits_ct<0u>(aot_mem.aot_direct_load32(ctx.gpr[4] + static_cast(0))); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 0u, 1u, 2u>(); ctx.vfpu_ctrl[1u] = 0x000010E5u; - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] / vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<32u, 0u, 0u, 1u, 0u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 32u, 1u, 3u>(); ctx.execute_vfpu_vocp_ct<32u, 0u, 1u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 2u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 2u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 2u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<1u, 0u, 2u, 22u>(); aot_mem.aot_direct_store32(ctx.gpr[21] + static_cast(0), ctx.vfpu_scalar_bits_ct<1u>()); aot_mem.aot_direct_store32(ctx.gpr[20] + static_cast(0), ctx.vfpu_scalar_bits_ct<33u>()); ctx.fpr[12] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[16] + static_cast(100))); @@ -2796,27 +2751,12 @@ L_088911CC: aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(396), std::bit_cast(ctx.fpr[12])); ctx.gpr[4] = (ctx.gpr[29] + static_cast(396)); ctx.set_vfpu_scalar_bits_ct<0u>(aot_mem.aot_direct_load32(ctx.gpr[4] + static_cast(0))); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 0u, 1u, 2u>(); ctx.vfpu_ctrl[1u] = 0x000010E5u; - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] / vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<32u, 0u, 0u, 1u, 0u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 32u, 1u, 3u>(); ctx.execute_vfpu_vocp_ct<32u, 0u, 1u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 2u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 2u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 2u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<1u, 0u, 2u, 22u>(); aot_mem.aot_direct_store32(ctx.gpr[19] + static_cast(0), ctx.vfpu_scalar_bits_ct<1u>()); aot_mem.aot_direct_store32(ctx.gpr[18] + static_cast(0), ctx.vfpu_scalar_bits_ct<33u>()); ctx.gpr[23] = (ctx.gpr[29] + static_cast(256)); @@ -2852,27 +2792,12 @@ L_08891280: aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(400), std::bit_cast(ctx.fpr[12])); ctx.gpr[4] = (ctx.gpr[29] + static_cast(400)); ctx.set_vfpu_scalar_bits_ct<0u>(aot_mem.aot_direct_load32(ctx.gpr[4] + static_cast(0))); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 0u, 1u, 2u>(); ctx.vfpu_ctrl[1u] = 0x000010E5u; - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] / vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<32u, 0u, 0u, 1u, 0u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 32u, 1u, 3u>(); ctx.execute_vfpu_vocp_ct<32u, 0u, 1u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 2u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 2u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 2u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<1u, 0u, 2u, 22u>(); aot_mem.aot_direct_store32(ctx.gpr[21] + static_cast(0), ctx.vfpu_scalar_bits_ct<1u>()); aot_mem.aot_direct_store32(ctx.gpr[20] + static_cast(0), ctx.vfpu_scalar_bits_ct<33u>()); ctx.fpr[12] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[16] + static_cast(100))); @@ -2885,27 +2810,12 @@ L_08891280: aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(404), std::bit_cast(ctx.fpr[12])); ctx.gpr[4] = (ctx.gpr[29] + static_cast(404)); ctx.set_vfpu_scalar_bits_ct<0u>(aot_mem.aot_direct_load32(ctx.gpr[4] + static_cast(0))); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 0u, 1u, 2u>(); ctx.vfpu_ctrl[1u] = 0x000010E5u; - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] / vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<32u, 0u, 0u, 1u, 0u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 32u, 1u, 3u>(); ctx.execute_vfpu_vocp_ct<32u, 0u, 1u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 2u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 2u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 2u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<1u, 0u, 2u, 22u>(); aot_mem.aot_direct_store32(ctx.gpr[19] + static_cast(0), ctx.vfpu_scalar_bits_ct<1u>()); aot_mem.aot_direct_store32(ctx.gpr[18] + static_cast(0), ctx.vfpu_scalar_bits_ct<33u>()); ctx.gpr[18] = (ctx.gpr[29] + static_cast(320)); @@ -3468,33 +3378,8 @@ L_08891710: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3766,19 +3651,9 @@ L_08891948: ctx.gpr[5] = (std::bit_cast(ctx.fpr[13])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -3865,33 +3740,8 @@ L_088919A0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4737,11 +4587,7 @@ L_088920C0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4757,10 +4603,7 @@ L_088920C0: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.fpr[12] = std::bit_cast(ctx.fpu_float_to_word_ct<3u>(ctx.fpr[12])); @@ -4817,10 +4660,7 @@ L_08892124: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -4898,11 +4738,7 @@ L_0889217C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[30] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5098,11 +4934,7 @@ L_088922FC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5516,11 +5348,7 @@ L_088926D4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[10] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5800,11 +5628,7 @@ L_088928AC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5994,11 +5818,7 @@ L_088929D0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[11] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6164,11 +5984,7 @@ L_08892AAC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[10] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6218,10 +6034,7 @@ L_08892AF8: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[2] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[2]); ctx.gpr[2] = (static_cast(static_cast(static_cast(aot_mem.aot_direct_load16(ctx.gpr[11] + static_cast(86)))))); @@ -6543,11 +6356,7 @@ L_08892D78: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6588,28 +6397,7 @@ L_08892D78: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<8u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<4u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<8u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } + ctx.execute_vfpu_vtfm_ct<0u, 4u, 8u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0036.cpp b/profiles/vcs/generated/generated_unit_0036.cpp index 069c68d..24b6ad4 100644 --- a/profiles/vcs/generated/generated_unit_0036.cpp +++ b/profiles/vcs/generated/generated_unit_0036.cpp @@ -1134,30 +1134,10 @@ L_08894150: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -1189,11 +1169,7 @@ L_08894150: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1389,11 +1365,7 @@ L_08894354: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1530,30 +1502,10 @@ L_08894418: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -1586,11 +1538,7 @@ L_08894418: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0037.cpp b/profiles/vcs/generated/generated_unit_0037.cpp index 2a844a2..0db0cc1 100644 --- a/profiles/vcs/generated/generated_unit_0037.cpp +++ b/profiles/vcs/generated/generated_unit_0037.cpp @@ -6945,11 +6945,7 @@ L_0889B514: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(160)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6979,10 +6975,7 @@ L_0889B514: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[22] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (static_cast(static_cast(static_cast(aot_mem.aot_direct_load16(ctx.gpr[17] + static_cast(86)))))); diff --git a/profiles/vcs/generated/generated_unit_0038.cpp b/profiles/vcs/generated/generated_unit_0038.cpp index 66dfe8a..3390ee7 100644 --- a/profiles/vcs/generated/generated_unit_0038.cpp +++ b/profiles/vcs/generated/generated_unit_0038.cpp @@ -1510,28 +1510,7 @@ L_0889C2F8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[23] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1703,28 +1682,7 @@ L_0889C43C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[22] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1795,11 +1753,7 @@ L_0889C4F4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1851,11 +1805,7 @@ L_0889C4F4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2072,11 +2022,7 @@ L_0889C6E8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[23] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2091,10 +2037,7 @@ L_0889C6E8: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[24] = std::bit_cast(ctx.gpr[4]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[24] < ctx.fpr[26])) ? 0x00800000u : 0u); @@ -7118,11 +7061,7 @@ L_0889EE44: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[9] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[9] + static_cast(0); @@ -7138,10 +7077,7 @@ L_0889EE44: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[6] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[6]); ctx.gpr[6] = (16256u << 16u); @@ -7239,11 +7175,7 @@ L_0889EF04: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -7355,11 +7287,7 @@ L_0889EFB0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -7491,11 +7419,7 @@ L_0889F06C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7539,11 +7463,7 @@ L_0889F06C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[11] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[11] + static_cast(0); @@ -7640,11 +7560,7 @@ L_0889F16C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[11] = (ctx.gpr[29] + static_cast(144)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[11] + static_cast(0); @@ -7685,11 +7601,7 @@ L_0889F16C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[3] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[3] + static_cast(0); @@ -7726,11 +7638,7 @@ L_0889F16C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[2] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7856,11 +7764,7 @@ L_0889F254: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(160)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7905,11 +7809,7 @@ L_0889F254: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(192)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8006,11 +7906,7 @@ L_0889F358: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(304)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8051,11 +7947,7 @@ L_0889F358: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(272)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -8092,11 +7984,7 @@ L_0889F358: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8541,10 +8429,7 @@ L_0889F68C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[22] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (static_cast(static_cast(static_cast(aot_mem.aot_direct_load8(ctx.gpr[16] + static_cast(1929)))))); @@ -8818,10 +8703,7 @@ L_0889F878: ctx.fpr[13] = std::bit_cast(std::bit_cast(ctx.fpr[12])); ctx.gpr[4] = (std::bit_cast(ctx.fpr[22])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -8873,24 +8755,10 @@ L_0889F8BC: { const float vfpu_constant = std::bit_cast(0x3FC90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 64u, 32u, 1u, 2u>(); + ctx.execute_vfpu_vec3_ct<64u, 32u, 0u, 1u, 1u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<64u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); goto L_0889F8E0; diff --git a/profiles/vcs/generated/generated_unit_0039.cpp b/profiles/vcs/generated/generated_unit_0039.cpp index a0c6c8f..c197018 100644 --- a/profiles/vcs/generated/generated_unit_0039.cpp +++ b/profiles/vcs/generated/generated_unit_0039.cpp @@ -1700,11 +1700,7 @@ L_088A04E4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1752,10 +1748,7 @@ L_088A0570: L_088A0578: ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 17u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); ctx.fpr[15] = std::bit_cast(ctx.gpr[4]); @@ -1793,10 +1786,7 @@ L_088A05D0: L_088A05D8: ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 17u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); ctx.fpr[15] = std::bit_cast(ctx.gpr[4]); @@ -2219,10 +2209,7 @@ L_088A0920: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[4]); ctx.gpr[19] = (0u | 0u); @@ -2570,11 +2557,7 @@ L_088A0B94: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2849,10 +2832,7 @@ L_088A0DC8: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[4]); ctx.fpr[22] = std::bit_cast(0u); @@ -6147,7 +6127,7 @@ L_088A25F0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6195,28 +6175,7 @@ L_088A2608: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6264,28 +6223,7 @@ L_088A2628: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<8u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 4u, 3u); - ctx.read_vfpu_vector_ct<8u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 4u, 8u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6342,33 +6280,8 @@ L_088A2648: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6417,10 +6330,7 @@ L_088A268C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6450,11 +6360,7 @@ L_088A26A0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6484,11 +6390,7 @@ L_088A26B8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6624,19 +6526,9 @@ L_088A2774: ctx.gpr[5] = (std::bit_cast(ctx.fpr[13])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -6693,10 +6585,7 @@ L_088A27D0: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -7423,10 +7312,7 @@ L_088A2CA8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8770,11 +8656,7 @@ L_088A37B8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8802,10 +8684,7 @@ L_088A37E4: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[13] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[28] + static_cast(-25144))); @@ -9564,11 +9443,7 @@ L_088A3D94: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(224)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -9721,10 +9596,7 @@ L_088A3ECC: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -9759,10 +9631,7 @@ L_088A3ECC: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); diff --git a/profiles/vcs/generated/generated_unit_0040.cpp b/profiles/vcs/generated/generated_unit_0040.cpp index 14eeb32..aff405d 100644 --- a/profiles/vcs/generated/generated_unit_0040.cpp +++ b/profiles/vcs/generated/generated_unit_0040.cpp @@ -1310,28 +1310,7 @@ L_088A41F0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -1355,11 +1334,7 @@ L_088A41F0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[7] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); @@ -1397,11 +1372,7 @@ L_088A41F0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1426,10 +1397,7 @@ L_088A41F0: { const float fs = ctx.fpr[12]; const float ft = ctx.fpr[13]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[12] = std::bit_cast(0x7FC00000u); else ctx.fpr[12] = fs * ft; } ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(48)); { const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); std::uint32_t vfpu_words[4]{}; @@ -2092,11 +2060,7 @@ L_088A478C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(384)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2156,10 +2120,7 @@ L_088A4820: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (15395u << 16u); @@ -2397,28 +2358,7 @@ L_088A4958: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(432)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -2456,11 +2396,7 @@ L_088A4958: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(464)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2474,15 +2410,8 @@ L_088A4958: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[22] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); @@ -2490,15 +2419,8 @@ L_088A4958: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[24] = std::bit_cast(ctx.gpr[4]); { std::uint32_t aot_run_words[3]{}; @@ -2750,28 +2672,7 @@ L_088A4BD0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(624)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2932,10 +2833,7 @@ L_088A4CF4: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (15897u << 16u); @@ -3040,28 +2938,7 @@ L_088A4D94: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(864)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -3405,11 +3282,7 @@ L_088A5058: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(1152)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -5914,10 +5787,7 @@ L_088A657C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[12] <= ctx.fpr[24])) ? 0x00800000u : 0u); @@ -6129,11 +5999,7 @@ L_088A66DC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6321,15 +6187,8 @@ L_088A68F8: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[24] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16068u << 16u); @@ -7847,19 +7706,9 @@ L_088A7358: ctx.gpr[5] = (std::bit_cast(ctx.fpr[14])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (17402u << 16u); @@ -8032,11 +7881,7 @@ L_088A74A0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(176)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8053,10 +7898,7 @@ L_088A74A0: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.fpr[14] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[28] + static_cast(7676))); diff --git a/profiles/vcs/generated/generated_unit_0041.cpp b/profiles/vcs/generated/generated_unit_0041.cpp index f9b4d94..c5d19b6 100644 --- a/profiles/vcs/generated/generated_unit_0041.cpp +++ b/profiles/vcs/generated/generated_unit_0041.cpp @@ -1341,10 +1341,7 @@ L_088A83D8: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[6] = (ctx.gpr[16] + static_cast(32)); @@ -1403,11 +1400,7 @@ L_088A8418: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1470,10 +1463,7 @@ L_088A84BC: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[6] = (ctx.gpr[16] + static_cast(32)); @@ -1532,11 +1522,7 @@ L_088A84FC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1592,10 +1578,7 @@ L_088A859C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[5] = (ctx.gpr[16] + static_cast(192)); @@ -1657,11 +1640,7 @@ L_088A85D8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1964,28 +1943,7 @@ L_088A87EC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(208)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2047,11 +2005,7 @@ L_088A8848: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2083,10 +2037,7 @@ L_088A8848: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.fpr[13] = ctx.fpr[12] + ctx.fpr[24]; @@ -2125,11 +2076,7 @@ L_088A8848: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2512,11 +2459,7 @@ L_088A8B74: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[17] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); @@ -2571,11 +2514,7 @@ L_088A8BEC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2790,7 +2729,7 @@ L_088A8E2C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[23] = (ctx.gpr[29] + static_cast(96)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[23] + static_cast(0); @@ -2814,11 +2753,7 @@ L_088A8E2C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2852,11 +2787,7 @@ L_088A8E9C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3474,11 +3405,7 @@ L_088A92EC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3496,10 +3423,7 @@ L_088A92EC: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -3958,10 +3882,7 @@ L_088A9694: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[4]); ctx.gpr[31] = (0x088A96B4u); @@ -3980,10 +3901,7 @@ L_088A96B4: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[20] <= ctx.fpr[12])) ? 0x00800000u : 0u); @@ -4012,10 +3930,7 @@ L_088A96E4: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (15820u << 16u); @@ -5308,19 +5223,9 @@ L_088AA1E4: ctx.gpr[5] = (std::bit_cast(ctx.fpr[20])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(64), std::bit_cast(ctx.fpr[12])); @@ -5331,19 +5236,9 @@ L_088AA1E4: ctx.gpr[5] = (std::bit_cast(ctx.fpr[20])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(68), std::bit_cast(ctx.fpr[12])); @@ -5358,19 +5253,9 @@ L_088AA260: ctx.gpr[5] = (std::bit_cast(ctx.fpr[20])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(72), std::bit_cast(ctx.fpr[12])); @@ -5411,28 +5296,7 @@ L_088AA260: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<8u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 4u, 3u); - ctx.read_vfpu_vector_ct<8u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 4u, 8u, 3u, 3u>(); ctx.gpr[16] = (ctx.gpr[29] + static_cast(96)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); @@ -5489,28 +5353,7 @@ L_088AA260: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[30] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5637,7 +5480,7 @@ L_088AA3CC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6034,10 +5877,7 @@ L_088AA6C4: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (15692u << 16u); @@ -7309,11 +7149,7 @@ L_088AB084: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[2] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[2] + static_cast(0); @@ -7329,10 +7165,7 @@ L_088AB084: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[2] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[2]); ctx.gpr[2] = (aot_mem.aot_direct_load32(ctx.gpr[28] + static_cast(8436))); @@ -9091,33 +8924,8 @@ L_088ABF6C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(144)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); diff --git a/profiles/vcs/generated/generated_unit_0042.cpp b/profiles/vcs/generated/generated_unit_0042.cpp index d461df8..56939e1 100644 --- a/profiles/vcs/generated/generated_unit_0042.cpp +++ b/profiles/vcs/generated/generated_unit_0042.cpp @@ -626,33 +626,8 @@ L_088AC000: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(160)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -843,33 +818,8 @@ L_088AC0E8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(320)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -933,33 +883,8 @@ L_088AC0E8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(336)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -1150,33 +1075,8 @@ L_088AC264: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(496)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1240,33 +1140,8 @@ L_088AC264: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(512)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -1479,11 +1354,7 @@ L_088AC4C0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2060,11 +1931,7 @@ L_088AC958: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2109,11 +1976,7 @@ L_088AC958: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2159,11 +2022,7 @@ L_088AC958: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2209,11 +2068,7 @@ L_088AC958: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2258,11 +2113,7 @@ L_088AC958: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2307,11 +2158,7 @@ L_088AC958: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2429,11 +2276,7 @@ L_088ACADC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(96)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -2474,11 +2317,7 @@ L_088ACADC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2732,11 +2571,7 @@ L_088ACD54: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2754,10 +2589,7 @@ L_088ACD54: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -2839,11 +2671,7 @@ L_088ACD54: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2888,11 +2716,7 @@ L_088ACD54: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3023,11 +2847,7 @@ L_088ACEB8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(240)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -3054,10 +2874,7 @@ L_088ACF0C: ctx.fpr[12] = ctx.fpr[14] + ctx.fpr[12]; ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -3174,11 +2991,7 @@ L_088ACF8C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(304)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -3205,10 +3018,7 @@ L_088ACFE0: ctx.fpr[12] = ctx.fpr[14] + ctx.fpr[12]; ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -3276,11 +3086,7 @@ L_088AD008: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3290,10 +3096,7 @@ L_088AD008: ctx.fpr[13] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (std::bit_cast(ctx.fpr[13])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -3344,11 +3147,7 @@ L_088AD008: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(160)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3356,10 +3155,7 @@ L_088AD008: aot_mem.aot_direct_store32_block(vfpu_address, vfpu_words); } ctx.gpr[5] = (std::bit_cast(ctx.fpr[13])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -3809,7 +3605,7 @@ L_088AD614: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[18] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); @@ -3833,7 +3629,7 @@ L_088AD614: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3967,10 +3763,7 @@ L_088AD708: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(432)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4042,10 +3835,7 @@ L_088AD768: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(448)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4090,11 +3880,7 @@ L_088AD78C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[20] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); @@ -4118,11 +3904,7 @@ L_088AD78C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4181,11 +3963,7 @@ L_088AD78C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4208,11 +3986,7 @@ L_088AD78C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4269,11 +4043,7 @@ L_088AD78C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4330,11 +4100,7 @@ L_088AD78C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4391,11 +4157,7 @@ L_088AD78C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4418,11 +4180,7 @@ L_088AD78C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4479,11 +4237,7 @@ L_088AD78C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4506,11 +4260,7 @@ L_088AD78C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4567,11 +4317,7 @@ L_088AD78C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4628,11 +4374,7 @@ L_088AD78C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4739,11 +4481,7 @@ L_088ADB58: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4766,11 +4504,7 @@ L_088ADB58: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4827,11 +4561,7 @@ L_088ADB58: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4854,11 +4584,7 @@ L_088ADB58: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4915,11 +4641,7 @@ L_088ADB58: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4942,11 +4664,7 @@ L_088ADB58: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5003,11 +4721,7 @@ L_088ADB58: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5030,11 +4744,7 @@ L_088ADB58: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7080,15 +6790,8 @@ L_088AF270: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[28] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[19] + static_cast(116), std::bit_cast(ctx.fpr[30])); @@ -7519,11 +7222,7 @@ L_088AF5CC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7571,10 +7270,7 @@ L_088AF62C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7602,7 +7298,7 @@ L_088AF62C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7621,10 +7317,7 @@ L_088AF62C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7815,11 +7508,7 @@ L_088AF7F0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[16] = (ctx.gpr[29] + static_cast(304)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); @@ -7851,10 +7540,7 @@ L_088AF7F0: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7937,7 +7623,7 @@ L_088AF884: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7968,10 +7654,7 @@ L_088AF884: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7996,7 +7679,7 @@ L_088AF884: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8027,10 +7710,7 @@ L_088AF884: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -8063,11 +7743,7 @@ L_088AF8FC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[16] = (ctx.gpr[29] + static_cast(352)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); @@ -8128,10 +7804,7 @@ L_088AF94C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -8186,10 +7859,7 @@ L_088AF990: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -8735,11 +8405,7 @@ L_088AFE14: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[16] = (ctx.gpr[29] + static_cast(496)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); @@ -8771,10 +8437,7 @@ L_088AFE14: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -8857,7 +8520,7 @@ L_088AFEA8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8888,10 +8551,7 @@ L_088AFEA8: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -8916,7 +8576,7 @@ L_088AFEA8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8947,10 +8607,7 @@ L_088AFEA8: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -8983,11 +8640,7 @@ L_088AFF20: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(544)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -9048,10 +8701,7 @@ L_088AFF70: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -9091,7 +8741,7 @@ L_088AFF70: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9122,10 +8772,7 @@ L_088AFF70: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); diff --git a/profiles/vcs/generated/generated_unit_0043.cpp b/profiles/vcs/generated/generated_unit_0043.cpp index f87f938..77bd285 100644 --- a/profiles/vcs/generated/generated_unit_0043.cpp +++ b/profiles/vcs/generated/generated_unit_0043.cpp @@ -1263,15 +1263,8 @@ L_088B0608: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[5]); aot_mem.aot_direct_store32(ctx.gpr[28] + static_cast(-24928), std::bit_cast(ctx.fpr[13])); @@ -1280,15 +1273,8 @@ L_088B0608: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[15] = std::bit_cast(ctx.gpr[5]); aot_mem.aot_direct_store32(ctx.gpr[28] + static_cast(-24924), std::bit_cast(ctx.fpr[15])); @@ -1548,33 +1534,8 @@ L_088B086C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1604,11 +1565,7 @@ L_088B0894: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1661,34 +1618,10 @@ L_088B0908: ctx.gpr[18] = (ctx.gpr[18] + static_cast(16)); ctx.execute_vfpu_vx2i_ct<0u, 2u, 2u, 3u>(); ctx.execute_vfpu_vx2i_ct<1u, 66u, 2u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<0u, 3u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<3u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(23u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 3u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<3u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(23u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 3u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<55u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<55u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 3u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<0u, 0u, 3u, 23u>(); + ctx.execute_vfpu_vi2f_ct<1u, 1u, 3u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 55u, 3u, 1u>(); + ctx.execute_vfpu_vec3_ct<1u, 1u, 55u, 3u, 0u>(); ctx.execute_vfpu_vcmp_ct<20u, 0u, 3u, 2u>(); { const bool branch_taken = ((ctx.vfpu_ctrl[3] >> 4u) & 1u) != 0u; // nop @@ -1748,78 +1681,27 @@ L_088B0960: ctx.set_vfpu_scalar_bits_ct<10u>(ctx.gpr[13]); ctx.set_vfpu_scalar_bits_ct<42u>(ctx.gpr[10]); ctx.execute_vfpu_vx2i_ct<4u, 8u, 2u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<4u, 4u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(23u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 4u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<4u, 4u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<4u, 4u, 4u, 23u>(); ctx.execute_vfpu_vx2i_ct<5u, 9u, 2u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<5u, 4u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(23u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 4u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<5u, 4u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<5u, 5u, 4u, 23u>(); ctx.execute_vfpu_vx2i_ct<6u, 10u, 2u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<6u, 4u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(23u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 4u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<6u, 4u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<6u, 6u, 4u, 23u>(); ctx.execute_vfpu_horizontal_ct<8u, 36u, 3u, true>(); ctx.execute_vfpu_horizontal_ct<40u, 37u, 3u, true>(); ctx.execute_vfpu_horizontal_ct<72u, 38u, 3u, true>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<4u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<8u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<12u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<5u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<8u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<13u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<6u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<8u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<14u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<12u, 4u, 8u, 3u, 1u>(); + ctx.execute_vfpu_vec3_ct<13u, 5u, 8u, 3u, 1u>(); + ctx.execute_vfpu_vec3_ct<14u, 6u, 8u, 3u, 1u>(); ctx.execute_vfpu_vdot_ct<15u, 12u, 12u, 3u>(); ctx.execute_vfpu_vdot_ct<47u, 13u, 13u, 3u>(); ctx.execute_vfpu_vdot_ct<79u, 14u, 14u, 3u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<20u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<8u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<12u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<12u, 20u, 8u, 3u, 1u>(); ctx.execute_vfpu_vminmax_ct<15u, 15u, 47u, 1u, true>(); ctx.execute_vfpu_vminmax_ct<15u, 15u, 79u, 1u, true>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<15u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<104u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<104u, 15u, 1u, 22u>(); ctx.execute_vfpu_vdot_ct<0u, 12u, 12u, 3u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<104u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<116u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<32u, 104u, 116u, 1u, 0u>(); + ctx.execute_vfpu_vec3_ct<32u, 32u, 32u, 1u, 2u>(); ctx.execute_vfpu_vcmp_ct<0u, 32u, 1u, 7u>(); { const bool branch_taken = ((ctx.vfpu_ctrl[3] >> 0u) & 1u) != 0u; // nop @@ -1829,33 +1711,15 @@ L_088B0960: goto L_088B0A1C; } L_088B0A1C: - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<5u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<4u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<9u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<6u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<4u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<10u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<20u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<4u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<15u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<9u, 5u, 4u, 3u, 1u>(); + ctx.execute_vfpu_vec3_ct<10u, 6u, 4u, 3u, 1u>(); + ctx.execute_vfpu_vec3_ct<15u, 20u, 4u, 3u, 1u>(); ctx.execute_vfpu_cross_quat_ct<17u, 10u, 9u, 3u>(); ctx.execute_vfpu_vdot_ct<110u, 17u, 17u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<110u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<110u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<110u, 110u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<7u, 17u, 110u, 3u>(); ctx.execute_vfpu_vdot_ct<12u, 7u, 15u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<12u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<12u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<12u, 12u, 1u, 1u>(); ctx.execute_vfpu_vcmp_ct<12u, 116u, 1u, 7u>(); { const bool branch_taken = ((ctx.vfpu_ctrl[3] >> 0u) & 1u) != 0u; // nop @@ -1870,34 +1734,12 @@ L_088B0A4C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_value); } { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<96u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<8u, 4u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<29u, 4u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<20u, 4u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<30u, 4u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<5u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<4u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<9u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<4u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<5u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<8u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<6u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<4u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<10u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<6u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<5u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<11u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<29u, 8u, 4u, 0u>(); + ctx.execute_vfpu_unary_ct<30u, 20u, 4u, 0u>(); + ctx.execute_vfpu_vec3_ct<9u, 5u, 4u, 3u, 1u>(); + ctx.execute_vfpu_vec3_ct<8u, 4u, 5u, 3u, 1u>(); + ctx.execute_vfpu_vec3_ct<10u, 6u, 4u, 3u, 1u>(); + ctx.execute_vfpu_vec3_ct<11u, 6u, 5u, 3u, 1u>(); ctx.execute_vfpu_cross_quat_ct<19u, 9u, 10u, 3u>(); ctx.execute_vfpu_cross_quat_ct<17u, 10u, 9u, 3u>(); ctx.execute_vfpu_cross_quat_ct<15u, 11u, 8u, 3u>(); @@ -1920,22 +1762,10 @@ L_088B0A84: ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } ctx.set_vfpu_scalar_bits_ct<127u>(aot_mem.aot_direct_load32(ctx.gpr[4] + static_cast(28))); ctx.gpr[4] = (ctx.gpr[4] + static_cast(64)); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<29u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<12u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<12u, 1u, 29u, 3u, 1u>(); ctx.execute_vfpu_vdot_ct<31u, 12u, 12u, 3u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<104u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<97u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<63u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<63u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<63u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<63u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<63u, 104u, 97u, 1u, 0u>(); + ctx.execute_vfpu_vec3_ct<63u, 63u, 63u, 1u, 2u>(); ctx.execute_vfpu_vcmp_ct<31u, 63u, 1u, 7u>(); { const bool branch_taken = ((ctx.vfpu_ctrl[3] >> 0u) & 1u) != 0u; // nop @@ -1946,20 +1776,9 @@ L_088B0A84: } L_088B0AAC: ctx.execute_vfpu_vdot_ct<2u, 1u, 7u, 3u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<2u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<103u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<98u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<34u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<98u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<98u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<63u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 2u, 103u, 1u, 1u>(); + ctx.execute_vfpu_unary_ct<34u, 98u, 1u, 1u>(); + ctx.execute_vfpu_vec3_ct<63u, 98u, 98u, 1u, 2u>(); ctx.execute_vfpu_vcmp_ct<34u, 97u, 1u, 7u>(); { const bool branch_taken = ((ctx.vfpu_ctrl[3] >> 0u) & 1u) != 0u; // nop @@ -1999,16 +1818,9 @@ L_088B0AE4: goto L_088B0AF0; } L_088B0AF0: - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<24u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<27u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<27u, 1u, 24u, 3u, 1u>(); ctx.execute_vfpu_vdot_ct<120u, 27u, 27u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<120u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<122u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<122u, 120u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<28u, 27u, 122u, 3u>(); ctx.gpr[12] = (aot_mem.aot_direct_load16(ctx.gpr[4] + static_cast(-20))); ctx.gpr[13] = (aot_mem.aot_direct_load16(ctx.gpr[24] + static_cast(-18))); @@ -2023,15 +1835,8 @@ L_088B0AF0: const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(-32); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; aot_mem.aot_direct_store32_block(vfpu_address, vfpu_words); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<120u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<123u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<97u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<123u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<124u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<123u, 120u, 1u, 22u>(); + ctx.execute_vfpu_vec3_ct<124u, 97u, 123u, 1u, 1u>(); aot_mem.aot_direct_store32(ctx.gpr[4] + static_cast(-16), ctx.vfpu_scalar_bits_ct<124u>()); goto L_088B0B28; L_088B0B28: @@ -2043,18 +1848,9 @@ L_088B0B28: goto L_088B0B30; } L_088B0B30: - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<30u, 4u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<20u, 4u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<116u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<117u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<116u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<118u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<20u, 30u, 4u, 0u>(); + ctx.execute_vfpu_unary_ct<117u, 116u, 1u, 0u>(); + ctx.execute_vfpu_unary_ct<118u, 116u, 1u, 0u>(); goto L_088B0B3C; L_088B0B3C: { const bool branch_taken = ctx.gpr[23] == ctx.gpr[7]; @@ -2106,14 +1902,8 @@ L_088B0B68: goto L_088B0B70; } L_088B0B70: - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<24u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<97u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<97u, 24u, 1u, 0u>(); + ctx.execute_vfpu_unary_ct<28u, 7u, 3u, 0u>(); ctx.gpr[13] = (aot_mem.aot_direct_load16(ctx.gpr[24] + static_cast(-2))); ctx.gpr[13] = (ctx.gpr[13] << 16u); ctx.set_vfpu_scalar_bits_ct<124u>(ctx.gpr[13]); @@ -2170,16 +1960,8 @@ L_088B0BA8: return; L_088B0BD0: ctx.gpr[2] = (0u + 0u); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<4u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<12u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<5u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<13u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<12u, 1u, 4u, 3u, 1u>(); + ctx.execute_vfpu_vec3_ct<13u, 1u, 5u, 3u, 1u>(); ctx.execute_vfpu_cross_quat_ct<18u, 9u, 12u, 3u>(); ctx.execute_vfpu_cross_quat_ct<16u, 10u, 12u, 3u>(); ctx.execute_vfpu_cross_quat_ct<14u, 11u, 13u, 3u>(); @@ -2219,10 +2001,7 @@ L_088B0C10: goto L_088B0C18; } L_088B0C18: - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<4u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<20u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<20u, 4u, 3u, 0u>(); { const bool branch_taken = 0u == 0u; // vflush: architectural no-op that retains VFPU prefixes if (branch_taken) { @@ -2239,10 +2018,7 @@ L_088B0C24: goto L_088B0C2C; } L_088B0C2C: - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<5u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<20u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<20u, 5u, 3u, 0u>(); { const bool branch_taken = 0u == 0u; // vflush: architectural no-op that retains VFPU prefixes if (branch_taken) { @@ -2251,14 +2027,8 @@ L_088B0C2C: goto L_088B0C38; } L_088B0C38: - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<4u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<20u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<5u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<21u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<20u, 4u, 3u, 0u>(); + ctx.execute_vfpu_unary_ct<21u, 5u, 3u, 0u>(); { const bool branch_taken = 0u == 0u; // vflush: architectural no-op that retains VFPU prefixes if (branch_taken) { @@ -2283,10 +2053,7 @@ L_088B0C50: goto L_088B0C58; } L_088B0C58: - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<6u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<20u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<20u, 6u, 3u, 0u>(); { const bool branch_taken = 0u == 0u; // vflush: architectural no-op that retains VFPU prefixes if (branch_taken) { @@ -2295,14 +2062,8 @@ L_088B0C58: goto L_088B0C64; } L_088B0C64: - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<4u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<20u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<6u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<21u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<20u, 4u, 3u, 0u>(); + ctx.execute_vfpu_unary_ct<21u, 6u, 3u, 0u>(); { const bool branch_taken = 0u == 0u; // vflush: architectural no-op that retains VFPU prefixes if (branch_taken) { @@ -2319,14 +2080,8 @@ L_088B0C74: goto L_088B0C7C; } L_088B0C7C: - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<5u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<20u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<6u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<21u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<20u, 5u, 3u, 0u>(); + ctx.execute_vfpu_unary_ct<21u, 6u, 3u, 0u>(); { const bool branch_taken = 0u == 0u; // vflush: architectural no-op that retains VFPU prefixes if (branch_taken) { @@ -2336,11 +2091,7 @@ L_088B0C7C: } L_088B0C8C: ctx.execute_vfpu_vscl_ct<25u, 7u, 98u, 3u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<25u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<24u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<24u, 1u, 25u, 3u, 1u>(); { const bool branch_taken = 0u == 0u; // vflush: architectural no-op that retains VFPU prefixes if (branch_taken) { @@ -2349,17 +2100,9 @@ L_088B0C8C: goto L_088B0C9C; } L_088B0C9C: - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<20u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<3u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<3u, 1u, 20u, 3u, 1u>(); ctx.execute_vfpu_vdot_ct<66u, 3u, 3u, 3u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<97u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<97u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<99u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<99u, 97u, 97u, 1u, 2u>(); ctx.execute_vfpu_vcmp_ct<66u, 99u, 1u, 3u>(); { const bool branch_taken = ((ctx.vfpu_ctrl[3] >> 0u) & 1u) == 0u; // vflush: architectural no-op that retains VFPU prefixes @@ -2369,10 +2112,7 @@ L_088B0C9C: goto L_088B0CB4; } L_088B0CB4: - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<20u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<24u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<24u, 20u, 3u, 0u>(); { const bool branch_taken = 0u == 0u; // vflush: architectural no-op that retains VFPU prefixes if (branch_taken) { @@ -2381,33 +2121,15 @@ L_088B0CB4: goto L_088B0CC0; } L_088B0CC0: - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<20u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<3u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<21u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<20u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<22u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<3u, 1u, 20u, 3u, 1u>(); + ctx.execute_vfpu_vec3_ct<22u, 21u, 20u, 3u, 1u>(); ctx.execute_vfpu_vdot_ct<116u, 22u, 22u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<116u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<117u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<117u, 116u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<23u, 22u, 117u, 3u>(); ctx.execute_vfpu_vdot_ct<66u, 3u, 3u, 3u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<97u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<97u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<99u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<99u, 97u, 97u, 1u, 2u>(); ctx.execute_vfpu_vdot_ct<118u, 3u, 23u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<20u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<24u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<24u, 20u, 3u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 99u, 1u, 3u>(); { const bool branch_taken = ((ctx.vfpu_ctrl[3] >> 0u) & 1u) != 0u; // vflush: architectural no-op that retains VFPU prefixes @@ -2426,16 +2148,8 @@ L_088B0CF0: goto L_088B0CFC; } L_088B0CFC: - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<118u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<118u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<119u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<119u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<115u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<119u, 118u, 118u, 1u, 2u>(); + ctx.execute_vfpu_vec3_ct<115u, 66u, 119u, 1u, 1u>(); ctx.execute_vfpu_vcmp_ct<115u, 99u, 1u, 7u>(); { const bool branch_taken = ((ctx.vfpu_ctrl[3] >> 0u) & 1u) != 0u; // vflush: architectural no-op that retains VFPU prefixes @@ -2445,24 +2159,10 @@ L_088B0CFC: goto L_088B0D10; } L_088B0D10: - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<116u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<114u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<99u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<115u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<113u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<113u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<112u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<118u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<112u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<121u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<114u, 116u, 1u, 22u>(); + ctx.execute_vfpu_vec3_ct<113u, 99u, 115u, 1u, 1u>(); + ctx.execute_vfpu_unary_ct<112u, 113u, 1u, 22u>(); + ctx.execute_vfpu_vec3_ct<121u, 118u, 112u, 1u, 1u>(); ctx.execute_vfpu_vcmp_ct<121u, 114u, 1u, 7u>(); { const bool branch_taken = ((ctx.vfpu_ctrl[3] >> 0u) & 1u) != 0u; // vflush: architectural no-op that retains VFPU prefixes @@ -2474,11 +2174,7 @@ L_088B0D10: L_088B0D2C: ctx.execute_vfpu_vminmax_ct<111u, 118u, 114u, 1u, false>(); ctx.execute_vfpu_vscl_ct<26u, 23u, 111u, 3u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<26u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<20u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<24u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<24u, 26u, 20u, 3u, 0u>(); goto L_088B0D38; L_088B0D38: ctx.gpr[2] = (ctx.gpr[2] + static_cast(1)); @@ -2526,26 +2222,11 @@ L_088B0D68: goto L_088B0D70; } L_088B0D70: - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<2u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<3u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<103u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<24u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<24u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<3u, 2u, 1u, 3u, 1u>(); + ctx.execute_vfpu_vec3_ct<24u, 103u, 24u, 1u, 1u>(); ctx.execute_vfpu_vdot_ct<56u, 3u, 7u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<56u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<56u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<24u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<56u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<24u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<56u, 56u, 1u, 16u>(); + ctx.execute_vfpu_vec3_ct<24u, 24u, 56u, 1u, 2u>(); ctx.execute_vfpu_vcmp_ct<24u, 127u, 1u, 6u>(); { const bool branch_taken = ((ctx.vfpu_ctrl[3] >> 0u) & 1u) != 0u; // nop @@ -2556,21 +2237,9 @@ L_088B0D70: } L_088B0D90: ctx.execute_vfpu_vscl_ct<8u, 3u, 24u, 3u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<8u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<4u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<12u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<5u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<13u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<1u, 8u, 1u, 3u, 0u>(); + ctx.execute_vfpu_vec3_ct<12u, 1u, 4u, 3u, 1u>(); + ctx.execute_vfpu_vec3_ct<13u, 1u, 5u, 3u, 1u>(); ctx.execute_vfpu_cross_quat_ct<18u, 9u, 12u, 3u>(); ctx.execute_vfpu_cross_quat_ct<16u, 10u, 12u, 3u>(); ctx.execute_vfpu_cross_quat_ct<14u, 11u, 13u, 3u>(); @@ -2636,60 +2305,10 @@ L_088B0E00: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<4u, 4u>(vfpu_value); } ctx.gpr[8] = (ctx.gpr[8] + static_cast(32)); - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<44u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<0u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 3u>(vfpu_result); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<44u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<4u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<5u, 3u>(vfpu_result); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<15u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<5u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<15u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<4u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<1u, 44u, 0u, 3u, 3u>(); + ctx.execute_vfpu_vtfm_ct<5u, 44u, 4u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 1u, 15u, 3u, 0u>(); + ctx.execute_vfpu_vec3_ct<4u, 5u, 15u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[9] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2751,51 +2370,12 @@ L_088B0E7C: ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.set_vfpu_scalar_bits_ct<101u>(aot_mem.aot_direct_load32(ctx.gpr[8] + static_cast(16))); ctx.gpr[8] = (ctx.gpr[8] + static_cast(32)); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<96u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<97u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<96u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<44u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<0u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 3u>(vfpu_result); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<15u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<35u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<24u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<29u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<35u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<25u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<97u, 96u, 1u, 0u>(); + ctx.execute_vfpu_unary_ct<98u, 96u, 1u, 0u>(); + ctx.execute_vfpu_vtfm_ct<1u, 44u, 0u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 1u, 15u, 3u, 0u>(); + ctx.execute_vfpu_vec3_ct<24u, 28u, 35u, 3u, 1u>(); + ctx.execute_vfpu_vec3_ct<25u, 29u, 35u, 3u, 0u>(); ctx.execute_vfpu_vcmp_ct<0u, 24u, 3u, 2u>(); { const bool branch_taken = ((ctx.vfpu_ctrl[3] >> 4u) & 1u) != 0u; // nop @@ -2899,51 +2479,12 @@ L_088B0F44: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<96u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<97u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<96u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<48u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<0u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 3u>(vfpu_result); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<30u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<35u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<24u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<31u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<35u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<25u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<19u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<97u, 96u, 1u, 0u>(); + ctx.execute_vfpu_unary_ct<98u, 96u, 1u, 0u>(); + ctx.execute_vfpu_vtfm_ct<1u, 48u, 0u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<24u, 30u, 35u, 3u, 1u>(); + ctx.execute_vfpu_vec3_ct<25u, 31u, 35u, 3u, 0u>(); + ctx.execute_vfpu_vec3_ct<0u, 1u, 19u, 3u, 0u>(); ctx.gpr[8] = (ctx.gpr[8] + static_cast(32)); ctx.execute_vfpu_vcmp_ct<0u, 24u, 3u, 2u>(); { const bool branch_taken = ((ctx.vfpu_ctrl[3] >> 4u) & 1u) != 0u; @@ -3027,16 +2568,8 @@ L_088B0FDC: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } ctx.gpr[8] = (ctx.gpr[8] + static_cast(48)); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<55u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<55u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 55u, 3u, 1u>(); + ctx.execute_vfpu_vec3_ct<1u, 1u, 55u, 3u, 0u>(); ctx.execute_vfpu_vcmp_ct<20u, 0u, 3u, 2u>(); { const bool branch_taken = ((ctx.vfpu_ctrl[3] >> 4u) & 1u) != 0u; // nop @@ -3108,11 +2641,7 @@ L_088B1028: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[8] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[8] + static_cast(0); @@ -3128,10 +2657,7 @@ L_088B1028: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[8] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[8]); ctx.fpr[12] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[5] + static_cast(12))); @@ -3189,10 +2715,7 @@ L_088B10DC: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -3241,11 +2764,7 @@ L_088B10DC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[9] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[9] + static_cast(0); @@ -3439,11 +2958,7 @@ L_088B12A8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -3459,10 +2974,7 @@ L_088B12A8: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[7] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[16] = std::bit_cast(ctx.gpr[7]); ctx.fpr[12] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[5] + static_cast(0))); @@ -3489,11 +3001,7 @@ L_088B12A8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[7] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); @@ -3681,11 +3189,7 @@ L_088B1424: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3708,11 +3212,7 @@ L_088B1424: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[7] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); @@ -3798,11 +3298,7 @@ L_088B149C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3818,10 +3314,7 @@ L_088B149C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); { const bool branch_taken = 0u == 0u; @@ -3850,11 +3343,7 @@ L_088B14CC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3870,10 +3359,7 @@ L_088B14CC: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); { const bool branch_taken = 0u == 0u; @@ -3961,11 +3447,7 @@ L_088B1554: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[8] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[8] + static_cast(0); @@ -4007,11 +3489,7 @@ L_088B1554: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[9] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[9] + static_cast(0); @@ -4148,11 +3626,7 @@ L_088B1664: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[8] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[8] + static_cast(0); @@ -4193,11 +3667,7 @@ L_088B1664: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(96)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4228,11 +3698,7 @@ L_088B1664: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[8] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[8] + static_cast(0); @@ -4251,10 +3717,7 @@ L_088B1664: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -4348,44 +3811,20 @@ L_088B1780: ctx.execute_vfpu_compare3_ct<13u, 3u, 4u, 3u, 7u>(); { float vfpu_value[4]{}; ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 3u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<12u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<13u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<12u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<12u, 12u, 13u, 3u, 0u>(); ctx.execute_vfpu_vcmp_ct<12u, 28u, 3u, 7u>(); { const bool branch_taken = ((ctx.vfpu_ctrl[3] >> 4u) & 1u) != 0u; - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<5u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<4u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<15u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<15u, 5u, 4u, 3u, 0u>(); if (branch_taken) { goto L_088B180C; } goto L_088B17B4; } L_088B17B4: - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<14u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<5u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<4u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<8u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<12u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<15u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<14u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<9u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<14u, 0u, 1u, 3u, 0u>(); + ctx.execute_vfpu_vec3_ct<8u, 5u, 4u, 3u, 1u>(); + ctx.execute_vfpu_vec3_ct<12u, 1u, 0u, 3u, 1u>(); + ctx.execute_vfpu_vec3_ct<9u, 15u, 14u, 3u, 1u>(); ctx.vfpu_ctrl[0u] = 0x000040C9u; ctx.vfpu_ctrl[1u] = 0x000043C6u; ctx.execute_vfpu_vdot_ct<16u, 8u, 12u, 3u>(); @@ -4488,60 +3927,10 @@ L_088B1820: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<5u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<19u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<19u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<3u, 3u>(vfpu_d); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<16u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<2u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<16u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<3u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 3u>(vfpu_result); } + ctx.execute_vfpu_vec3_ct<2u, 0u, 19u, 3u, 1u>(); + ctx.execute_vfpu_vec3_ct<3u, 1u, 19u, 3u, 1u>(); + ctx.execute_vfpu_vtfm_ct<0u, 16u, 2u, 3u, 3u>(); + ctx.execute_vfpu_vtfm_ct<1u, 16u, 3u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4556,11 +3945,7 @@ L_088B1820: ctx.execute_vfpu_compare3_ct<13u, 3u, 4u, 3u, 7u>(); { float vfpu_value[4]{}; ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 3u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<12u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<13u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<12u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<12u, 12u, 13u, 3u, 0u>(); ctx.execute_vfpu_vcmp_ct<12u, 28u, 3u, 7u>(); { const bool branch_taken = ((ctx.vfpu_ctrl[3] >> 4u) & 1u) != 0u; // nop @@ -4570,31 +3955,11 @@ L_088B1820: goto L_088B187C; } L_088B187C: - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<5u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<4u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<15u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<14u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<5u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<4u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<8u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<12u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<15u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<14u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<9u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<15u, 5u, 4u, 3u, 0u>(); + ctx.execute_vfpu_vec3_ct<14u, 0u, 1u, 3u, 0u>(); + ctx.execute_vfpu_vec3_ct<8u, 5u, 4u, 3u, 1u>(); + ctx.execute_vfpu_vec3_ct<12u, 1u, 0u, 3u, 1u>(); + ctx.execute_vfpu_vec3_ct<9u, 15u, 14u, 3u, 1u>(); ctx.vfpu_ctrl[0u] = 0x000040C9u; ctx.vfpu_ctrl[1u] = 0x000043C6u; ctx.execute_vfpu_vdot_ct<16u, 8u, 12u, 3u>(); @@ -4949,22 +4314,11 @@ L_088B1B74: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<6u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<5u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<4u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<9u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<6u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<4u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<10u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<9u, 5u, 4u, 3u, 1u>(); + ctx.execute_vfpu_vec3_ct<10u, 6u, 4u, 3u, 1u>(); ctx.execute_vfpu_cross_quat_ct<17u, 10u, 9u, 3u>(); ctx.execute_vfpu_vdot_ct<110u, 17u, 17u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<110u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<110u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<110u, 110u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<7u, 17u, 110u, 3u>(); ctx.execute_vfpu_vdot_ct<103u, 7u, 4u, 3u>(); ctx.execute_vfpu_vdot_ct<24u, 1u, 7u, 3u>(); @@ -5002,54 +4356,19 @@ L_088B1BC8: goto L_088B1BD0; } L_088B1BD0: - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<2u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<3u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<103u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<24u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<24u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<3u, 2u, 1u, 3u, 1u>(); + ctx.execute_vfpu_vec3_ct<24u, 103u, 24u, 1u, 1u>(); ctx.execute_vfpu_vdot_ct<56u, 3u, 7u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<56u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<56u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<24u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<56u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<24u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<56u, 56u, 1u, 16u>(); + ctx.execute_vfpu_vec3_ct<24u, 24u, 56u, 1u, 2u>(); ctx.execute_vfpu_vscl_ct<8u, 3u, 24u, 3u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<8u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<1u, 8u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<4u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<5u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<8u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<6u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<5u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<11u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<4u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<12u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<5u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<13u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<8u, 4u, 5u, 3u, 1u>(); + ctx.execute_vfpu_vec3_ct<11u, 6u, 5u, 3u, 1u>(); + ctx.execute_vfpu_vec3_ct<12u, 1u, 4u, 3u, 1u>(); + ctx.execute_vfpu_vec3_ct<13u, 1u, 5u, 3u, 1u>(); ctx.execute_vfpu_cross_quat_ct<19u, 9u, 10u, 3u>(); ctx.execute_vfpu_cross_quat_ct<15u, 11u, 8u, 3u>(); ctx.execute_vfpu_cross_quat_ct<18u, 9u, 12u, 3u>(); @@ -6765,11 +6084,7 @@ L_088B27E0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(496)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6779,10 +6094,7 @@ L_088B27E0: ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -6817,11 +6129,7 @@ L_088B27E0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(512)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7025,11 +6333,7 @@ L_088B29D8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(576)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -7208,11 +6512,7 @@ L_088B2BB8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(672)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -7371,11 +6671,7 @@ L_088B2D78: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(768)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -7461,11 +6757,7 @@ L_088B2E64: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7507,10 +6799,7 @@ L_088B2EB8: ctx.fpr[16] = std::sqrt(ctx.fpr[12]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[16])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); std::uint32_t vfpu_words[4]{}; @@ -7612,70 +6901,23 @@ L_088B2FA8: ctx.set_vfpu_scalar_bits_ct<10u>(ctx.gpr[13]); ctx.set_vfpu_scalar_bits_ct<42u>(ctx.gpr[10]); ctx.execute_vfpu_vx2i_ct<2u, 8u, 2u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<2u, 4u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(23u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 4u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 4u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 2u, 4u, 23u>(); ctx.execute_vfpu_vx2i_ct<3u, 9u, 2u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<3u, 4u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(23u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 4u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<3u, 4u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<3u, 3u, 4u, 23u>(); ctx.execute_vfpu_vx2i_ct<4u, 10u, 2u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<4u, 4u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(23u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 4u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<4u, 4u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<4u, 4u, 4u, 23u>(); { float vfpu_value[4]{}; ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<3u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<2u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<5u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<4u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<2u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<6u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<5u, 3u, 2u, 3u, 1u>(); + ctx.execute_vfpu_vec3_ct<6u, 4u, 2u, 3u, 1u>(); ctx.execute_vfpu_vdot_ct<8u, 5u, 5u, 3u>(); ctx.execute_vfpu_vdot_ct<40u, 6u, 6u, 3u>(); ctx.execute_vfpu_vminmax_ct<98u, 8u, 40u, 1u, true>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<98u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<99u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<2u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<7u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<99u, 98u, 1u, 22u>(); + ctx.execute_vfpu_vec3_ct<7u, 1u, 2u, 3u, 1u>(); ctx.execute_vfpu_vdot_ct<72u, 7u, 7u, 3u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<99u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<97u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<100u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<100u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<100u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<101u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<100u, 99u, 97u, 1u, 0u>(); + ctx.execute_vfpu_vec3_ct<101u, 100u, 100u, 1u, 2u>(); ctx.gpr[2] = (0u | 0u); ctx.execute_vfpu_vcmp_ct<72u, 101u, 1u, 7u>(); { const bool branch_taken = ((ctx.vfpu_ctrl[3] >> 0u) & 1u) != 0u; @@ -8527,51 +7769,16 @@ L_088B36D0: ctx.set_vfpu_scalar_bits_ct<10u>(ctx.gpr[13]); ctx.set_vfpu_scalar_bits_ct<42u>(ctx.gpr[10]); ctx.execute_vfpu_vx2i_ct<4u, 8u, 2u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<4u, 4u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(23u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 4u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<4u, 4u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<4u, 4u, 4u, 23u>(); ctx.execute_vfpu_vx2i_ct<5u, 9u, 2u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<5u, 4u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(23u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 4u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<5u, 4u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<5u, 5u, 4u, 23u>(); ctx.execute_vfpu_vx2i_ct<6u, 10u, 2u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<6u, 4u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(23u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 4u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<6u, 4u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<5u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<4u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<9u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<6u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<4u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<10u, 3u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<6u, 6u, 4u, 23u>(); + ctx.execute_vfpu_vec3_ct<9u, 5u, 4u, 3u, 1u>(); + ctx.execute_vfpu_vec3_ct<10u, 6u, 4u, 3u, 1u>(); ctx.execute_vfpu_cross_quat_ct<17u, 10u, 9u, 3u>(); ctx.execute_vfpu_vdot_ct<110u, 17u, 17u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<110u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<110u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<110u, 110u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<7u, 17u, 110u, 3u>(); ctx.execute_vfpu_vdot_ct<103u, 7u, 4u, 3u>(); ctx.execute_vfpu_vdot_ct<24u, 1u, 7u, 3u>(); @@ -8609,54 +7816,19 @@ L_088B3784: goto L_088B378C; } L_088B378C: - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<2u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<3u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<103u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<24u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<24u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<3u, 2u, 1u, 3u, 1u>(); + ctx.execute_vfpu_vec3_ct<24u, 103u, 24u, 1u, 1u>(); ctx.execute_vfpu_vdot_ct<56u, 3u, 7u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<56u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<56u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<24u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<56u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<24u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<56u, 56u, 1u, 16u>(); + ctx.execute_vfpu_vec3_ct<24u, 24u, 56u, 1u, 2u>(); ctx.execute_vfpu_vscl_ct<8u, 3u, 24u, 3u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<8u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<1u, 8u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<4u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<5u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<8u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<6u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<5u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<11u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<4u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<12u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<5u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<13u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<8u, 4u, 5u, 3u, 1u>(); + ctx.execute_vfpu_vec3_ct<11u, 6u, 5u, 3u, 1u>(); + ctx.execute_vfpu_vec3_ct<12u, 1u, 4u, 3u, 1u>(); + ctx.execute_vfpu_vec3_ct<13u, 1u, 5u, 3u, 1u>(); ctx.execute_vfpu_cross_quat_ct<19u, 9u, 10u, 3u>(); ctx.execute_vfpu_cross_quat_ct<15u, 11u, 8u, 3u>(); ctx.execute_vfpu_cross_quat_ct<18u, 9u, 12u, 3u>(); @@ -8984,24 +8156,8 @@ L_088B3A64: ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } ctx.execute_vfpu_vx2i_ct<0u, 2u, 2u, 3u>(); ctx.execute_vfpu_vx2i_ct<1u, 66u, 2u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<0u, 3u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<3u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(23u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 3u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<3u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(23u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 3u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 3u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<0u, 0u, 3u, 23u>(); + ctx.execute_vfpu_vi2f_ct<1u, 1u, 3u, 23u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9629,24 +8785,8 @@ L_088B3FCC: ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } ctx.execute_vfpu_vx2i_ct<0u, 2u, 2u, 3u>(); ctx.execute_vfpu_vx2i_ct<1u, 66u, 2u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<0u, 3u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<3u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(23u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 3u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<3u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(23u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 3u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 3u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<0u, 0u, 3u, 23u>(); + ctx.execute_vfpu_vi2f_ct<1u, 1u, 3u, 23u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0044.cpp b/profiles/vcs/generated/generated_unit_0044.cpp index f70147b..54c0327 100644 --- a/profiles/vcs/generated/generated_unit_0044.cpp +++ b/profiles/vcs/generated/generated_unit_0044.cpp @@ -1360,26 +1360,10 @@ L_088B4414: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_vx2i(0u, 2u, 2u, 3u); - ctx.execute_vfpu_vx2i(1u, 66u, 2u, 3u); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<0u, 3u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<3u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(23u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 3u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<3u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(23u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 3u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 3u>(vfpu_d); } + ctx.execute_vfpu_vx2i_ct<0u, 2u, 2u, 3u>(); + ctx.execute_vfpu_vx2i_ct<1u, 66u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<0u, 0u, 3u, 23u>(); + ctx.execute_vfpu_vi2f_ct<1u, 1u, 3u, 23u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1641,28 +1625,7 @@ L_088B45D8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1930,32 +1893,8 @@ L_088B4738: ctx.write_vfpu_vector_with_destination_prefix_ct<39u, 3u>(vfpu_value); } { float vfpu_value[4]{}; ctx.write_vfpu_vector_with_destination_prefix_ct<35u, 3u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<19u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 7u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<7u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<7u, 36u, 19u, 3u, 3u>(); + ctx.execute_vfpu_unary_ct<7u, 7u, 3u, 2u>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<99u, 1u>(vfpu_value); } { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; @@ -2000,55 +1939,13 @@ L_088B4738: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<29u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<104u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<117u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<104u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<118u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<104u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<116u, 1u>(vfpu_d); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 44u, 3u); - ctx.read_vfpu_vector_ct<8u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 20u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<20u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<15u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<20u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<55u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<16u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<29u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<55u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<17u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<117u, 104u, 1u, 0u>(); + ctx.execute_vfpu_unary_ct<118u, 104u, 1u, 0u>(); + ctx.execute_vfpu_unary_ct<116u, 104u, 1u, 0u>(); + ctx.execute_vfpu_vtfm_ct<20u, 44u, 8u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<20u, 20u, 15u, 3u, 0u>(); + ctx.execute_vfpu_vec3_ct<16u, 28u, 55u, 3u, 1u>(); + ctx.execute_vfpu_vec3_ct<17u, 29u, 55u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<20u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2074,38 +1971,11 @@ L_088B47E4: L_088B47F0: ctx.gpr[2] = (0u + static_cast(1)); ctx.gpr[2] = (ctx.gpr[2] & 255u); - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 0u, 3u); - ctx.read_vfpu_vector_ct<3u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 35u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<35u, 0u, 3u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.write_vfpu_vector_with_destination_prefix_ct<3u, 3u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<35u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<35u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<19u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<39u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<35u, 35u, 3u, 2u>(); + ctx.execute_vfpu_unary_ct<39u, 19u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.write_vfpu_vector_with_destination_prefix_ct<7u, 3u>(vfpu_value); } { float vfpu_s[16]{}, vfpu_t[16]{}, vfpu_d[16]{}; @@ -2636,28 +2506,7 @@ L_088B4C7C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2955,28 +2804,7 @@ L_088B4EE4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7838,10 +7666,7 @@ L_088B77C8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); diff --git a/profiles/vcs/generated/generated_unit_0045.cpp b/profiles/vcs/generated/generated_unit_0045.cpp index 5275c1d..ece68c0 100644 --- a/profiles/vcs/generated/generated_unit_0045.cpp +++ b/profiles/vcs/generated/generated_unit_0045.cpp @@ -1349,7 +1349,7 @@ L_088B8344: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1381,10 +1381,7 @@ L_088B8344: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -1413,7 +1410,7 @@ L_088B8344: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0047.cpp b/profiles/vcs/generated/generated_unit_0047.cpp index 2cbd20c..669638b 100644 --- a/profiles/vcs/generated/generated_unit_0047.cpp +++ b/profiles/vcs/generated/generated_unit_0047.cpp @@ -4968,11 +4968,7 @@ L_088C2C20: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[30] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0048.cpp b/profiles/vcs/generated/generated_unit_0048.cpp index 07cff54..f0e5173 100644 --- a/profiles/vcs/generated/generated_unit_0048.cpp +++ b/profiles/vcs/generated/generated_unit_0048.cpp @@ -4496,15 +4496,8 @@ L_088C5CD0: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; diff --git a/profiles/vcs/generated/generated_unit_0049.cpp b/profiles/vcs/generated/generated_unit_0049.cpp index a2df66b..2c7a4eb 100644 --- a/profiles/vcs/generated/generated_unit_0049.cpp +++ b/profiles/vcs/generated/generated_unit_0049.cpp @@ -5904,11 +5904,7 @@ L_088CA7B0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8805,33 +8801,8 @@ L_088CBD30: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(160)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); diff --git a/profiles/vcs/generated/generated_unit_0051.cpp b/profiles/vcs/generated/generated_unit_0051.cpp index 93b3989..1b40635 100644 --- a/profiles/vcs/generated/generated_unit_0051.cpp +++ b/profiles/vcs/generated/generated_unit_0051.cpp @@ -4473,10 +4473,7 @@ L_088D1FC4: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[4]); { const bool branch_taken = static_cast(ctx.gpr[19]) <= 0; @@ -4897,10 +4894,7 @@ L_088D2314: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16256u << 16u); diff --git a/profiles/vcs/generated/generated_unit_0054.cpp b/profiles/vcs/generated/generated_unit_0054.cpp index 23de2f1..503e29b 100644 --- a/profiles/vcs/generated/generated_unit_0054.cpp +++ b/profiles/vcs/generated/generated_unit_0054.cpp @@ -4198,15 +4198,8 @@ L_088DDF90: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -4221,15 +4214,8 @@ L_088DDFB4: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -4538,10 +4524,7 @@ L_088DE1EC: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -5186,15 +5169,8 @@ L_088DE9BC: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); ctx.fpr[12] = ctx.fpr[22] - ctx.fpr[14]; @@ -5244,15 +5220,8 @@ L_088DEA38: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[15] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[15]; const float ft = ctx.fpr[13]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[20] = std::bit_cast(0x7FC00000u); else ctx.fpr[20] = fs * ft; } @@ -5311,10 +5280,7 @@ L_088DEAB8: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); diff --git a/profiles/vcs/generated/generated_unit_0056.cpp b/profiles/vcs/generated/generated_unit_0056.cpp index 0fb3605..a2af0bc 100644 --- a/profiles/vcs/generated/generated_unit_0056.cpp +++ b/profiles/vcs/generated/generated_unit_0056.cpp @@ -1030,11 +1030,7 @@ L_088E4000: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[21] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); @@ -1054,10 +1050,7 @@ L_088E4000: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -1101,11 +1094,7 @@ L_088E4000: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[19] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); @@ -1202,11 +1191,7 @@ L_088E4110: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2163,11 +2148,7 @@ L_088E48AC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2891,11 +2872,7 @@ L_088E4E34: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2910,10 +2887,7 @@ L_088E4E34: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[22] = std::bit_cast(ctx.gpr[5]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[22] < ctx.fpr[24])) ? 0x00800000u : 0u); @@ -4057,11 +4031,7 @@ L_088E5670: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[16] = (ctx.gpr[29] + static_cast(96)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); @@ -4080,10 +4050,7 @@ L_088E5670: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -5420,33 +5387,8 @@ L_088E6090: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5769,33 +5711,8 @@ L_088E62B0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5876,33 +5793,8 @@ L_088E62F0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6654,10 +6546,7 @@ L_088E68E4: L_088E68EC: ctx.gpr[6] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[6]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 17u>(); ctx.gpr[6] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[6]); ctx.fpr[14] = std::bit_cast(ctx.gpr[6]); @@ -8698,11 +8587,7 @@ L_088E78F0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -8732,10 +8617,7 @@ L_088E78F0: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[6] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[6]); ctx.gpr[6] = (16332u << 16u); @@ -9368,11 +9250,7 @@ L_088E7D98: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -9397,11 +9275,7 @@ L_088E7D98: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); diff --git a/profiles/vcs/generated/generated_unit_0057.cpp b/profiles/vcs/generated/generated_unit_0057.cpp index 18a95df..587ffbe 100644 --- a/profiles/vcs/generated/generated_unit_0057.cpp +++ b/profiles/vcs/generated/generated_unit_0057.cpp @@ -4038,28 +4038,7 @@ L_088E96CC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -4198,11 +4177,7 @@ L_088E97D0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[22] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4990,11 +4965,7 @@ L_088E9E0C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(208)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5013,10 +4984,7 @@ L_088E9E0C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -5253,11 +5221,7 @@ L_088E9FAC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[17] = (ctx.gpr[29] + static_cast(272)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); @@ -5358,11 +5322,7 @@ L_088EA018: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(240)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5386,11 +5346,7 @@ L_088EA018: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5564,11 +5520,7 @@ L_088EA154: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5729,11 +5681,7 @@ L_088EA2CC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(448)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -6388,28 +6336,7 @@ L_088EA804: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7066,11 +6993,7 @@ L_088EAD5C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[16] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); @@ -7089,10 +7012,7 @@ L_088EAD5C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7202,11 +7122,7 @@ L_088EAE48: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(260), ctx.gpr[18]); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[19] = (ctx.gpr[29] + static_cast(96)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); @@ -7225,10 +7141,7 @@ L_088EAE48: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); diff --git a/profiles/vcs/generated/generated_unit_0058.cpp b/profiles/vcs/generated/generated_unit_0058.cpp index dc7d229..77924d8 100644 --- a/profiles/vcs/generated/generated_unit_0058.cpp +++ b/profiles/vcs/generated/generated_unit_0058.cpp @@ -1912,10 +1912,7 @@ L_088EC49C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16640u << 16u); @@ -1939,10 +1936,7 @@ L_088EC4C4: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (16704u << 16u); @@ -1959,10 +1953,7 @@ L_088EC4C4: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16768u << 16u); @@ -2335,11 +2326,7 @@ L_088EC72C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[16] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); @@ -2415,10 +2402,7 @@ L_088EC798: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16800u << 16u); @@ -2426,10 +2410,7 @@ L_088EC798: { const float fs = ctx.fpr[12]; const float ft = ctx.fpr[13]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[12] = std::bit_cast(0x7FC00000u); else ctx.fpr[12] = fs * ft; } ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<1u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<1u, 1u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -4565,11 +4546,7 @@ L_088ED6F8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4756,11 +4733,7 @@ L_088ED820: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4849,11 +4822,7 @@ L_088ED884: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5576,11 +5545,7 @@ L_088EDD08: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5669,11 +5634,7 @@ L_088EDD6C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(144)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5824,11 +5785,7 @@ L_088EDE70: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(192)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5937,11 +5894,7 @@ L_088EDEF4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(224)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6698,11 +6651,7 @@ L_088EE428: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[19] = (ctx.gpr[29] + static_cast(288)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); @@ -6751,11 +6700,7 @@ L_088EE494: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7050,11 +6995,7 @@ L_088EE6DC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[22] = (ctx.gpr[29] + static_cast(368)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[22] + static_cast(0); @@ -7103,11 +7044,7 @@ L_088EE748: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7600,11 +7537,7 @@ L_088EEAF8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(1108), ctx.gpr[4]); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); @@ -7710,10 +7643,7 @@ L_088EEBA8: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (17096u << 16u); @@ -8236,11 +8166,7 @@ L_088EEF64: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(160)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8281,11 +8207,7 @@ L_088EEF64: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(144)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8347,11 +8269,7 @@ L_088EEFE0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(224)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8392,11 +8310,7 @@ L_088EEFE0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(208)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8475,11 +8389,7 @@ L_088EF098: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(272)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8508,10 +8418,7 @@ L_088EF0EC: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[12] = ctx.fpr[13] / ctx.fpr[12]; @@ -8593,10 +8500,7 @@ L_088EF180: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -8807,11 +8711,7 @@ L_088EF2D8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(336)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8852,11 +8752,7 @@ L_088EF2D8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(320)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -8910,11 +8806,7 @@ L_088EF340: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(400)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8955,11 +8847,7 @@ L_088EF340: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(384)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -9027,11 +8915,7 @@ L_088EF3E4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[22] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9397,10 +9281,7 @@ L_088EF64C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -9792,10 +9673,7 @@ L_088EF918: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -9826,10 +9704,7 @@ L_088EF918: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[12] <= ctx.fpr[20])) ? 0x00800000u : 0u); @@ -9943,11 +9818,7 @@ L_088EF9A4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(576)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); diff --git a/profiles/vcs/generated/generated_unit_0059.cpp b/profiles/vcs/generated/generated_unit_0059.cpp index 5fdfa90..39cee20 100644 --- a/profiles/vcs/generated/generated_unit_0059.cpp +++ b/profiles/vcs/generated/generated_unit_0059.cpp @@ -1248,19 +1248,9 @@ L_088F0210: ctx.gpr[5] = (std::bit_cast(ctx.fpr[13])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[26] = std::bit_cast(ctx.gpr[4]); ctx.fpr[12] = ctx.fpr[30] - ctx.fpr[26]; @@ -1681,19 +1671,9 @@ L_088F0594: ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[22] = std::bit_cast(ctx.gpr[4]); ctx.fpr[12] = ctx.fpr[30] - ctx.fpr[22]; @@ -4190,19 +4170,9 @@ L_088F1918: ctx.gpr[5] = (std::bit_cast(ctx.fpr[13])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[13] = ctx.fpr[30] - ctx.fpr[12]; @@ -4389,19 +4359,9 @@ L_088F1AAC: ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[13] = ctx.fpr[30] - ctx.fpr[12]; @@ -5287,11 +5247,7 @@ L_088F2204: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[20] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); @@ -5333,28 +5289,7 @@ L_088F2204: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -5401,28 +5336,7 @@ L_088F2250: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<8u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 4u, 3u); - ctx.read_vfpu_vector_ct<8u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 4u, 8u, 3u, 3u>(); ctx.gpr[19] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); @@ -5460,11 +5374,7 @@ L_088F2250: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[21] = (ctx.gpr[29] + static_cast(96)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); @@ -5506,28 +5416,7 @@ L_088F2250: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<8u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 4u, 3u); - ctx.read_vfpu_vector_ct<8u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 4u, 8u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5569,11 +5458,7 @@ L_088F22B8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5749,33 +5634,8 @@ L_088F239C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6157,11 +6017,7 @@ L_088F2684: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -6269,11 +6125,7 @@ L_088F271C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6359,11 +6211,7 @@ L_088F271C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(96)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6995,11 +6843,7 @@ L_088F2C70: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7014,10 +6858,7 @@ L_088F2C70: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[8] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[8]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[13] < ctx.fpr[12])) ? 0x00800000u : 0u); @@ -7386,11 +7227,7 @@ L_088F2EB0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -7406,10 +7243,7 @@ L_088F2EB0: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (16800u << 16u); @@ -7453,15 +7287,8 @@ L_088F2F5C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[15] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (15948u << 16u); @@ -7556,15 +7383,8 @@ L_088F30A0: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); @@ -7572,15 +7392,8 @@ L_088F30A0: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); { const std::uint32_t aot_run_words[3]{std::bit_cast(ctx.fpr[13]), std::bit_cast(ctx.fpr[14]), std::bit_cast(ctx.fpr[22])}; @@ -7646,11 +7459,7 @@ L_088F3134: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -7666,10 +7475,7 @@ L_088F3134: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (16736u << 16u); @@ -7713,15 +7519,8 @@ L_088F31E0: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[15] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16128u << 16u); @@ -7812,15 +7611,8 @@ L_088F3314: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); @@ -7828,15 +7620,8 @@ L_088F3314: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); { const std::uint32_t aot_run_words[3]{std::bit_cast(ctx.fpr[13]), std::bit_cast(ctx.fpr[14]), std::bit_cast(ctx.fpr[22])}; @@ -7891,11 +7676,7 @@ L_088F33C8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7910,10 +7691,7 @@ L_088F33C8: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[8] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[8]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[13] < ctx.fpr[12])) ? 0x00800000u : 0u); @@ -8351,15 +8129,8 @@ L_088F3754: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[13]; const float ft = ctx.fpr[22]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[12] = std::bit_cast(0x7FC00000u); else ctx.fpr[12] = fs * ft; } @@ -8380,15 +8151,8 @@ L_088F3798: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[13]; const float ft = ctx.fpr[22]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[13] = std::bit_cast(0x7FC00000u); else ctx.fpr[13] = fs * ft; } @@ -8586,11 +8350,7 @@ L_088F392C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8605,10 +8365,7 @@ L_088F392C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[12] < ctx.fpr[20])) ? 0x00800000u : 0u); @@ -9544,11 +9301,7 @@ L_088F3F2C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); diff --git a/profiles/vcs/generated/generated_unit_0060.cpp b/profiles/vcs/generated/generated_unit_0060.cpp index 08eff26..f287cb5 100644 --- a/profiles/vcs/generated/generated_unit_0060.cpp +++ b/profiles/vcs/generated/generated_unit_0060.cpp @@ -6397,15 +6397,8 @@ L_088F69A0: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[5]); { const float fs = ctx.fpr[14]; const float ft = ctx.fpr[12]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[14] = std::bit_cast(0x7FC00000u); else ctx.fpr[14] = fs * ft; } @@ -6414,15 +6407,8 @@ L_088F69A0: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[15] = std::bit_cast(ctx.gpr[5]); { const float fs = ctx.fpr[15]; const float ft = ctx.fpr[12]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[13] = std::bit_cast(0x7FC00000u); else ctx.fpr[13] = fs * ft; } @@ -9045,15 +9031,8 @@ L_088F7EC0: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[13]; const float ft = ctx.fpr[20]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[12] = std::bit_cast(0x7FC00000u); else ctx.fpr[12] = fs * ft; } @@ -9074,15 +9053,8 @@ L_088F7F04: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[13]; const float ft = ctx.fpr[20]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[13] = std::bit_cast(0x7FC00000u); else ctx.fpr[13] = fs * ft; } @@ -9161,11 +9133,7 @@ L_088F7F6C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[30] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9188,11 +9156,7 @@ L_088F7F6C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[23] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9207,10 +9171,7 @@ L_088F7F6C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); { const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); @@ -9231,11 +9192,7 @@ L_088F7F6C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[22] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9250,10 +9207,7 @@ L_088F7F6C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.fpr[13] = ctx.fpr[13] + ctx.fpr[24]; diff --git a/profiles/vcs/generated/generated_unit_0061.cpp b/profiles/vcs/generated/generated_unit_0061.cpp index af25180..2cee075 100644 --- a/profiles/vcs/generated/generated_unit_0061.cpp +++ b/profiles/vcs/generated/generated_unit_0061.cpp @@ -1178,11 +1178,7 @@ L_088F8000: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8685,33 +8681,8 @@ L_088FB584: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9050,11 +9021,7 @@ L_088FB7D8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0062.cpp b/profiles/vcs/generated/generated_unit_0062.cpp index 7af4d54..edc862e 100644 --- a/profiles/vcs/generated/generated_unit_0062.cpp +++ b/profiles/vcs/generated/generated_unit_0062.cpp @@ -6524,11 +6524,7 @@ L_088FEEE8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[21] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); @@ -6544,10 +6540,7 @@ L_088FEEE8: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[6] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[6]); ctx.fpr[12] = ctx.fpr[13] / ctx.fpr[12]; @@ -6574,10 +6567,7 @@ L_088FEF80: ctx.fpr[13] = static_cast(static_cast(std::bit_cast(ctx.fpr[13]))); ctx.gpr[4] = (std::bit_cast(ctx.fpr[13])); ctx.set_vfpu_scalar_bits_ct<1u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<1u, 1u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -6642,11 +6632,7 @@ L_088FEFC0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7583,15 +7569,8 @@ L_088FF990: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[16] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[3] + static_cast(0), std::bit_cast(ctx.fpr[16])); @@ -7600,15 +7579,8 @@ L_088FF990: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[16] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[2] + static_cast(0), std::bit_cast(ctx.fpr[16])); diff --git a/profiles/vcs/generated/generated_unit_0063.cpp b/profiles/vcs/generated/generated_unit_0063.cpp index c74a87f..45862bc 100644 --- a/profiles/vcs/generated/generated_unit_0063.cpp +++ b/profiles/vcs/generated/generated_unit_0063.cpp @@ -1764,11 +1764,7 @@ L_089004F0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -2633,11 +2629,7 @@ L_08900BB0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(96)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3046,19 +3038,9 @@ L_08900F68: ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(20), std::bit_cast(ctx.fpr[13])); @@ -3069,19 +3051,9 @@ L_08900F68: ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(24), std::bit_cast(ctx.fpr[13])); @@ -3092,19 +3064,9 @@ L_08900F68: ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(28), std::bit_cast(ctx.fpr[13])); @@ -3115,19 +3077,9 @@ L_08900F68: ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(32), std::bit_cast(ctx.fpr[13])); @@ -3138,19 +3090,9 @@ L_08900F68: ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(36), std::bit_cast(ctx.fpr[13])); @@ -3161,19 +3103,9 @@ L_08900F68: ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(40), std::bit_cast(ctx.fpr[12])); @@ -3395,10 +3327,7 @@ L_08901218: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16384u << 16u); @@ -3448,11 +3377,7 @@ L_089012FC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(144)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3468,10 +3393,7 @@ L_089012FC: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (17008u << 16u); @@ -3523,10 +3445,7 @@ L_08901378: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[13] = std::bit_cast(0u); @@ -3789,11 +3708,7 @@ L_08901534: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(160)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3809,10 +3724,7 @@ L_08901534: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); goto L_08901578; @@ -4133,10 +4045,7 @@ L_089017B4: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (ctx.gpr[29] + static_cast(112)); @@ -4150,10 +4059,7 @@ L_089017B4: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[13] <= ctx.fpr[14])) ? 0x00800000u : 0u); @@ -4177,10 +4083,7 @@ L_089017F4: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); goto L_0890180C; @@ -6346,11 +6249,7 @@ L_08902968: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[23] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6368,10 +6267,7 @@ L_08902968: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -6389,10 +6285,7 @@ L_08902968: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (14979u << 16u); @@ -6964,19 +6857,9 @@ L_08902E6C: { const float vfpu_constant = std::bit_cast(0x3FC90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 64u, 32u, 1u, 2u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[15] < ctx.fpr[26])) ? 0x00800000u : 0u); @@ -7003,10 +6886,7 @@ L_08902EA4: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[15] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[15]; const float ft = ctx.fpr[12]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[12] = std::bit_cast(0x7FC00000u); else ctx.fpr[12] = fs * ft; } @@ -7094,11 +6974,7 @@ L_08902F68: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7137,19 +7013,9 @@ L_08902F9C: { const float vfpu_constant = std::bit_cast(0x3FC90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 64u, 32u, 1u, 2u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(132), std::bit_cast(ctx.fpr[12])); @@ -8427,11 +8293,7 @@ L_08903B08: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[22] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8585,11 +8447,7 @@ L_08903C74: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0064.cpp b/profiles/vcs/generated/generated_unit_0064.cpp index 620be67..21d4906 100644 --- a/profiles/vcs/generated/generated_unit_0064.cpp +++ b/profiles/vcs/generated/generated_unit_0064.cpp @@ -2635,33 +2635,8 @@ L_08904B64: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2719,11 +2694,7 @@ L_08904BA8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2753,11 +2724,7 @@ L_08904BC0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4418,19 +4385,9 @@ L_08905964: ctx.gpr[5] = (std::bit_cast(ctx.fpr[13])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -4474,10 +4431,7 @@ L_089059BC: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -4976,33 +4930,8 @@ L_08905CEC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5144,30 +5073,10 @@ L_08905E50: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -5186,15 +5095,7 @@ L_08905E94: ctx.gpr[9] = (static_cast(static_cast(static_cast(aot_mem.aot_direct_load16(ctx.gpr[4] + static_cast(2)))))); ctx.set_vfpu_scalar_bits_ct<1u>(ctx.gpr[8]); ctx.set_vfpu_scalar_bits_ct<33u>(ctx.gpr[9]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); aot_mem.aot_direct_store32(ctx.gpr[5] + static_cast(0), ctx.vfpu_scalar_bits_ct<2u>()); aot_mem.aot_direct_store32(ctx.gpr[5] + static_cast(4), ctx.vfpu_scalar_bits_ct<34u>()); jump_target = ctx.gpr[31]; @@ -5286,11 +5187,7 @@ L_08905F10: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -5342,11 +5239,7 @@ L_08905F10: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5370,11 +5263,7 @@ L_08905F10: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6345,11 +6234,7 @@ L_08906638: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7021,11 +6906,7 @@ L_08906B24: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[7] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); @@ -7054,10 +6935,7 @@ L_08906B24: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[7] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[7]); ctx.gpr[10] = (ctx.gpr[29] + static_cast(32)); @@ -7100,11 +6978,7 @@ L_08906BAC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[9] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7132,10 +7006,7 @@ L_08906BAC: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[7] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[7]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[13] < ctx.fpr[12])) ? 0x00800000u : 0u); @@ -7186,11 +7057,7 @@ L_08906C08: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[10] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7218,10 +7085,7 @@ L_08906C08: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[7] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[7]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[12] < ctx.fpr[13])) ? 0x00800000u : 0u); @@ -8279,11 +8143,7 @@ L_08907250: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8799,15 +8659,8 @@ L_0890759C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); ctx.fpr[14] = std::bit_cast(std::bit_cast(ctx.fpr[14]) ^ 0x80000000u); @@ -8821,15 +8674,8 @@ L_0890759C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(28), std::bit_cast(ctx.fpr[13])); @@ -8865,10 +8711,7 @@ L_08907660: L_0890767C: ctx.gpr[5] = (std::bit_cast(ctx.fpr[15])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 17u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[17] = std::bit_cast(ctx.gpr[5]); ctx.fpr[18] = std::bit_cast(ctx.gpr[5]); diff --git a/profiles/vcs/generated/generated_unit_0065.cpp b/profiles/vcs/generated/generated_unit_0065.cpp index 28675fd..fabbea5 100644 --- a/profiles/vcs/generated/generated_unit_0065.cpp +++ b/profiles/vcs/generated/generated_unit_0065.cpp @@ -7005,30 +7005,10 @@ L_0890A9E0: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -7575,11 +7555,7 @@ L_0890ADEC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8586,11 +8562,7 @@ L_0890B570: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8619,10 +8591,7 @@ L_0890B570: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16448u << 16u); diff --git a/profiles/vcs/generated/generated_unit_0066.cpp b/profiles/vcs/generated/generated_unit_0066.cpp index 9c2d7df..928a709 100644 --- a/profiles/vcs/generated/generated_unit_0066.cpp +++ b/profiles/vcs/generated/generated_unit_0066.cpp @@ -4043,11 +4043,7 @@ L_0890D55C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4075,10 +4071,7 @@ L_0890D55C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[12] < ctx.fpr[20])) ? 0x00800000u : 0u); @@ -4880,11 +4873,7 @@ L_0890DB5C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5547,10 +5536,7 @@ L_0890E008: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[6] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[6]); ctx.gpr[6] = (16128u << 16u); @@ -5592,10 +5578,7 @@ L_0890E098: aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(56), std::bit_cast(ctx.fpr[12])); ctx.gpr[5] = (std::bit_cast(ctx.fpr[20])); ctx.set_vfpu_scalar_bits_ct<1u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<1u, 1u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -5778,11 +5761,7 @@ L_0890E1D4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7123,11 +7102,7 @@ L_0890EAD0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7829,11 +7804,7 @@ L_0890F090: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7957,11 +7928,7 @@ L_0890F188: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8461,10 +8428,7 @@ L_0890F530: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -8510,11 +8474,7 @@ L_0890F554: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(96)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -8548,10 +8508,7 @@ L_0890F554: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -8605,11 +8562,7 @@ L_0890F5A0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8827,11 +8780,7 @@ L_0890F748: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8886,28 +8835,7 @@ L_0890F748: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<8u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<4u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<8u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } + ctx.execute_vfpu_vtfm_ct<0u, 4u, 8u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9034,28 +8962,7 @@ L_0890F7E0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9080,11 +8987,7 @@ L_0890F7E0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9157,28 +9060,7 @@ L_0890F850: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -9268,11 +9150,7 @@ L_0890F8AC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -9355,28 +9233,7 @@ L_0890F924: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -9640,11 +9497,7 @@ L_0890FAF8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9659,10 +9512,7 @@ L_0890FAF8: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16256u << 16u); @@ -10150,15 +10000,7 @@ L_0890FEA0: ctx.gpr[9] = (static_cast(static_cast(static_cast(aot_mem.aot_direct_load16(ctx.gpr[9] + static_cast(2)))))); ctx.set_vfpu_scalar_bits_ct<1u>(ctx.gpr[8]); ctx.set_vfpu_scalar_bits_ct<33u>(ctx.gpr[9]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); aot_mem.aot_direct_store32(ctx.gpr[16] + static_cast(0), ctx.vfpu_scalar_bits_ct<2u>()); aot_mem.aot_direct_store32(ctx.gpr[16] + static_cast(4), ctx.vfpu_scalar_bits_ct<34u>()); goto L_0890FEC0; @@ -10240,30 +10082,10 @@ L_0890FF04: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } diff --git a/profiles/vcs/generated/generated_unit_0067.cpp b/profiles/vcs/generated/generated_unit_0067.cpp index b23731b..c063e3f 100644 --- a/profiles/vcs/generated/generated_unit_0067.cpp +++ b/profiles/vcs/generated/generated_unit_0067.cpp @@ -4279,11 +4279,7 @@ L_089115A8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4301,10 +4297,7 @@ L_089115A8: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -5743,10 +5736,7 @@ L_08912018: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -6139,10 +6129,7 @@ L_08912300: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7241,10 +7228,7 @@ L_08912B08: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (15267u << 16u); @@ -7582,11 +7566,7 @@ L_08912DC4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8399,11 +8379,7 @@ L_08913318: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); diff --git a/profiles/vcs/generated/generated_unit_0068.cpp b/profiles/vcs/generated/generated_unit_0068.cpp index 6544553..dbacd56 100644 --- a/profiles/vcs/generated/generated_unit_0068.cpp +++ b/profiles/vcs/generated/generated_unit_0068.cpp @@ -1546,33 +1546,8 @@ L_089146C8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2756,10 +2731,7 @@ L_0891503C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); { const bool branch_taken = ctx.gpr[20] == 0u; @@ -4707,11 +4679,7 @@ L_08915FF0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(448)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -6831,11 +6799,7 @@ L_08917000: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(832)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -6948,11 +6912,7 @@ L_089170B8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7150,11 +7110,7 @@ L_089171F4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[22] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0069.cpp b/profiles/vcs/generated/generated_unit_0069.cpp index cc09ef8..7178388 100644 --- a/profiles/vcs/generated/generated_unit_0069.cpp +++ b/profiles/vcs/generated/generated_unit_0069.cpp @@ -1413,11 +1413,7 @@ L_08918420: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -3063,10 +3059,7 @@ L_0891906C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[12] <= ctx.fpr[20])) ? 0x00800000u : 0u); @@ -3261,11 +3254,7 @@ L_089191A8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3518,11 +3507,7 @@ L_08919350: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[22] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3537,10 +3522,7 @@ L_08919350: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[12] < ctx.fpr[20])) ? 0x00800000u : 0u); @@ -4045,15 +4027,8 @@ L_08919778: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.fpr[14] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[16] + static_cast(792))); @@ -4063,15 +4038,8 @@ L_08919778: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[15] = std::bit_cast(ctx.gpr[4]); ctx.fpr[16] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[16] + static_cast(796))); @@ -4084,15 +4052,8 @@ L_08919778: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[13]; const float ft = ctx.fpr[16]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[13] = std::bit_cast(0x7FC00000u); else ctx.fpr[13] = fs * ft; } @@ -4101,15 +4062,8 @@ L_08919778: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[15] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[15]; const float ft = ctx.fpr[14]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[14] = std::bit_cast(0x7FC00000u); else ctx.fpr[14] = fs * ft; } @@ -4464,15 +4418,8 @@ L_08919AE8: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[12]; const float ft = ctx.fpr[22]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[12] = std::bit_cast(0x7FC00000u); else ctx.fpr[12] = fs * ft; } @@ -4483,15 +4430,8 @@ L_08919AE8: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[12]; const float ft = ctx.fpr[22]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[12] = std::bit_cast(0x7FC00000u); else ctx.fpr[12] = fs * ft; } @@ -4916,7 +4856,7 @@ L_08919E48: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -4942,11 +4882,7 @@ L_08919E48: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5228,10 +5164,7 @@ L_0891A080: L_0891A088: ctx.gpr[4] = (std::bit_cast(ctx.fpr[13])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 17u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); ctx.fpr[15] = std::bit_cast(ctx.gpr[4]); @@ -5917,11 +5850,7 @@ L_0891A618: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6079,11 +6008,7 @@ L_0891A75C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6125,11 +6050,7 @@ L_0891A75C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(160)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6854,10 +6775,7 @@ L_0891AE2C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -6868,10 +6786,7 @@ L_0891AE2C: ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<1u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<1u, 1u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -7324,19 +7239,9 @@ L_0891B2B0: ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[20])); @@ -7408,19 +7313,9 @@ L_0891B368: ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[4]); goto L_0891B39C; @@ -7515,11 +7410,7 @@ L_0891B414: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(480)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); diff --git a/profiles/vcs/generated/generated_unit_0070.cpp b/profiles/vcs/generated/generated_unit_0070.cpp index 2dd0c7e..404df4f 100644 --- a/profiles/vcs/generated/generated_unit_0070.cpp +++ b/profiles/vcs/generated/generated_unit_0070.cpp @@ -2553,11 +2553,7 @@ L_0891CB24: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2631,11 +2627,7 @@ L_0891CB88: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3056,11 +3048,7 @@ L_0891CE50: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3446,30 +3434,10 @@ L_0891D0F4: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -3503,11 +3471,7 @@ L_0891D144: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3747,11 +3711,7 @@ L_0891D328: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3793,11 +3753,7 @@ L_0891D37C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3998,33 +3954,8 @@ L_0891D4A0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4098,11 +4029,7 @@ L_0891D514: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4120,10 +4047,7 @@ L_0891D514: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -4250,11 +4174,7 @@ L_0891D5C0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[19] = (ctx.gpr[29] + static_cast(656)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); @@ -4273,10 +4193,7 @@ L_0891D5C0: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -4620,11 +4537,7 @@ L_0891D758: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(720)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4748,11 +4661,7 @@ L_0891D7DC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(752)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4861,11 +4770,7 @@ L_0891D8D4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4944,11 +4849,7 @@ L_0891D950: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4964,10 +4865,7 @@ L_0891D950: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((!(std::isnan(ctx.fpr[12]) || std::isnan(ctx.fpr[20])) && ctx.fpr[12] == ctx.fpr[20])) ? 0x00800000u : 0u); @@ -4993,10 +4891,7 @@ L_0891D98C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -5112,11 +5007,7 @@ L_0891DA10: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[9] = (ctx.gpr[29] + static_cast(368)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[9] + static_cast(0); @@ -5135,10 +5026,7 @@ L_0891DA10: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -5300,11 +5188,7 @@ L_0891DACC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[11] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5431,11 +5315,7 @@ L_0891DB5C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[3] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5902,11 +5782,7 @@ L_0891DE74: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5925,10 +5801,7 @@ L_0891DE74: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -5988,11 +5861,7 @@ L_0891DE74: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6064,11 +5933,7 @@ L_0891DF30: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(1040)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6092,11 +5957,7 @@ L_0891DF30: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[30] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6246,11 +6107,7 @@ L_0891E064: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6427,11 +6284,7 @@ L_0891E170: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6450,10 +6303,7 @@ L_0891E170: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -6513,11 +6363,7 @@ L_0891E170: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6589,11 +6435,7 @@ L_0891E22C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(1168)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6617,11 +6459,7 @@ L_0891E22C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[30] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6771,11 +6609,7 @@ L_0891E360: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6927,11 +6761,7 @@ L_0891E43C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6946,10 +6776,7 @@ L_0891E43C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[11] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[11]); ctx.fpr[12] = ctx.fpr[12] + ctx.fpr[13]; @@ -6999,11 +6826,7 @@ L_0891E49C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7018,10 +6841,7 @@ L_0891E49C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[11] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[11]); ctx.fpr[13] = ctx.fpr[13] + ctx.fpr[14]; diff --git a/profiles/vcs/generated/generated_unit_0071.cpp b/profiles/vcs/generated/generated_unit_0071.cpp index 36d8f1f..a0e299b 100644 --- a/profiles/vcs/generated/generated_unit_0071.cpp +++ b/profiles/vcs/generated/generated_unit_0071.cpp @@ -1344,11 +1344,7 @@ L_08920090: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2858,11 +2854,7 @@ L_08920CB0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(240)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2872,10 +2864,7 @@ L_08920CB0: ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -2936,33 +2925,8 @@ L_08920CB0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(256)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -3000,11 +2964,7 @@ L_08920CB0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3131,11 +3091,7 @@ L_08920E20: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(320)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3154,10 +3110,7 @@ L_08920E20: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); diff --git a/profiles/vcs/generated/generated_unit_0072.cpp b/profiles/vcs/generated/generated_unit_0072.cpp index 1ec1b6f..90ae90f 100644 --- a/profiles/vcs/generated/generated_unit_0072.cpp +++ b/profiles/vcs/generated/generated_unit_0072.cpp @@ -2943,11 +2943,7 @@ L_08924B9C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5110,11 +5106,7 @@ L_08925AD0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(96)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5198,11 +5190,7 @@ L_08925B44: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5290,11 +5278,7 @@ L_08925BAC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5382,11 +5366,7 @@ L_08925C24: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5520,11 +5500,7 @@ L_08925D48: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(144)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5540,10 +5516,7 @@ L_08925D48: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (0u | 0u); @@ -5870,33 +5843,8 @@ L_08925F48: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(208)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -6128,11 +6076,7 @@ L_0892614C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6992,11 +6936,7 @@ L_089268B0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7049,11 +6989,7 @@ L_08926930: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7509,11 +7445,7 @@ L_08926C9C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7529,10 +7461,7 @@ L_08926C9C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16480u << 16u); diff --git a/profiles/vcs/generated/generated_unit_0073.cpp b/profiles/vcs/generated/generated_unit_0073.cpp index 8f6c6df..55550fd 100644 --- a/profiles/vcs/generated/generated_unit_0073.cpp +++ b/profiles/vcs/generated/generated_unit_0073.cpp @@ -1925,11 +1925,7 @@ L_08928628: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(96)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1970,11 +1966,7 @@ L_08928628: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2039,11 +2031,7 @@ L_089286B0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(160)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2084,11 +2072,7 @@ L_089286B0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(144)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2167,11 +2151,7 @@ L_08928768: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(208)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2199,10 +2179,7 @@ L_089287B8: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[12] = ctx.fpr[13] / ctx.fpr[12]; @@ -2284,10 +2261,7 @@ L_0892884C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -3404,28 +3378,7 @@ L_08928F4C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4285,11 +4238,7 @@ L_0892962C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4474,11 +4423,7 @@ L_08929750: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4663,11 +4608,7 @@ L_08929874: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(96)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4850,11 +4791,7 @@ L_08929990: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4973,11 +4910,7 @@ L_08929A4C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(192)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8010,11 +7943,7 @@ L_0892AF40: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[7] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); @@ -8111,11 +8040,7 @@ L_0892AFD4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -8139,11 +8064,7 @@ L_0892AFD4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8159,10 +8080,7 @@ L_0892AFD4: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16544u << 16u); @@ -8347,11 +8265,7 @@ L_0892B110: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8361,10 +8275,7 @@ L_0892B110: ctx.fpr[12] = std::bit_cast(ctx.gpr[7]); ctx.gpr[7] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[7]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -9675,11 +9586,7 @@ L_0892B9AC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(208)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); diff --git a/profiles/vcs/generated/generated_unit_0074.cpp b/profiles/vcs/generated/generated_unit_0074.cpp index b09a545..d614cf9 100644 --- a/profiles/vcs/generated/generated_unit_0074.cpp +++ b/profiles/vcs/generated/generated_unit_0074.cpp @@ -1115,11 +1115,7 @@ L_0892C178: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[19] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); @@ -1245,11 +1241,7 @@ L_0892C268: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -1607,15 +1599,8 @@ L_0892C514: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[5]); { const float fs = ctx.fpr[24]; const float ft = ctx.fpr[14]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[13] = std::bit_cast(0x7FC00000u); else ctx.fpr[13] = fs * ft; } @@ -1627,15 +1612,8 @@ L_0892C514: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[5]); { const float fs = ctx.fpr[24]; const float ft = ctx.fpr[13]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[12] = std::bit_cast(0x7FC00000u); else ctx.fpr[12] = fs * ft; } @@ -1676,15 +1654,8 @@ L_0892C5A8: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[24]; const float ft = ctx.fpr[12]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[12] = std::bit_cast(0x7FC00000u); else ctx.fpr[12] = fs * ft; } @@ -1695,15 +1666,8 @@ L_0892C5A8: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[24]; const float ft = ctx.fpr[12]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[12] = std::bit_cast(0x7FC00000u); else ctx.fpr[12] = fs * ft; } @@ -1722,15 +1686,8 @@ L_0892C5FC: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[24]; const float ft = ctx.fpr[13]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[12] = std::bit_cast(0x7FC00000u); else ctx.fpr[12] = fs * ft; } @@ -1742,15 +1699,8 @@ L_0892C5FC: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[24]; const float ft = ctx.fpr[13]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[12] = std::bit_cast(0x7FC00000u); else ctx.fpr[12] = fs * ft; } @@ -2694,11 +2644,7 @@ L_0892CD80: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2766,10 +2712,7 @@ L_0892CE04: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -2919,28 +2862,7 @@ L_0892CF14: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2965,11 +2887,7 @@ L_0892CF14: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(176)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3673,11 +3591,7 @@ L_0892D3D4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3687,10 +3601,7 @@ L_0892D3D4: ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -4110,33 +4021,8 @@ L_0892D734: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(224)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -4356,33 +4242,8 @@ L_0892D894: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(240)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -4425,11 +4286,7 @@ L_0892D8C4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -4876,33 +4733,8 @@ L_0892DB44: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(336)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5084,33 +4916,8 @@ L_0892DC64: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(336)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5292,33 +5099,8 @@ L_0892DD84: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(336)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5500,33 +5282,8 @@ L_0892DEA4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(336)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5725,33 +5482,8 @@ L_0892DFDC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[7] = (ctx.gpr[29] + static_cast(400)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); @@ -5823,33 +5555,8 @@ L_0892DFDC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6025,33 +5732,8 @@ L_0892E158: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[7] = (ctx.gpr[29] + static_cast(416)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); @@ -6123,33 +5805,8 @@ L_0892E158: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8220,33 +7877,8 @@ L_0892EEE4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -9986,11 +9618,7 @@ L_0892FCA4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -10032,11 +9660,7 @@ L_0892FCA4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(144)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -10077,11 +9701,7 @@ L_0892FCA4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -10144,11 +9764,7 @@ L_0892FE14: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(240)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -10187,11 +9803,7 @@ L_0892FE14: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -10251,11 +9863,7 @@ L_0892FEA4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(176)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -10271,10 +9879,7 @@ L_0892FEA4: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[17] + static_cast(4), std::bit_cast(ctx.fpr[12])); @@ -10291,10 +9896,7 @@ L_0892FEA4: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -10321,10 +9923,7 @@ L_0892FEA4: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -10396,7 +9995,7 @@ L_0892FEA4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0075.cpp b/profiles/vcs/generated/generated_unit_0075.cpp index 8cf1e32..d8f451f 100644 --- a/profiles/vcs/generated/generated_unit_0075.cpp +++ b/profiles/vcs/generated/generated_unit_0075.cpp @@ -1049,11 +1049,7 @@ L_0893008C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -1114,11 +1110,7 @@ L_089300EC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(336)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1180,11 +1172,7 @@ L_0893012C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(384)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -1242,11 +1230,7 @@ L_0893012C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1284,11 +1268,7 @@ L_0893012C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1336,11 +1316,7 @@ L_089301EC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(432)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -1403,11 +1379,7 @@ L_0893022C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(480)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1672,10 +1644,7 @@ L_08930430: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -2481,10 +2450,7 @@ L_089309DC: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.fpr[14] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[28] + static_cast(7676))); @@ -2817,11 +2783,7 @@ L_08930CB0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3130,11 +3092,7 @@ L_08930F0C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(160)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3371,10 +3329,7 @@ L_089310DC: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[6] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[6]); ctx.gpr[6] = (16168u << 16u); @@ -4124,10 +4079,7 @@ L_089316BC: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -5572,11 +5524,7 @@ L_0893216C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5591,10 +5539,7 @@ L_0893216C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[31] = (0x08932198u); @@ -5723,11 +5668,7 @@ L_089322D0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -7541,11 +7482,7 @@ L_0893310C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[18] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); @@ -7577,10 +7514,7 @@ L_0893310C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7759,11 +7693,7 @@ L_089332BC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8536,11 +8466,7 @@ L_089339F4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -8572,10 +8498,7 @@ L_089339F4: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -8738,11 +8661,7 @@ L_08933B70: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); diff --git a/profiles/vcs/generated/generated_unit_0076.cpp b/profiles/vcs/generated/generated_unit_0076.cpp index 462774a..004cfb8 100644 --- a/profiles/vcs/generated/generated_unit_0076.cpp +++ b/profiles/vcs/generated/generated_unit_0076.cpp @@ -2677,17 +2677,9 @@ L_08934A78: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - ctx.execute_vfpu_vrot(1u, 64u, 2u, 4u); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<33u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] / vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_vrot_ct<1u, 64u, 2u, 4u>(); + ctx.execute_vfpu_vec3_ct<0u, 33u, 1u, 1u, 3u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (aot_mem.aot_direct_load32(ctx.gpr[28] + static_cast(8896))); @@ -2826,17 +2818,9 @@ L_08934B9C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - ctx.execute_vfpu_vrot(1u, 64u, 2u, 4u); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<33u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] / vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_vrot_ct<1u, 64u, 2u, 4u>(); + ctx.execute_vfpu_vec3_ct<0u, 33u, 1u, 1u, 3u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[4]); ctx.gpr[31] = (0x08934C44u); @@ -6545,17 +6529,9 @@ L_08936770: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - ctx.execute_vfpu_vrot(1u, 64u, 2u, 4u); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<33u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] / vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_vrot_ct<1u, 64u, 2u, 4u>(); + ctx.execute_vfpu_vec3_ct<0u, 33u, 1u, 1u, 3u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (aot_mem.aot_direct_load32(ctx.gpr[28] + static_cast(8896))); @@ -9745,17 +9721,9 @@ L_08937E6C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - ctx.execute_vfpu_vrot(1u, 64u, 2u, 4u); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<33u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] / vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_vrot_ct<1u, 64u, 2u, 4u>(); + ctx.execute_vfpu_vec3_ct<0u, 33u, 1u, 1u, 3u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (aot_mem.aot_direct_load32(ctx.gpr[28] + static_cast(8896))); diff --git a/profiles/vcs/generated/generated_unit_0077.cpp b/profiles/vcs/generated/generated_unit_0077.cpp index 978b792..c78ffc3 100644 --- a/profiles/vcs/generated/generated_unit_0077.cpp +++ b/profiles/vcs/generated/generated_unit_0077.cpp @@ -1882,15 +1882,8 @@ L_08938668: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[13]; const float ft = ctx.fpr[22]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[12] = std::bit_cast(0x7FC00000u); else ctx.fpr[12] = fs * ft; } @@ -1911,15 +1904,8 @@ L_089386AC: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[13]; const float ft = ctx.fpr[22]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[13] = std::bit_cast(0x7FC00000u); else ctx.fpr[13] = fs * ft; } diff --git a/profiles/vcs/generated/generated_unit_0078.cpp b/profiles/vcs/generated/generated_unit_0078.cpp index 27ba6a6..341b58b 100644 --- a/profiles/vcs/generated/generated_unit_0078.cpp +++ b/profiles/vcs/generated/generated_unit_0078.cpp @@ -7177,33 +7177,8 @@ L_0893EFFC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7249,19 +7224,9 @@ L_0893F084: ctx.gpr[6] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[6]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[22] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (aot_mem.aot_direct_load8(ctx.gpr[5] + static_cast(5))); @@ -7433,11 +7398,7 @@ L_0893F1D8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7460,11 +7421,7 @@ L_0893F1D8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7479,10 +7436,7 @@ L_0893F1D8: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.fpr[12] = ctx.fpr[12] / ctx.fpr[13]; @@ -7520,11 +7474,7 @@ L_0893F1D8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[22] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7560,11 +7510,7 @@ L_0893F1D8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7572,10 +7518,7 @@ L_0893F1D8: ctx.fpr[12] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[28] + static_cast(7676))); ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -7738,28 +7681,7 @@ L_0893F318: ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[5]); - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 4u); - ctx.read_vfpu_vector_ct<12u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 14u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<14u, 36u, 12u, 4u, 3u>(); ctx.vfpu_ctrl[1u] = 0x00000000u; ctx.execute_vfpu_vcmp_ct<14u, 0u, 4u, 3u>(); ctx.gpr[5] = (0u + static_cast(0)); @@ -8348,33 +8270,8 @@ L_0893F78C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8398,11 +8295,7 @@ L_0893F78C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8418,10 +8311,7 @@ L_0893F78C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16320u << 16u); @@ -9029,11 +8919,7 @@ L_0893FCDC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -9096,11 +8982,7 @@ L_0893FD48: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); diff --git a/profiles/vcs/generated/generated_unit_0079.cpp b/profiles/vcs/generated/generated_unit_0079.cpp index 12d3ad0..e33d717 100644 --- a/profiles/vcs/generated/generated_unit_0079.cpp +++ b/profiles/vcs/generated/generated_unit_0079.cpp @@ -726,11 +726,7 @@ L_08940000: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1156,33 +1152,8 @@ L_089402CC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[21] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); @@ -1266,33 +1237,8 @@ L_0894033C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1321,11 +1267,7 @@ L_08940374: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1340,10 +1282,7 @@ L_08940374: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16320u << 16u); @@ -1586,10 +1525,7 @@ L_08940588: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(176)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1906,11 +1842,7 @@ L_08940818: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1934,11 +1866,7 @@ L_08940818: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[18] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); @@ -2000,11 +1928,7 @@ L_08940818: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(160)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2037,11 +1961,7 @@ L_08940818: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(832)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -2068,7 +1988,7 @@ L_08940818: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(816)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -2093,11 +2013,7 @@ L_08940818: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(800)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -2122,11 +2038,7 @@ L_08940818: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[20] = (ctx.gpr[29] + static_cast(176)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); @@ -2156,11 +2068,7 @@ L_08940918: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[19] = (ctx.gpr[29] + static_cast(192)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); @@ -2179,10 +2087,7 @@ L_08940918: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -2267,11 +2172,7 @@ L_089409A0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[16] = (ctx.gpr[29] + static_cast(208)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); @@ -2315,11 +2216,7 @@ L_089409A0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2366,11 +2263,7 @@ L_089409F4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2404,10 +2297,7 @@ L_08940A40: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(240)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2576,10 +2466,7 @@ L_08940C2C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(416)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3028,11 +2915,7 @@ L_08940F88: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3048,10 +2931,7 @@ L_08940F88: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); { const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); @@ -3135,11 +3015,7 @@ L_08941028: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -3155,10 +3031,7 @@ L_08941028: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[6] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[6]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[12] <= ctx.fpr[14])) ? 0x00800000u : 0u); @@ -3193,11 +3066,7 @@ L_0894107C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[9] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[9] + static_cast(0); @@ -3221,11 +3090,7 @@ L_0894107C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[10] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[10] + static_cast(0); @@ -3241,10 +3106,7 @@ L_0894107C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[11] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[11]); ctx.fpr[12] = ctx.fpr[14] / ctx.fpr[12]; @@ -3283,11 +3145,7 @@ L_0894107C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[8] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[8] + static_cast(0); @@ -3326,11 +3184,7 @@ L_0894107C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[9] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3338,10 +3192,7 @@ L_0894107C: ctx.fpr[12] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[28] + static_cast(7676))); ctx.gpr[7] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[7]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[9] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -3437,11 +3288,7 @@ L_0894116C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(144)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -3457,10 +3304,7 @@ L_0894116C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[6] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[6]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[12] <= ctx.fpr[14])) ? 0x00800000u : 0u); @@ -3495,11 +3339,7 @@ L_089411C0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[8] = (ctx.gpr[29] + static_cast(208)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[8] + static_cast(0); @@ -3523,11 +3363,7 @@ L_089411C0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[9] = (ctx.gpr[29] + static_cast(224)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[9] + static_cast(0); @@ -3543,10 +3379,7 @@ L_089411C0: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[9] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[9]); ctx.fpr[12] = ctx.fpr[14] / ctx.fpr[12]; @@ -3585,11 +3418,7 @@ L_089411C0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(176)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -3677,11 +3506,7 @@ L_08941278: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(240)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -3697,10 +3522,7 @@ L_08941278: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[12] <= ctx.fpr[14])) ? 0x00800000u : 0u); @@ -3735,11 +3557,7 @@ L_089412CC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[7] = (ctx.gpr[29] + static_cast(304)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); @@ -3763,11 +3581,7 @@ L_089412CC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[8] = (ctx.gpr[29] + static_cast(320)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[8] + static_cast(0); @@ -3783,10 +3597,7 @@ L_089412CC: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[8] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[8]); ctx.fpr[12] = ctx.fpr[14] / ctx.fpr[12]; @@ -3825,11 +3636,7 @@ L_089412CC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(272)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -3921,11 +3728,7 @@ L_0894137C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(336)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3968,10 +3771,7 @@ L_089413CC: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -4036,11 +3836,7 @@ L_089413CC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(368)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7961,11 +7757,7 @@ L_08943468: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8068,11 +7860,7 @@ L_08943564: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8176,11 +7964,7 @@ L_08943650: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8282,11 +8066,7 @@ L_0894373C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8942,11 +8722,7 @@ L_08943D90: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9039,11 +8815,7 @@ L_08943E74: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9137,11 +8909,7 @@ L_08943F44: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0080.cpp b/profiles/vcs/generated/generated_unit_0080.cpp index ffb1e3c..d771628 100644 --- a/profiles/vcs/generated/generated_unit_0080.cpp +++ b/profiles/vcs/generated/generated_unit_0080.cpp @@ -1054,11 +1054,7 @@ L_08944014: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1366,11 +1362,7 @@ L_08944240: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(672)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1386,10 +1378,7 @@ L_08944240: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16544u << 16u); @@ -6430,33 +6419,8 @@ L_08946B14: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6511,30 +6475,10 @@ L_08946B58: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } diff --git a/profiles/vcs/generated/generated_unit_0081.cpp b/profiles/vcs/generated/generated_unit_0081.cpp index e5ff487..879add5 100644 --- a/profiles/vcs/generated/generated_unit_0081.cpp +++ b/profiles/vcs/generated/generated_unit_0081.cpp @@ -2845,15 +2845,8 @@ L_08948D04: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (49216u << 16u); @@ -2865,15 +2858,8 @@ L_08948D04: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16448u << 16u); @@ -4687,11 +4673,7 @@ L_08949AE8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -4707,10 +4689,7 @@ L_08949AE8: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[5]); ctx.gpr[31] = (0x08949B1Cu); @@ -4973,11 +4952,7 @@ L_08949CFC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0082.cpp b/profiles/vcs/generated/generated_unit_0082.cpp index 350db72..61b585b 100644 --- a/profiles/vcs/generated/generated_unit_0082.cpp +++ b/profiles/vcs/generated/generated_unit_0082.cpp @@ -3272,11 +3272,7 @@ L_0894D0AC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3463,11 +3459,7 @@ L_0894D24C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[17] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); @@ -3534,10 +3526,7 @@ L_0894D2D0: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -4582,11 +4571,7 @@ L_0894DA88: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4630,10 +4615,7 @@ L_0894DA88: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -4676,11 +4658,7 @@ L_0894DA88: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4876,11 +4854,7 @@ L_0894DC8C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4895,10 +4869,7 @@ L_0894DC8C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[12] <= ctx.fpr[20])) ? 0x00800000u : 0u); @@ -5097,11 +5068,7 @@ L_0894DE14: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5116,10 +5083,7 @@ L_0894DE14: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[12] <= ctx.fpr[20])) ? 0x00800000u : 0u); @@ -5336,11 +5300,7 @@ L_0894DFD0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5396,10 +5356,7 @@ L_0894E02C: ctx.fpr[12] = std::sqrt(ctx.fpr[12]); ctx.gpr[9] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<1u>(ctx.gpr[9]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<1u, 1u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -6498,11 +6455,7 @@ L_0894E854: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6577,11 +6530,7 @@ L_0894E8B0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6645,11 +6594,7 @@ L_0894E8F4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6731,11 +6676,7 @@ L_0894E954: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6798,11 +6739,7 @@ L_0894E998: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(160)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -6821,10 +6758,7 @@ L_0894E998: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -6844,10 +6778,7 @@ L_0894E998: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -6880,24 +6811,10 @@ L_0894E998: { const float vfpu_constant = std::bit_cast(0x3FC90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 64u, 32u, 1u, 2u>(); + ctx.execute_vfpu_vec3_ct<64u, 32u, 0u, 1u, 1u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<64u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(240), 0u); @@ -6919,15 +6836,8 @@ L_0894EA60: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[20] < ctx.fpr[13])) ? 0x00800000u : 0u); @@ -6980,10 +6890,7 @@ L_0894EAC4: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7169,33 +7076,8 @@ L_0894EBB0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(448)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8231,15 +8113,8 @@ L_0894F30C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (49216u << 16u); @@ -8251,15 +8126,8 @@ L_0894F30C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16448u << 16u); @@ -9680,11 +9548,7 @@ L_0894FEBC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(512)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -9821,11 +9685,7 @@ L_0894FFA0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(560)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); diff --git a/profiles/vcs/generated/generated_unit_0083.cpp b/profiles/vcs/generated/generated_unit_0083.cpp index 44a7298..9a44848 100644 --- a/profiles/vcs/generated/generated_unit_0083.cpp +++ b/profiles/vcs/generated/generated_unit_0083.cpp @@ -1102,11 +1102,7 @@ L_08950068: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(320)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1210,10 +1206,7 @@ L_08950120: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -1250,24 +1243,10 @@ L_08950120: { const float vfpu_constant = std::bit_cast(0x3FC90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 64u, 32u, 1u, 2u>(); + ctx.execute_vfpu_vec3_ct<64u, 32u, 0u, 1u, 1u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<64u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (17204u << 16u); @@ -1371,10 +1350,7 @@ L_08950244: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -1411,24 +1387,10 @@ L_08950244: { const float vfpu_constant = std::bit_cast(0x3FC90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 64u, 32u, 1u, 2u>(); + ctx.execute_vfpu_vec3_ct<64u, 32u, 0u, 1u, 1u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<64u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (17204u << 16u); @@ -1567,10 +1529,7 @@ L_08950380: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -1715,11 +1674,7 @@ L_0895046C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(656)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1738,10 +1693,7 @@ L_0895046C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -1978,10 +1930,7 @@ L_08950640: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -2016,24 +1965,10 @@ L_08950640: { const float vfpu_constant = std::bit_cast(0x3FC90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 64u, 32u, 1u, 2u>(); + ctx.execute_vfpu_vec3_ct<64u, 32u, 0u, 1u, 1u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<64u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[12]; const float ft = ctx.fpr[24]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[22] = std::bit_cast(0x7FC00000u); else ctx.fpr[22] = fs * ft; } @@ -3264,19 +3199,9 @@ L_08950F78: ctx.gpr[5] = (std::bit_cast(ctx.fpr[13])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[13] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[16] + static_cast(3168))); @@ -4974,11 +4899,7 @@ L_08951BE4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5673,11 +5594,7 @@ L_08952018: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(160)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -5948,11 +5865,7 @@ L_0895224C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(288)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5999,11 +5912,7 @@ L_08952270: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(304)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8025,11 +7934,7 @@ L_08953074: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8287,11 +8192,7 @@ L_08953250: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8316,11 +8217,7 @@ L_08953250: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8338,10 +8235,7 @@ L_08953250: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -8608,11 +8502,7 @@ L_089533DC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[23] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9404,28 +9294,7 @@ L_08953934: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9482,33 +9351,8 @@ L_08953954: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9529,10 +9373,7 @@ L_0895397C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0084.cpp b/profiles/vcs/generated/generated_unit_0084.cpp index a7fac43..56d6990 100644 --- a/profiles/vcs/generated/generated_unit_0084.cpp +++ b/profiles/vcs/generated/generated_unit_0084.cpp @@ -2085,10 +2085,7 @@ L_08954810: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); ctx.gpr[17] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); @@ -2167,28 +2164,7 @@ L_08954870: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2203,10 +2179,7 @@ L_08954870: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2809,11 +2782,7 @@ L_08954D34: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<27u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<27u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<27u, 27u, 0u, 3u, 1u>(); ctx.execute_vfpu_vdot_ct<0u, 27u, 27u, 3u>(); aot_mem.aot_direct_store32(ctx.gpr[18] + static_cast(0), ctx.vfpu_scalar_bits_ct<0u>()); { const bool branch_taken = 0u == 0u; @@ -2867,11 +2836,7 @@ L_08954D74: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<27u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<27u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<27u, 27u, 0u, 3u, 1u>(); ctx.execute_vfpu_vdot_ct<0u, 27u, 27u, 3u>(); aot_mem.aot_direct_store32(ctx.gpr[5] + static_cast(0), ctx.vfpu_scalar_bits_ct<0u>()); ctx.gpr[5] = (aot_mem.aot_direct_load32(ctx.gpr[4] + static_cast(0))); @@ -3672,33 +3637,8 @@ L_08955444: ctx.gpr[10] = (ctx.gpr[28] + static_cast(-17792)); ctx.set_vfpu_scalar_bits_ct<76u>(aot_mem.aot_direct_load32(ctx.gpr[10] + static_cast(0))); ctx.execute_vfpu_vh2f_ct<21u, 12u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<117u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<76u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<117u, 1u>(vfpu_d); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 4u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<21u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<14u, 4u>(vfpu_result); } + ctx.execute_vfpu_vec3_ct<117u, 117u, 76u, 1u, 2u>(); + ctx.execute_vfpu_vtfm_ct<14u, 36u, 21u, 4u, 3u>(); ctx.vfpu_ctrl[1u] = 0x000000FFu; ctx.execute_vfpu_vcmp_ct<14u, 21u, 4u, 3u>(); { const bool branch_taken = ((ctx.vfpu_ctrl[3] >> 5u) & 1u) == 0u; @@ -3709,11 +3649,7 @@ L_08955444: goto L_08955474; } L_08955474: - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<13u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<27u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<13u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<13u, 13u, 27u, 3u, 1u>(); ctx.execute_vfpu_vdot_ct<12u, 13u, 13u, 3u>(); aot_mem.aot_direct_store32(ctx.gpr[9] + static_cast(0), ctx.vfpu_scalar_bits_ct<12u>()); aot_mem.aot_direct_store32(ctx.gpr[9] + static_cast(4), ctx.gpr[5]); @@ -3822,28 +3758,7 @@ L_08955544: goto L_0895554C; } L_0895554C: - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<40u, 4u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<21u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<14u, 4u>(vfpu_result); } + ctx.execute_vfpu_vtfm_ct<14u, 40u, 21u, 4u, 3u>(); ctx.vfpu_ctrl[0u] = 0x00000FE4u; ctx.vfpu_ctrl[1u] = 0x000000FFu; ctx.execute_vfpu_vcmp_ct<14u, 21u, 4u, 3u>(); @@ -7814,11 +7729,7 @@ L_08957500: ctx.set_vfpu_scalar_bits_ct<27u>(ctx.gpr[25]); ctx.set_vfpu_scalar_bits_ct<59u>(ctx.gpr[2]); ctx.set_vfpu_scalar_bits_ct<91u>(ctx.gpr[3]); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<27u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<15u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<27u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<27u, 27u, 15u, 3u, 1u>(); { float vfpu_value[4]{}; vfpu_value[3u] = 1.0f; ctx.write_vfpu_vector_with_destination_prefix_ct<59u, 4u>(vfpu_value); } { float vfpu_value[4]{}; @@ -7827,10 +7738,7 @@ L_08957500: ctx.write_vfpu_vector_with_destination_prefix_ct<12u, 4u>(vfpu_value); } { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<13u, 2u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<12u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<12u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<12u, 12u, 3u, 2u>(); { float vfpu_s[16]{}, vfpu_t[16]{}, vfpu_d[16]{}; ctx.read_vfpu_matrix_ct<36u, 4u>(vfpu_s); ctx.read_vfpu_matrix_ct<24u, 4u>(vfpu_t); diff --git a/profiles/vcs/generated/generated_unit_0085.cpp b/profiles/vcs/generated/generated_unit_0085.cpp index 89b7693..45d34d7 100644 --- a/profiles/vcs/generated/generated_unit_0085.cpp +++ b/profiles/vcs/generated/generated_unit_0085.cpp @@ -1301,10 +1301,7 @@ L_08958474: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[16] + static_cast(96), std::bit_cast(ctx.fpr[12])); @@ -1319,10 +1316,7 @@ L_08958474: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); aot_mem.aot_direct_store32(ctx.gpr[16] + static_cast(100), std::bit_cast(ctx.fpr[12])); @@ -1337,10 +1331,7 @@ L_08958474: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[6] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[6]); aot_mem.aot_direct_store32(ctx.gpr[16] + static_cast(104), std::bit_cast(ctx.fpr[12])); @@ -1449,28 +1440,7 @@ L_08958474: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<8u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<4u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<8u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } + ctx.execute_vfpu_vtfm_ct<0u, 4u, 8u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1857,11 +1827,7 @@ L_0895890C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1876,10 +1842,7 @@ L_0895890C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[26] = std::bit_cast(ctx.gpr[4]); ctx.fpr[22] = std::bit_cast(std::bit_cast(ctx.fpr[24])); @@ -1958,11 +1921,7 @@ L_089589B8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(96)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1978,10 +1937,7 @@ L_089589B8: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[22] = std::bit_cast(ctx.gpr[4]); goto L_089589E0; @@ -2075,11 +2031,7 @@ L_08958A78: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2095,10 +2047,7 @@ L_08958A78: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[4]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[22] <= ctx.fpr[20])) ? 0x00800000u : 0u); @@ -2359,11 +2308,7 @@ L_08958BE0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2420,11 +2365,7 @@ L_08958C40: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2770,10 +2711,7 @@ L_08958F60: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(96)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3024,33 +2962,8 @@ L_08959118: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(384)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3177,10 +3090,7 @@ L_08959258: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3343,10 +3253,7 @@ L_08959414: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); ctx.gpr[16] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); @@ -3968,33 +3875,8 @@ L_089598B0: ctx.set_vfpu_scalar_bits_ct<44u>(aot_mem.aot_direct_load32(ctx.gpr[17] + static_cast(8))); ctx.set_vfpu_scalar_bits_ct<76u>(aot_mem.aot_direct_load32(ctx.gpr[23] + static_cast(0))); ctx.execute_vfpu_vh2f_ct<21u, 12u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<117u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<76u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<117u, 1u>(vfpu_d); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 4u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<21u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<14u, 4u>(vfpu_result); } + ctx.execute_vfpu_vec3_ct<117u, 117u, 76u, 1u, 2u>(); + ctx.execute_vfpu_vtfm_ct<14u, 36u, 21u, 4u, 3u>(); ctx.vfpu_ctrl[1u] = 0x000000FFu; ctx.execute_vfpu_vcmp_ct<14u, 21u, 4u, 3u>(); { const bool branch_taken = ((ctx.vfpu_ctrl[3] >> 5u) & 1u) == 0u; @@ -4064,33 +3946,8 @@ L_08959930: ctx.set_vfpu_scalar_bits_ct<44u>(aot_mem.aot_direct_load32(ctx.gpr[17] + static_cast(8))); ctx.set_vfpu_scalar_bits_ct<76u>(aot_mem.aot_direct_load32(ctx.gpr[23] + static_cast(0))); ctx.execute_vfpu_vh2f_ct<21u, 12u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<117u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<76u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<117u, 1u>(vfpu_d); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 4u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<21u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<14u, 4u>(vfpu_result); } + ctx.execute_vfpu_vec3_ct<117u, 117u, 76u, 1u, 2u>(); + ctx.execute_vfpu_vtfm_ct<14u, 36u, 21u, 4u, 3u>(); ctx.vfpu_ctrl[1u] = 0x000000FFu; ctx.execute_vfpu_vcmp_ct<14u, 21u, 4u, 3u>(); { const bool branch_taken = ((ctx.vfpu_ctrl[3] >> 5u) & 1u) == 0u; @@ -4474,33 +4331,8 @@ L_08959BF0: ctx.set_vfpu_scalar_bits_ct<44u>(aot_mem.aot_direct_load32(ctx.gpr[16] + static_cast(8))); ctx.set_vfpu_scalar_bits_ct<76u>(aot_mem.aot_direct_load32(ctx.gpr[23] + static_cast(0))); ctx.execute_vfpu_vh2f_ct<21u, 12u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<117u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<76u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<117u, 1u>(vfpu_d); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 4u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<21u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<14u, 4u>(vfpu_result); } + ctx.execute_vfpu_vec3_ct<117u, 117u, 76u, 1u, 2u>(); + ctx.execute_vfpu_vtfm_ct<14u, 36u, 21u, 4u, 3u>(); ctx.vfpu_ctrl[1u] = 0x000000FFu; ctx.execute_vfpu_vcmp_ct<14u, 21u, 4u, 3u>(); { const bool branch_taken = ((ctx.vfpu_ctrl[3] >> 5u) & 1u) == 0u; @@ -4629,33 +4461,8 @@ L_08959CD8: ctx.set_vfpu_scalar_bits_ct<44u>(aot_mem.aot_direct_load32(ctx.gpr[16] + static_cast(8))); ctx.set_vfpu_scalar_bits_ct<76u>(aot_mem.aot_direct_load32(ctx.gpr[23] + static_cast(0))); ctx.execute_vfpu_vh2f_ct<21u, 12u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<117u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<76u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<117u, 1u>(vfpu_d); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 4u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<21u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<14u, 4u>(vfpu_result); } + ctx.execute_vfpu_vec3_ct<117u, 117u, 76u, 1u, 2u>(); + ctx.execute_vfpu_vtfm_ct<14u, 36u, 21u, 4u, 3u>(); ctx.vfpu_ctrl[1u] = 0x000000FFu; ctx.execute_vfpu_vcmp_ct<14u, 21u, 4u, 3u>(); { const bool branch_taken = ((ctx.vfpu_ctrl[3] >> 5u) & 1u) == 0u; @@ -5077,33 +4884,8 @@ L_0895A03C: ctx.set_vfpu_scalar_bits_ct<44u>(aot_mem.aot_direct_load32(ctx.gpr[16] + static_cast(8))); ctx.set_vfpu_scalar_bits_ct<76u>(aot_mem.aot_direct_load32(ctx.gpr[22] + static_cast(0))); ctx.execute_vfpu_vh2f_ct<21u, 12u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<117u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<76u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<117u, 1u>(vfpu_d); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 4u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<21u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<14u, 4u>(vfpu_result); } + ctx.execute_vfpu_vec3_ct<117u, 117u, 76u, 1u, 2u>(); + ctx.execute_vfpu_vtfm_ct<14u, 36u, 21u, 4u, 3u>(); ctx.vfpu_ctrl[1u] = 0x000000FFu; ctx.execute_vfpu_vcmp_ct<14u, 21u, 4u, 3u>(); { const bool branch_taken = ((ctx.vfpu_ctrl[3] >> 5u) & 1u) == 0u; @@ -6701,24 +6483,8 @@ L_0895AFEC: ctx.set_vfpu_scalar_bits_ct<14u>(aot_mem.aot_direct_load32(ctx.gpr[18] + static_cast(20))); ctx.execute_vfpu_vx2i_ct<12u, 13u, 2u, 3u>(); ctx.execute_vfpu_vx2i_ct<13u, 14u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<12u, 4u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(31u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 4u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<12u, 4u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<13u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(31u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<13u, 2u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<12u, 12u, 4u, 31u>(); + ctx.execute_vfpu_vi2f_ct<13u, 13u, 2u, 31u>(); ctx.execute_vfpu_vcmp_ct<52u, 31u, 3u, 6u>(); ctx.execute_vfpu_vcmov_ct<1u, 108u, 1u, 0u, true>(); ctx.execute_vfpu_vcmov_ct<1u, 12u, 1u, 0u, false>(); @@ -7814,11 +7580,7 @@ L_0895B934: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<43u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<3u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<15u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<3u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<3u, 3u, 15u, 3u, 1u>(); ctx.execute_vfpu_vmmov_ct<24u, 0u, 4u>(); { float vfpu_value[4]{}; ctx.write_vfpu_vector_with_destination_prefix_ct<31u, 4u>(vfpu_value); } @@ -8024,41 +7786,12 @@ L_0895BAA4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<8u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<27u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<15u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<3u, 3u>(vfpu_d); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<56u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<4u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<5u, 3u>(vfpu_result); } + ctx.execute_vfpu_vec3_ct<3u, 27u, 15u, 3u, 0u>(); + ctx.execute_vfpu_vtfm_ct<5u, 56u, 4u, 3u, 3u>(); ctx.execute_vfpu_vscl_ct<0u, 24u, 8u, 3u>(); ctx.execute_vfpu_vscl_ct<1u, 25u, 40u, 3u>(); ctx.execute_vfpu_vscl_ct<2u, 26u, 72u, 3u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<3u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<5u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<3u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<3u, 3u, 5u, 3u, 0u>(); ctx.gpr[12] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.gpr[13] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.gpr[14] = (ctx.vfpu_scalar_bits_ct<64u>()); diff --git a/profiles/vcs/generated/generated_unit_0086.cpp b/profiles/vcs/generated/generated_unit_0086.cpp index 028ceb5..4adae89 100644 --- a/profiles/vcs/generated/generated_unit_0086.cpp +++ b/profiles/vcs/generated/generated_unit_0086.cpp @@ -1673,15 +1673,8 @@ L_0895CA18: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -1696,15 +1689,8 @@ L_0895CA3C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -2853,28 +2839,7 @@ L_0895D3C0: ctx.fpr[13] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (std::bit_cast(ctx.fpr[13])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[5]); - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 4u); - ctx.read_vfpu_vector_ct<12u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 14u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<14u, 36u, 12u, 4u, 3u>(); ctx.vfpu_ctrl[1u] = 0x00000000u; ctx.execute_vfpu_vcmp_ct<14u, 0u, 4u, 3u>(); ctx.gpr[5] = (0u + static_cast(0)); @@ -4033,11 +3998,7 @@ L_0895DD2C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4065,10 +4026,7 @@ L_0895DD2C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[10] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[10]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[13] <= ctx.fpr[12])) ? 0x00800000u : 0u); @@ -4116,11 +4074,7 @@ L_0895DD68: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4608,15 +4562,8 @@ L_0895E110: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[15] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (std::bit_cast(ctx.fpr[14])); @@ -4624,15 +4571,8 @@ L_0895E110: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[5]); ctx.fpr[16] = std::bit_cast(std::bit_cast(ctx.fpr[14]) ^ 0x80000000u); @@ -4740,11 +4680,7 @@ L_0895E22C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4760,10 +4696,7 @@ L_0895E22C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[17] = std::bit_cast(ctx.gpr[5]); ctx.fpr[12] = ctx.fpr[12] / ctx.fpr[17]; @@ -4862,11 +4795,7 @@ L_0895E2B8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[9] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[9] + static_cast(0); @@ -4927,15 +4856,8 @@ L_0895E36C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[15] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[14])); @@ -4943,15 +4865,8 @@ L_0895E36C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[16] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (aot_mem.aot_direct_load32(ctx.gpr[16] + static_cast(32))); @@ -5104,15 +5019,8 @@ L_0895E4D8: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[15] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[14])); @@ -5120,15 +5028,8 @@ L_0895E4D8: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[16] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (aot_mem.aot_direct_load32(ctx.gpr[16] + static_cast(32))); @@ -5403,15 +5304,8 @@ L_0895E6E8: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[15] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[14])); @@ -5419,15 +5313,8 @@ L_0895E6E8: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[16] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (aot_mem.aot_direct_load32(ctx.gpr[16] + static_cast(32))); @@ -5484,15 +5371,8 @@ L_0895E7AC: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[15] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[14])); @@ -5500,15 +5380,8 @@ L_0895E7AC: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[16] = std::bit_cast(ctx.gpr[4]); ctx.fpr[14] = std::bit_cast(std::bit_cast(ctx.fpr[15]) ^ 0x80000000u); @@ -6238,28 +6111,7 @@ L_0895ED58: ctx.write_vfpu_vector_ct<39u, 4u>(vfpu_value); } ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 4u); - ctx.read_vfpu_vector_ct<12u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 14u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<14u, 36u, 12u, 4u, 3u>(); ctx.vfpu_ctrl[1u] = 0x00000000u; ctx.execute_vfpu_vcmp_ct<14u, 0u, 4u, 3u>(); ctx.gpr[4] = (0u + static_cast(0)); @@ -6434,11 +6286,7 @@ L_0895EF04: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6775,11 +6623,7 @@ L_0895F21C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6812,10 +6656,7 @@ L_0895F21C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -6859,7 +6700,7 @@ L_0895F21C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6891,10 +6732,7 @@ L_0895F21C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -6919,7 +6757,7 @@ L_0895F21C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6996,11 +6834,7 @@ L_0895F308: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7135,10 +6969,7 @@ L_0895F420: ctx.fpr[12] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[28] + static_cast(7676))); ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<1u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<1u, 1u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -7417,11 +7248,7 @@ L_0895F628: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7436,10 +7263,7 @@ L_0895F628: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (ctx.gpr[4] + static_cast(64)); @@ -7494,28 +7318,7 @@ L_0895F628: ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[5]); - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 4u); - ctx.read_vfpu_vector_ct<12u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 14u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<14u, 36u, 12u, 4u, 3u>(); ctx.vfpu_ctrl[1u] = 0x00000000u; ctx.execute_vfpu_vcmp_ct<14u, 0u, 4u, 3u>(); ctx.gpr[5] = (0u + static_cast(0)); @@ -7923,11 +7726,7 @@ L_0895FAF4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7942,10 +7741,7 @@ L_0895FAF4: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (16840u << 16u); @@ -8010,28 +7806,7 @@ L_0895FB40: ctx.write_vfpu_vector_ct<39u, 4u>(vfpu_value); } ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 4u); - ctx.read_vfpu_vector_ct<12u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 14u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<14u, 36u, 12u, 4u, 3u>(); ctx.vfpu_ctrl[1u] = 0x00000000u; ctx.execute_vfpu_vcmp_ct<14u, 0u, 4u, 3u>(); ctx.gpr[4] = (0u + static_cast(0)); @@ -8377,11 +8152,7 @@ L_0895FDB0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8477,11 +8248,7 @@ L_0895FE24: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(176)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8565,11 +8332,7 @@ L_0895FE90: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[23] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8617,11 +8380,7 @@ L_0895FEEC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[22] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8637,10 +8396,7 @@ L_0895FEEC: L_0895FF38: ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[22] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -8717,11 +8473,7 @@ L_0895FF88: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -8737,10 +8489,7 @@ L_0895FF88: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (16864u << 16u); diff --git a/profiles/vcs/generated/generated_unit_0087.cpp b/profiles/vcs/generated/generated_unit_0087.cpp index 461c489..3062d23 100644 --- a/profiles/vcs/generated/generated_unit_0087.cpp +++ b/profiles/vcs/generated/generated_unit_0087.cpp @@ -988,28 +988,7 @@ LOCAL_DISPATCH: } } L_08960000: - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 4u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<12u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<14u, 4u>(vfpu_result); } + ctx.execute_vfpu_vtfm_ct<14u, 36u, 12u, 4u, 3u>(); ctx.vfpu_ctrl[1u] = 0x00000000u; ctx.execute_vfpu_vcmp_ct<14u, 0u, 4u, 3u>(); ctx.gpr[4] = (0u + static_cast(0)); @@ -2478,10 +2457,7 @@ L_08960BAC: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (15436u << 16u); @@ -3894,33 +3870,8 @@ L_089617F8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6382,10 +6333,7 @@ L_08962BFC: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (16076u << 16u); @@ -6482,10 +6430,7 @@ L_08962CA0: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (15820u << 16u); @@ -7677,10 +7622,7 @@ L_08963618: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (15692u << 16u); @@ -8193,11 +8135,7 @@ L_0896399C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(96)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8213,10 +8151,7 @@ L_0896399C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16672u << 16u); diff --git a/profiles/vcs/generated/generated_unit_0088.cpp b/profiles/vcs/generated/generated_unit_0088.cpp index bb82198..2f6e665 100644 --- a/profiles/vcs/generated/generated_unit_0088.cpp +++ b/profiles/vcs/generated/generated_unit_0088.cpp @@ -952,33 +952,8 @@ L_08964000: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1277,51 +1252,23 @@ L_0896433C: L_089644F8: ctx.gpr[8] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<4u>(ctx.gpr[8]); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<4u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<4u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 4u, 4u, 1u, 2u>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 0u>(); { float vfpu_value[4]{}; ctx.write_vfpu_vector_with_destination_prefix_ct<96u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 64u, 1u, 17u>(); { const float vfpu_constant = std::bit_cast(0x3FC90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_vec3_ct<0u, 64u, 0u, 1u, 1u>(); ctx.execute_vfpu_vcmp_ct<4u, 96u, 1u, 2u>(); if (((ctx.vfpu_ctrl[3] >> 5u) & 1u) != 0u) { - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 2u>(); goto L_08964534; } goto L_08964534; @@ -1342,10 +1289,7 @@ L_08964540: ctx.execute_vfpu_vcmp_ct<4u, 4u, 2u, 1u>(); ctx.vfpu_ctrl[0u] = 0x000010E5u; ctx.execute_vfpu_vcmov_ct<4u, 4u, 1u, 5u, false>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<4u, 2u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 2u; ++i) vfpu_d[i] = std::fabs(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<5u, 2u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<5u, 4u, 2u, 1u>(); { const float vfpu_constant = std::bit_cast(0x3FC90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } @@ -1353,57 +1297,24 @@ L_08964540: { float vfpu_value[4]{}; ctx.write_vfpu_vector_with_destination_prefix_ct<68u, 1u>(vfpu_value); } if (((ctx.vfpu_ctrl[3] >> 5u) & 1u) != 0u) { - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<4u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<36u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] / vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<8u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<8u, 4u, 36u, 1u, 3u>(); goto L_0896457C; } goto L_08964578; L_08964578: - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<36u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<4u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] / vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<8u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<8u, 36u, 4u, 1u, 3u>(); goto L_0896457C; L_0896457C: - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<8u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<8u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 8u, 8u, 1u, 2u>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 32u, 1u, 0u>(); + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 17u>(); + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 64u, 1u, 2u>(); if (((ctx.vfpu_ctrl[3] >> 5u) & 1u) != 0u) { - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 64u, 0u, 1u, 1u>(); goto L_089645A0; } goto L_089645A0; @@ -1421,42 +1332,23 @@ L_089645A0: } L_089645B0: if (((ctx.vfpu_ctrl[3] >> 5u) & 1u) != 0u) { - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<68u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 68u, 0u, 1u, 1u>(); goto L_089645D0; } goto L_089645B8; L_089645B8: - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<96u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<96u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<96u, 96u, 1u, 2u>(); { const bool branch_taken = 0u == 0u; - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<96u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 96u, 1u, 0u>(); if (branch_taken) { goto L_089645D0; } goto L_089645C4; } L_089645C4: - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<96u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 96u, 0u, 1u, 1u>(); if (((ctx.vfpu_ctrl[3] >> 5u) & 1u) != 0u) { - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<96u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 96u, 0u, 1u, 1u>(); goto L_089645D0; } goto L_089645D0; @@ -6407,11 +6299,7 @@ L_089671C8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6426,10 +6314,7 @@ L_089671C8: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (17224u << 16u); @@ -6611,11 +6496,7 @@ L_08967338: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6630,10 +6511,7 @@ L_08967338: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[12] < ctx.fpr[20])) ? 0x00800000u : 0u); diff --git a/profiles/vcs/generated/generated_unit_0089.cpp b/profiles/vcs/generated/generated_unit_0089.cpp index 7483045..50a4d55 100644 --- a/profiles/vcs/generated/generated_unit_0089.cpp +++ b/profiles/vcs/generated/generated_unit_0089.cpp @@ -1087,11 +1087,7 @@ L_089680C0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -1601,11 +1597,7 @@ L_089684A4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -2683,15 +2675,8 @@ L_08968D2C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[7] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[7]); ctx.gpr[7] = (std::bit_cast(ctx.fpr[13])); @@ -2699,15 +2684,8 @@ L_08968D2C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[7] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[7]); ctx.fpr[13] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[5] + static_cast(0))); @@ -3943,15 +3921,8 @@ L_08969520: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -3966,15 +3937,8 @@ L_08969544: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -4011,11 +3975,7 @@ L_08969568: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -4067,11 +4027,7 @@ L_08969568: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4095,11 +4051,7 @@ L_08969568: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4819,11 +4771,7 @@ L_08969A94: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5100,10 +5048,7 @@ L_08969D18: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16128u << 16u); @@ -5168,11 +5113,7 @@ L_08969D60: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5244,10 +5185,7 @@ L_08969E28: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (ctx.gpr[16] + static_cast(32)); @@ -5306,11 +5244,7 @@ L_08969E68: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7398,19 +7332,9 @@ L_0896AE88: ctx.gpr[5] = (std::bit_cast(ctx.fpr[13])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[13] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[16] + static_cast(1596))); @@ -7740,19 +7664,9 @@ L_0896B14C: ctx.gpr[5] = (std::bit_cast(ctx.fpr[13])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[13] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[16] + static_cast(1596))); @@ -8580,10 +8494,7 @@ L_0896B800: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (0u | 0u); @@ -8742,15 +8653,8 @@ L_0896B928: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); @@ -8758,15 +8662,8 @@ L_0896B928: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); ctx.fpr[12] = std::bit_cast(std::bit_cast(ctx.fpr[13]) ^ 0x80000000u); @@ -8808,28 +8705,7 @@ L_0896B928: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9161,10 +9037,7 @@ L_0896BC1C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -9386,19 +9259,9 @@ L_0896BDA0: { const float vfpu_constant = std::bit_cast(0x3FC90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 64u, 32u, 1u, 2u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[22] = std::bit_cast(ctx.gpr[4]); ctx.fpr[22] = std::bit_cast(std::bit_cast(ctx.fpr[22]) ^ 0x80000000u); @@ -9571,15 +9434,8 @@ L_0896BF04: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (15523u << 16u); @@ -9708,15 +9564,8 @@ L_0896BFE4: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.pc = 0x0896C000u; return; } diff --git a/profiles/vcs/generated/generated_unit_0090.cpp b/profiles/vcs/generated/generated_unit_0090.cpp index c060f78..c55b36b 100644 --- a/profiles/vcs/generated/generated_unit_0090.cpp +++ b/profiles/vcs/generated/generated_unit_0090.cpp @@ -1178,15 +1178,8 @@ L_0896C0C0: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16256u << 16u); @@ -2261,10 +2254,7 @@ L_0896C874: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (15948u << 16u); @@ -2606,10 +2596,7 @@ L_0896CAD0: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (15948u << 16u); @@ -3109,33 +3096,8 @@ L_0896CE08: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7245,11 +7207,7 @@ L_0896EEC4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7400,11 +7358,7 @@ L_0896EFEC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9112,15 +9066,8 @@ L_0896FCF4: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[8] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[8]); ctx.gpr[8] = (std::bit_cast(ctx.fpr[12])); @@ -9128,15 +9075,8 @@ L_0896FCF4: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[8] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[8]); ctx.gpr[8] = (aot_mem.aot_direct_load32(ctx.gpr[4] + static_cast(0))); diff --git a/profiles/vcs/generated/generated_unit_0091.cpp b/profiles/vcs/generated/generated_unit_0091.cpp index e12e7d7..182ca10 100644 --- a/profiles/vcs/generated/generated_unit_0091.cpp +++ b/profiles/vcs/generated/generated_unit_0091.cpp @@ -2120,28 +2120,7 @@ L_089706C8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3600,19 +3579,9 @@ L_089712D0: ctx.gpr[5] = (std::bit_cast(ctx.fpr[0])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[16] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[12]; const float ft = ctx.fpr[16]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[12] = std::bit_cast(0x7FC00000u); else ctx.fpr[12] = fs * ft; } @@ -3707,10 +3676,7 @@ L_08971408: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -3938,15 +3904,8 @@ L_089715F8: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); @@ -3954,15 +3913,8 @@ L_089715F8: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (aot_mem.aot_direct_load32(ctx.gpr[16] + static_cast(0))); @@ -4055,10 +4007,7 @@ L_089716C0: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); diff --git a/profiles/vcs/generated/generated_unit_0092.cpp b/profiles/vcs/generated/generated_unit_0092.cpp index 09f360f..1514211 100644 --- a/profiles/vcs/generated/generated_unit_0092.cpp +++ b/profiles/vcs/generated/generated_unit_0092.cpp @@ -4117,10 +4117,7 @@ L_08975938: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -5957,30 +5954,10 @@ L_08976968: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -6078,30 +6055,10 @@ L_08976A2C: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -6139,30 +6096,10 @@ L_08976A2C: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -6189,11 +6126,7 @@ L_08976A2C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6378,11 +6311,7 @@ L_08976C3C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7186,30 +7115,10 @@ L_08977290: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -7597,30 +7506,10 @@ L_089775CC: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -7646,11 +7535,7 @@ L_089775CC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7938,30 +7823,10 @@ L_0897784C: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -8069,15 +7934,7 @@ L_08977908: ctx.gpr[9] = (static_cast(static_cast(static_cast(aot_mem.aot_direct_load16(ctx.gpr[9] + static_cast(2)))))); ctx.set_vfpu_scalar_bits_ct<1u>(ctx.gpr[8]); ctx.set_vfpu_scalar_bits_ct<33u>(ctx.gpr[9]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); ctx.gpr[7] = (ctx.gpr[29] + static_cast(48)); aot_mem.aot_direct_store32(ctx.gpr[7] + static_cast(0), ctx.vfpu_scalar_bits_ct<2u>()); aot_mem.aot_direct_store32(ctx.gpr[7] + static_cast(4), ctx.vfpu_scalar_bits_ct<34u>()); @@ -8114,10 +7971,7 @@ L_089779AC: L_089779B8: ctx.gpr[7] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[7]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 17u>(); ctx.gpr[7] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[7]); ctx.fpr[14] = std::bit_cast(ctx.gpr[7]); @@ -8209,11 +8063,7 @@ L_08977A24: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8300,11 +8150,7 @@ L_08977B50: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8337,10 +8183,7 @@ L_08977B50: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -8688,30 +8531,10 @@ L_08977E08: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } diff --git a/profiles/vcs/generated/generated_unit_0093.cpp b/profiles/vcs/generated/generated_unit_0093.cpp index a10f6e2..ba3782d 100644 --- a/profiles/vcs/generated/generated_unit_0093.cpp +++ b/profiles/vcs/generated/generated_unit_0093.cpp @@ -974,30 +974,10 @@ L_08978160: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -1024,11 +1004,7 @@ L_08978160: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[8] = (ctx.gpr[3] | 0u); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[8] + static_cast(0); @@ -1160,15 +1136,7 @@ L_089782BC: ctx.gpr[9] = (static_cast(static_cast(static_cast(aot_mem.aot_direct_load16(ctx.gpr[9] + static_cast(2)))))); ctx.set_vfpu_scalar_bits_ct<1u>(ctx.gpr[8]); ctx.set_vfpu_scalar_bits_ct<33u>(ctx.gpr[9]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(0), ctx.vfpu_scalar_bits_ct<2u>()); aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(4), ctx.vfpu_scalar_bits_ct<34u>()); ctx.gpr[9] = (aot_mem.aot_direct_load32(ctx.gpr[6] + static_cast(4))); @@ -1176,15 +1144,7 @@ L_089782BC: ctx.gpr[9] = (static_cast(static_cast(static_cast(aot_mem.aot_direct_load16(ctx.gpr[9] + static_cast(2)))))); ctx.set_vfpu_scalar_bits_ct<1u>(ctx.gpr[8]); ctx.set_vfpu_scalar_bits_ct<33u>(ctx.gpr[9]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(8)); aot_mem.aot_direct_store32(ctx.gpr[4] + static_cast(0), ctx.vfpu_scalar_bits_ct<2u>()); aot_mem.aot_direct_store32(ctx.gpr[4] + static_cast(4), ctx.vfpu_scalar_bits_ct<34u>()); @@ -1328,30 +1288,10 @@ L_08978410: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -1538,30 +1478,10 @@ L_089785AC: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -1810,15 +1730,8 @@ L_08978808: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.fpr[13] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[29] + static_cast(156))); @@ -1830,15 +1743,8 @@ L_08978808: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[15] = std::bit_cast(ctx.gpr[5]); { const float fs = ctx.fpr[13]; const float ft = ctx.fpr[15]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[13] = std::bit_cast(0x7FC00000u); else ctx.fpr[13] = fs * ft; } @@ -1880,10 +1786,7 @@ L_089788AC: L_089788B4: ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 17u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[5]); ctx.fpr[14] = std::bit_cast(ctx.gpr[5]); @@ -1921,10 +1824,7 @@ L_0897890C: L_08978914: ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 17u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[5]); ctx.fpr[14] = std::bit_cast(ctx.gpr[5]); @@ -1991,30 +1891,10 @@ L_08978988: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -2064,30 +1944,10 @@ L_089789E4: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -2355,30 +2215,10 @@ L_08978C50: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -3114,30 +2954,10 @@ L_089792F0: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -3164,11 +2984,7 @@ L_089792F0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[13] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3885,11 +3701,7 @@ L_08979958: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4009,15 +3821,8 @@ L_08979A54: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(200), std::bit_cast(ctx.fpr[12])); @@ -4026,15 +3831,8 @@ L_08979A54: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[30] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[23] + static_cast(0), 0u); @@ -4192,30 +3990,14 @@ L_08979BE0: ctx.gpr[9] = (static_cast(static_cast(static_cast(aot_mem.aot_direct_load16(ctx.gpr[9] + static_cast(2)))))); ctx.set_vfpu_scalar_bits_ct<1u>(ctx.gpr[8]); ctx.set_vfpu_scalar_bits_ct<33u>(ctx.gpr[9]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); aot_mem.aot_direct_store32(ctx.gpr[23] + static_cast(0), ctx.vfpu_scalar_bits_ct<2u>()); aot_mem.aot_direct_store32(ctx.gpr[23] + static_cast(4), ctx.vfpu_scalar_bits_ct<34u>()); ctx.gpr[8] = (static_cast(static_cast(static_cast(aot_mem.aot_direct_load16(ctx.gpr[17] + static_cast(0)))))); ctx.gpr[9] = (static_cast(static_cast(static_cast(aot_mem.aot_direct_load16(ctx.gpr[17] + static_cast(2)))))); ctx.set_vfpu_scalar_bits_ct<1u>(ctx.gpr[8]); ctx.set_vfpu_scalar_bits_ct<33u>(ctx.gpr[9]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); aot_mem.aot_direct_store32(ctx.gpr[22] + static_cast(0), ctx.vfpu_scalar_bits_ct<2u>()); aot_mem.aot_direct_store32(ctx.gpr[22] + static_cast(4), ctx.vfpu_scalar_bits_ct<34u>()); ctx.gpr[4] = (aot_mem.aot_direct_load8(ctx.gpr[29] + static_cast(204))); @@ -4269,30 +4051,10 @@ L_08979C5C: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -4884,30 +4646,10 @@ L_0897A1B8: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -4935,11 +4677,7 @@ L_0897A1B8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[7] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); @@ -5174,11 +4912,7 @@ L_0897A3C8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[22] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5193,10 +4927,7 @@ L_0897A3C8: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[12] <= ctx.fpr[20])) ? 0x00800000u : 0u); @@ -5352,11 +5083,7 @@ L_0897A538: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(144)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5390,10 +5117,7 @@ L_0897A538: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -5765,15 +5489,7 @@ L_0897A8B4: ctx.gpr[9] = (static_cast(static_cast(static_cast(aot_mem.aot_direct_load16(ctx.gpr[9] + static_cast(2)))))); ctx.set_vfpu_scalar_bits_ct<1u>(ctx.gpr[8]); ctx.set_vfpu_scalar_bits_ct<33u>(ctx.gpr[9]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(88)); aot_mem.aot_direct_store32(ctx.gpr[4] + static_cast(0), ctx.vfpu_scalar_bits_ct<2u>()); aot_mem.aot_direct_store32(ctx.gpr[4] + static_cast(4), ctx.vfpu_scalar_bits_ct<34u>()); @@ -5881,15 +5597,7 @@ L_0897A9EC: ctx.gpr[9] = (static_cast(static_cast(static_cast(aot_mem.aot_direct_load16(ctx.gpr[9] + static_cast(2)))))); ctx.set_vfpu_scalar_bits_ct<1u>(ctx.gpr[8]); ctx.set_vfpu_scalar_bits_ct<33u>(ctx.gpr[9]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(112)); aot_mem.aot_direct_store32(ctx.gpr[4] + static_cast(0), ctx.vfpu_scalar_bits_ct<2u>()); aot_mem.aot_direct_store32(ctx.gpr[4] + static_cast(4), ctx.vfpu_scalar_bits_ct<34u>()); @@ -6107,11 +5815,7 @@ L_0897ABEC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6437,15 +6141,7 @@ L_0897AED4: ctx.gpr[9] = (static_cast(static_cast(static_cast(aot_mem.aot_direct_load16(ctx.gpr[9] + static_cast(2)))))); ctx.set_vfpu_scalar_bits_ct<1u>(ctx.gpr[8]); ctx.set_vfpu_scalar_bits_ct<33u>(ctx.gpr[9]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(88)); aot_mem.aot_direct_store32(ctx.gpr[4] + static_cast(0), ctx.vfpu_scalar_bits_ct<2u>()); aot_mem.aot_direct_store32(ctx.gpr[4] + static_cast(4), ctx.vfpu_scalar_bits_ct<34u>()); @@ -6579,15 +6275,7 @@ L_0897B06C: ctx.gpr[9] = (static_cast(static_cast(static_cast(aot_mem.aot_direct_load16(ctx.gpr[9] + static_cast(2)))))); ctx.set_vfpu_scalar_bits_ct<1u>(ctx.gpr[8]); ctx.set_vfpu_scalar_bits_ct<33u>(ctx.gpr[9]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(112)); aot_mem.aot_direct_store32(ctx.gpr[4] + static_cast(0), ctx.vfpu_scalar_bits_ct<2u>()); aot_mem.aot_direct_store32(ctx.gpr[4] + static_cast(4), ctx.vfpu_scalar_bits_ct<34u>()); @@ -6712,30 +6400,10 @@ L_0897B158: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -6777,30 +6445,10 @@ L_0897B158: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -6866,11 +6514,7 @@ L_0897B158: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6985,28 +6629,7 @@ L_0897B2E0: ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 4u); - ctx.read_vfpu_vector_ct<12u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 14u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<14u, 36u, 12u, 4u, 3u>(); ctx.vfpu_ctrl[1u] = 0x00000000u; ctx.execute_vfpu_vcmp_ct<14u, 0u, 4u, 3u>(); ctx.gpr[4] = (0u + static_cast(0)); diff --git a/profiles/vcs/generated/generated_unit_0095.cpp b/profiles/vcs/generated/generated_unit_0095.cpp index 863f5f1..915a37f 100644 --- a/profiles/vcs/generated/generated_unit_0095.cpp +++ b/profiles/vcs/generated/generated_unit_0095.cpp @@ -1002,11 +1002,7 @@ L_08980100: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1034,10 +1030,7 @@ L_08980100: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[3] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[3]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[13] <= ctx.fpr[12])) ? 0x00800000u : 0u); @@ -2893,33 +2886,8 @@ L_08980DCC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5337,11 +5305,7 @@ L_089820DC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[19] = (ctx.gpr[29] + static_cast(192)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); @@ -5360,10 +5324,7 @@ L_089820DC: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -5391,11 +5352,7 @@ L_089820DC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(208)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -5411,10 +5368,7 @@ L_089820DC: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16384u << 16u); @@ -5454,10 +5408,7 @@ L_08982190: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7370,10 +7321,7 @@ L_0898344C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7634,15 +7582,8 @@ L_08983888: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[3] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[3]); { const float fs = ctx.fpr[0]; const float ft = ctx.fpr[15]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[19] = std::bit_cast(0x7FC00000u); else ctx.fpr[19] = fs * ft; } @@ -7988,11 +7929,7 @@ L_08983C28: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8007,10 +7944,7 @@ L_08983C28: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[14] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[17] = std::bit_cast(ctx.gpr[14]); ctx.gpr[14] = (16908u << 16u); diff --git a/profiles/vcs/generated/generated_unit_0096.cpp b/profiles/vcs/generated/generated_unit_0096.cpp index 9c1952e..7ad8cca 100644 --- a/profiles/vcs/generated/generated_unit_0096.cpp +++ b/profiles/vcs/generated/generated_unit_0096.cpp @@ -9351,24 +9351,10 @@ L_08987CC8: { const float vfpu_constant = std::bit_cast(0x3FC90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 64u, 32u, 1u, 2u>(); + ctx.execute_vfpu_vec3_ct<64u, 32u, 0u, 1u, 1u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<64u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[16] + static_cast(0), std::bit_cast(ctx.fpr[14])); @@ -9388,15 +9374,8 @@ L_08987D00: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); ctx.fpr[12] = ctx.fpr[13] / ctx.fpr[14]; @@ -9448,12 +9427,12 @@ L_08987D6C: ctx.gpr[6] = (ctx.gpr[6] & 65535u); ctx.gpr[5] = (ctx.gpr[5] & 65535u); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[5]); - ctx.execute_vfpu_vh2f(0u, 0u, 1u); + ctx.execute_vfpu_vh2f_ct<0u, 0u, 1u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (ctx.gpr[6] & 65535u); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[5]); - ctx.execute_vfpu_vh2f(0u, 0u, 1u); + ctx.execute_vfpu_vh2f_ct<0u, 0u, 1u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[5]); ctx.fpr[12] = ctx.fpr[12] - ctx.fpr[13]; @@ -9461,7 +9440,7 @@ L_08987D6C: ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[5]); { float vfpu_value[4]{}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - ctx.execute_vfpu_vf2h(64u, 0u, 2u); + ctx.execute_vfpu_vf2h_ct<64u, 0u, 2u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<64u>()); aot_mem.aot_direct_store16(ctx.gpr[4] + static_cast(0), static_cast(ctx.gpr[5])); jump_target = ctx.gpr[31]; @@ -9476,7 +9455,7 @@ L_08987DD0: ctx.gpr[5] = (aot_mem.aot_direct_load16(ctx.gpr[5] + static_cast(4))); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[6]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - ctx.execute_vfpu_vh2f(1u, 0u, 2u); + ctx.execute_vfpu_vh2f_ct<1u, 0u, 2u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9496,42 +9475,15 @@ L_08987DF4: goto L_08987DFC; } L_08987DFC: - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<35u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<35u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<99u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<99u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<3u, 4u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 4u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<60u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<3u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<3u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<3u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<35u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<124u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<67u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<92u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<99u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<99u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<99u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<35u, 35u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<99u, 99u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<28u, 3u, 4u, 0u>(); + ctx.execute_vfpu_unary_ct<3u, 60u, 1u, 0u>(); + ctx.execute_vfpu_unary_ct<3u, 3u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<35u, 28u, 1u, 0u>(); + ctx.execute_vfpu_unary_ct<67u, 124u, 1u, 0u>(); + ctx.execute_vfpu_unary_ct<99u, 92u, 1u, 0u>(); + ctx.execute_vfpu_unary_ct<99u, 99u, 1u, 2u>(); jump_target = ctx.gpr[31]; // nop local_pc = jump_target; @@ -9539,14 +9491,8 @@ L_08987DFC: ctx.pc = jump_target; return; L_08987E28: - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<67u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<67u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<99u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<99u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<67u, 67u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<99u, 99u, 1u, 2u>(); ctx.gpr[1] = (0u + static_cast(1)); { const bool branch_taken = ctx.gpr[24] != ctx.gpr[1]; // nop @@ -9556,34 +9502,13 @@ L_08987E28: goto L_08987E3C; } L_08987E3C: - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<3u, 4u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 4u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<124u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<3u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<3u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<3u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<92u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<35u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<60u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<67u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<67u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<67u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<99u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 3u, 4u, 0u>(); + ctx.execute_vfpu_unary_ct<3u, 124u, 1u, 0u>(); + ctx.execute_vfpu_unary_ct<3u, 3u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<35u, 92u, 1u, 0u>(); + ctx.execute_vfpu_unary_ct<67u, 60u, 1u, 0u>(); + ctx.execute_vfpu_unary_ct<67u, 67u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<99u, 28u, 1u, 0u>(); goto L_08987E58; L_08987E58: jump_target = ctx.gpr[31]; diff --git a/profiles/vcs/generated/generated_unit_0097.cpp b/profiles/vcs/generated/generated_unit_0097.cpp index 399a443..7caad1c 100644 --- a/profiles/vcs/generated/generated_unit_0097.cpp +++ b/profiles/vcs/generated/generated_unit_0097.cpp @@ -1882,11 +1882,7 @@ L_08988554: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2206,11 +2202,7 @@ L_08988848: } L_08988890: ctx.set_vfpu_scalar_bits_ct<32u>(aot_mem.aot_direct_load32(ctx.gpr[11] + static_cast(12))); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 32u, 1u, 1u>(); ctx.execute_vfpu_vcmp_ct<0u, 28u, 1u, 7u>(); if (((ctx.vfpu_ctrl[3] >> 0u) & 1u) != 0u) { aot_mem.aot_direct_store32(ctx.gpr[4] + static_cast(16), ctx.vfpu_scalar_bits_ct<0u>()); @@ -2257,10 +2249,7 @@ L_089888DC: } goto L_089888E8; L_089888E8: - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 28u, 1u, 0u>(); aot_mem.aot_direct_store32(ctx.gpr[4] + static_cast(16), ctx.vfpu_scalar_bits_ct<28u>()); { const bool branch_taken = 0u == 0u; ctx.gpr[12] = (ctx.gpr[12] - ctx.gpr[16]); @@ -2277,11 +2266,7 @@ L_08988900: ctx.gpr[8] = (aot_mem.aot_direct_load_word_left(ctx.gpr[12] + static_cast(11), ctx.gpr[8])); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[8]); ctx.execute_vfpu_vh2f_ct<1u, 32u, 1u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<0u, 28u, 1u, 2u>(); if (((ctx.vfpu_ctrl[3] >> 0u) & 1u) != 0u) { ctx.gpr[15] = (ctx.gpr[12] | 0u); @@ -2313,31 +2298,19 @@ L_0898892C: ctx.execute_vfpu_vh2f_ct<1u, 16u, 2u>(); ctx.execute_vfpu_vh2f_ct<2u, 80u, 2u>(); ctx.execute_vfpu_vdot_ct<16u, 1u, 2u, 4u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<16u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<16u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<16u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<16u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<16u, 16u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<16u, 16u, 1u, 23u>(); ctx.execute_vfpu_vocp_ct<16u, 16u, 1u>(); ctx.execute_vfpu_vcmp_ct<16u, 28u, 1u, 1u>(); { const bool branch_taken = ((ctx.vfpu_ctrl[3] >> 0u) & 1u) != 0u; - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<16u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<48u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<48u, 16u, 1u, 18u>(); if (branch_taken) { goto L_08988980; } goto L_0898897C; } L_0898897C: - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<48u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<48u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<48u, 48u, 1u, 16u>(); goto L_08988980; L_08988980: aot_mem.aot_direct_store32(ctx.gpr[4] + static_cast(0), ctx.vfpu_scalar_bits_ct<16u>()); @@ -2361,11 +2334,7 @@ L_08988998: } L_089889A8: ctx.set_vfpu_scalar_bits_ct<44u>(ctx.gpr[8]); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<12u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<44u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<12u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<12u, 12u, 44u, 1u, 2u>(); goto L_089889B0; L_089889B0: ctx.execute_vfpu_vcmp_ct<12u, 28u, 1u, 7u>(); @@ -2383,22 +2352,14 @@ L_089889BC: ctx.execute_vfpu_vh2f_ct<20u, 32u, 1u>(); ctx.execute_vfpu_vcmp_ct<20u, 28u, 1u, 1u>(); { const bool branch_taken = ((ctx.vfpu_ctrl[3] >> 0u) & 1u) != 0u; - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<20u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<52u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<52u, 20u, 0u, 1u, 1u>(); if (branch_taken) { goto L_089889DC; } goto L_089889D8; } L_089889D8: - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<52u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<20u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] / vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<20u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<20u, 52u, 20u, 1u, 3u>(); goto L_089889DC; L_089889DC: ctx.gpr[8] = (ctx.gpr[25] & 2u); @@ -2422,17 +2383,9 @@ L_089889E8: ctx.set_vfpu_scalar_bits_ct<96u>(ctx.gpr[9]); ctx.execute_vfpu_vh2f_ct<1u, 0u, 2u>(); ctx.execute_vfpu_vh2f_ct<2u, 64u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<2u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<2u, 2u, 1u, 3u, 1u>(); ctx.execute_vfpu_vscl_ct<2u, 2u, 20u, 3u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<2u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<1u, 1u, 2u, 3u, 0u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 12u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -2474,28 +2427,13 @@ L_08988A64: ctx.set_vfpu_scalar_bits_ct<96u>(ctx.gpr[9]); ctx.execute_vfpu_vh2f_ct<2u, 64u, 2u>(); ctx.execute_vfpu_vocp_ct<52u, 20u, 1u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<16u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<20u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<4u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<16u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<52u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<36u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<4u, 2u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 2u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<8u, 2u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<4u, 16u, 20u, 1u, 2u>(); + ctx.execute_vfpu_vec3_ct<36u, 16u, 52u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<8u, 4u, 2u, 18u>(); ctx.execute_vfpu_vscl_ct<4u, 8u, 48u, 2u>(); ctx.execute_vfpu_vscl_ct<2u, 2u, 4u, 4u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 36u, 4u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 4u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<2u, 4u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 4u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<1u, 1u, 2u, 4u, 0u>(); goto L_08988AA0; L_08988AA0: ctx.execute_vfpu_vscl_ct<1u, 1u, 12u, 4u>(); @@ -2548,11 +2486,7 @@ L_08988AB8: } L_08988B08: ctx.set_vfpu_scalar_bits_ct<32u>(aot_mem.aot_direct_load32(ctx.gpr[11] + static_cast(12))); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 32u, 1u, 1u>(); ctx.execute_vfpu_vcmp_ct<0u, 28u, 1u, 7u>(); if (((ctx.vfpu_ctrl[3] >> 0u) & 1u) != 0u) { aot_mem.aot_direct_store32(ctx.gpr[4] + static_cast(16), ctx.vfpu_scalar_bits_ct<0u>()); @@ -2599,10 +2533,7 @@ L_08988B54: } goto L_08988B60; L_08988B60: - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 28u, 1u, 0u>(); aot_mem.aot_direct_store32(ctx.gpr[4] + static_cast(16), ctx.vfpu_scalar_bits_ct<28u>()); { const bool branch_taken = 0u == 0u; ctx.gpr[12] = (ctx.gpr[12] - ctx.gpr[16]); @@ -2619,11 +2550,7 @@ L_08988B78: ctx.gpr[8] = (aot_mem.aot_direct_load_word_left(ctx.gpr[12] + static_cast(11), ctx.gpr[8])); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[8]); ctx.execute_vfpu_vh2f_ct<1u, 32u, 1u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<0u, 28u, 1u, 2u>(); if (((ctx.vfpu_ctrl[3] >> 0u) & 1u) != 0u) { ctx.gpr[15] = (ctx.gpr[12] | 0u); @@ -2655,31 +2582,19 @@ L_08988BA4: ctx.execute_vfpu_vh2f_ct<1u, 16u, 2u>(); ctx.execute_vfpu_vh2f_ct<2u, 80u, 2u>(); ctx.execute_vfpu_vdot_ct<16u, 1u, 2u, 4u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<16u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<16u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<16u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<16u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<16u, 16u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<16u, 16u, 1u, 23u>(); ctx.execute_vfpu_vocp_ct<16u, 16u, 1u>(); ctx.execute_vfpu_vcmp_ct<16u, 28u, 1u, 1u>(); { const bool branch_taken = ((ctx.vfpu_ctrl[3] >> 0u) & 1u) != 0u; - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<16u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<48u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<48u, 16u, 1u, 18u>(); if (branch_taken) { goto L_08988BF8; } goto L_08988BF4; } L_08988BF4: - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<48u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<48u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<48u, 48u, 1u, 16u>(); goto L_08988BF8; L_08988BF8: aot_mem.aot_direct_store32(ctx.gpr[4] + static_cast(0), ctx.vfpu_scalar_bits_ct<16u>()); @@ -2703,11 +2618,7 @@ L_08988C10: } L_08988C20: ctx.set_vfpu_scalar_bits_ct<44u>(ctx.gpr[8]); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<12u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<44u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<12u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<12u, 12u, 44u, 1u, 2u>(); goto L_08988C28; L_08988C28: ctx.execute_vfpu_vcmp_ct<12u, 28u, 1u, 7u>(); @@ -2725,22 +2636,14 @@ L_08988C34: ctx.execute_vfpu_vh2f_ct<20u, 32u, 1u>(); ctx.execute_vfpu_vcmp_ct<20u, 28u, 1u, 1u>(); { const bool branch_taken = ((ctx.vfpu_ctrl[3] >> 0u) & 1u) != 0u; - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<20u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<52u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<52u, 20u, 0u, 1u, 1u>(); if (branch_taken) { goto L_08988C54; } goto L_08988C50; } L_08988C50: - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<52u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<20u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] / vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<20u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<20u, 52u, 20u, 1u, 3u>(); goto L_08988C54; L_08988C54: ctx.gpr[8] = (ctx.gpr[25] & 2u); @@ -2764,22 +2667,11 @@ L_08988C60: ctx.set_vfpu_scalar_bits_ct<96u>(ctx.gpr[9]); ctx.execute_vfpu_vh2f_ct<1u, 0u, 2u>(); ctx.execute_vfpu_vh2f_ct<2u, 64u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<2u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<2u, 2u, 1u, 3u, 1u>(); ctx.execute_vfpu_vscl_ct<2u, 2u, 20u, 3u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<2u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<1u, 1u, 2u, 3u, 0u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 12u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<1u, 1u, 1u, 2u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2812,10 +2704,7 @@ L_08988CB4: goto L_08988CE0; } L_08988CE0: - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 4u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<3u, 4u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<3u, 1u, 4u, 0u>(); aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(8), ctx.gpr[31]); ctx.gpr[31] = (0x08988CF0u); // nop @@ -2823,10 +2712,7 @@ L_08988CE0: return; L_08988CF0: ctx.gpr[31] = (aot_mem.aot_direct_load32(ctx.gpr[29] + static_cast(8))); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<3u, 4u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 4u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<1u, 3u, 4u, 0u>(); goto L_08988CF8; L_08988CF8: ctx.gpr[8] = (aot_mem.aot_direct_load_word_right(ctx.gpr[12] + static_cast(0), ctx.gpr[8])); @@ -2836,10 +2722,7 @@ L_08988CF8: ctx.set_vfpu_scalar_bits_ct<64u>(ctx.gpr[8]); ctx.set_vfpu_scalar_bits_ct<96u>(ctx.gpr[9]); ctx.execute_vfpu_vh2f_ct<2u, 64u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<2u, 4u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<3u, 4u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<3u, 2u, 4u, 0u>(); aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(8), ctx.gpr[31]); ctx.gpr[31] = (0x08988D24u); // nop @@ -2847,33 +2730,15 @@ L_08988CF8: return; L_08988D24: ctx.gpr[31] = (aot_mem.aot_direct_load32(ctx.gpr[29] + static_cast(8))); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<3u, 4u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 4u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<2u, 3u, 4u, 0u>(); ctx.execute_vfpu_vocp_ct<52u, 20u, 1u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<16u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<20u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<4u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<16u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<52u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<36u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<4u, 2u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 2u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<8u, 2u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<4u, 16u, 20u, 1u, 2u>(); + ctx.execute_vfpu_vec3_ct<36u, 16u, 52u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<8u, 4u, 2u, 18u>(); ctx.execute_vfpu_vscl_ct<4u, 8u, 48u, 2u>(); ctx.execute_vfpu_vscl_ct<2u, 2u, 4u, 4u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 36u, 4u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 4u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<2u, 4u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 4u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<1u, 1u, 2u, 4u, 0u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 12u, 4u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -4579,30 +4444,10 @@ L_089899C0: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } diff --git a/profiles/vcs/generated/generated_unit_0098.cpp b/profiles/vcs/generated/generated_unit_0098.cpp index 6390f99..07e4f8a 100644 --- a/profiles/vcs/generated/generated_unit_0098.cpp +++ b/profiles/vcs/generated/generated_unit_0098.cpp @@ -7715,19 +7715,9 @@ L_0898E464: ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[13]; const float ft = ctx.fpr[26]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[13] = std::bit_cast(0x7FC00000u); else ctx.fpr[13] = fs * ft; } @@ -7745,19 +7735,9 @@ L_0898E498: ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.fpr[13] = std::bit_cast(std::bit_cast(ctx.fpr[13]) ^ 0x80000000u); @@ -7780,19 +7760,9 @@ L_0898E4E0: ctx.gpr[6] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[5]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[6]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); { const float fs = ctx.fpr[12]; const float ft = ctx.fpr[26]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[12] = std::bit_cast(0x7FC00000u); else ctx.fpr[12] = fs * ft; } @@ -7810,19 +7780,9 @@ L_0898E514: ctx.gpr[6] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[5]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[6]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.fpr[12] = std::bit_cast(std::bit_cast(ctx.fpr[12]) ^ 0x80000000u); diff --git a/profiles/vcs/generated/generated_unit_0101.cpp b/profiles/vcs/generated/generated_unit_0101.cpp index 3cb513c..9f2fffc 100644 --- a/profiles/vcs/generated/generated_unit_0101.cpp +++ b/profiles/vcs/generated/generated_unit_0101.cpp @@ -1424,11 +1424,7 @@ L_089988BC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[21] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); @@ -1478,10 +1474,7 @@ L_08998960: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -1509,7 +1502,7 @@ L_08998960: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[21] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); @@ -1621,11 +1614,7 @@ L_08998A10: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1756,33 +1745,8 @@ L_08998B70: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1840,11 +1804,7 @@ L_08998BB4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1874,11 +1834,7 @@ L_08998BCC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2378,15 +2334,8 @@ L_08998FA4: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -2401,15 +2350,8 @@ L_08998FC8: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -2424,17 +2366,9 @@ L_08998FEC: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - ctx.execute_vfpu_vrot(1u, 64u, 2u, 4u); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<33u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] / vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_vrot_ct<1u, 64u, 2u, 4u>(); + ctx.execute_vfpu_vec3_ct<0u, 33u, 1u, 1u, 3u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -2448,19 +2382,9 @@ L_08999014: ctx.gpr[5] = (std::bit_cast(ctx.fpr[13])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -2480,10 +2404,7 @@ L_08999040: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -2534,11 +2455,7 @@ L_0899906C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -2565,7 +2482,7 @@ L_0899906C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -2590,11 +2507,7 @@ L_0899906C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2618,11 +2531,7 @@ L_0899906C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2929,11 +2838,7 @@ L_089992B0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3179,17 +3084,9 @@ L_089994CC: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - ctx.execute_vfpu_vrot(1u, 64u, 2u, 4u); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<33u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] / vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_vrot_ct<1u, 64u, 2u, 4u>(); + ctx.execute_vfpu_vec3_ct<0u, 33u, 1u, 1u, 3u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[4]); ctx.gpr[31] = (0x0899952Cu); @@ -3242,11 +3139,7 @@ L_08999588: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3389,11 +3282,7 @@ L_0899967C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3918,11 +3807,7 @@ L_08999A60: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3959,11 +3844,7 @@ L_08999A60: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4000,11 +3881,7 @@ L_08999A60: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4186,11 +4063,7 @@ L_08999BB4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4335,15 +4208,8 @@ L_08999D28: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.fpr[14] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[16] + static_cast(312))); @@ -4411,11 +4277,7 @@ L_08999DB8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4585,11 +4447,7 @@ L_08999EC8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4687,11 +4545,7 @@ L_08999F90: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4723,10 +4577,7 @@ L_08999F90: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -4791,10 +4642,7 @@ L_0899A020: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -4819,7 +4667,7 @@ L_0899A020: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4838,10 +4686,7 @@ L_0899A020: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -4866,7 +4711,7 @@ L_0899A020: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4947,11 +4792,7 @@ L_0899A0B0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4983,10 +4824,7 @@ L_0899A0B0: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -5061,11 +4899,7 @@ L_0899A144: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5134,10 +4968,7 @@ L_0899A1E8: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -5162,7 +4993,7 @@ L_0899A1E8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5181,10 +5012,7 @@ L_0899A1E8: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -5209,7 +5037,7 @@ L_0899A1E8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5240,10 +5068,7 @@ L_0899A1E8: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -5459,11 +5284,7 @@ L_0899A3EC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5570,11 +5391,7 @@ L_0899A4A8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5612,11 +5429,7 @@ L_0899A4A8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -5635,10 +5448,7 @@ L_0899A4A8: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -5737,11 +5547,7 @@ L_0899A590: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5873,11 +5679,7 @@ L_0899A6B0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[18] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); @@ -5906,7 +5708,7 @@ L_0899A6B0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5925,10 +5727,7 @@ L_0899A6B0: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -5971,10 +5770,7 @@ L_0899A744: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -6022,10 +5818,7 @@ L_0899A794: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16181u << 16u); @@ -6063,10 +5856,7 @@ L_0899A7E4: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -6106,10 +5896,7 @@ L_0899A820: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -6159,11 +5946,7 @@ L_0899A868: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6191,10 +5974,7 @@ L_0899A868: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.fpr[12] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[28] + static_cast(-13636))); @@ -6255,11 +6035,7 @@ L_0899A8D0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6351,11 +6127,7 @@ L_0899A934: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6376,10 +6148,7 @@ L_0899A934: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -6422,11 +6191,7 @@ L_0899A934: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6468,7 +6233,7 @@ L_0899A934: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6793,11 +6558,7 @@ L_0899AC64: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[22] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6812,10 +6573,7 @@ L_0899AC64: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[24] = std::bit_cast(ctx.gpr[4]); ctx.fpr[12] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[28] + static_cast(-13632))); @@ -6889,11 +6647,7 @@ L_0899ACE4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6924,10 +6678,7 @@ L_0899ACE4: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7079,17 +6830,9 @@ L_0899AE60: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - ctx.execute_vfpu_vrot(1u, 64u, 2u, 4u); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<33u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] / vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_vrot_ct<1u, 64u, 2u, 4u>(); + ctx.execute_vfpu_vec3_ct<0u, 33u, 1u, 1u, 3u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[22] = std::bit_cast(ctx.gpr[4]); ctx.gpr[31] = (0x0899AEA4u); @@ -7170,11 +6913,7 @@ L_0899AEC4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7357,11 +7096,7 @@ L_0899B048: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7377,10 +7112,7 @@ L_0899B048: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[4]); ctx.gpr[31] = (0x0899B09Cu); @@ -8586,10 +8318,7 @@ L_0899BA7C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); diff --git a/profiles/vcs/generated/generated_unit_0102.cpp b/profiles/vcs/generated/generated_unit_0102.cpp index 6f0dac1..a2f3c45 100644 --- a/profiles/vcs/generated/generated_unit_0102.cpp +++ b/profiles/vcs/generated/generated_unit_0102.cpp @@ -1117,11 +1117,7 @@ L_0899C208: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2133,19 +2129,9 @@ L_0899C95C: ctx.gpr[5] = (std::bit_cast(ctx.fpr[16])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[15] = std::bit_cast(ctx.gpr[4]); ctx.fpr[16] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[28] + static_cast(6824))); @@ -2492,11 +2478,7 @@ L_0899CC40: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(144)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2512,10 +2494,7 @@ L_0899CC40: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (15948u << 16u); @@ -2589,11 +2568,7 @@ L_0899CC98: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(144)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2765,19 +2740,9 @@ L_0899CDF4: ctx.gpr[5] = (std::bit_cast(ctx.fpr[13])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); @@ -2837,11 +2802,7 @@ L_0899CDF4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(256)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -2916,19 +2877,9 @@ L_0899CEBC: ctx.gpr[5] = (std::bit_cast(ctx.fpr[13])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[24] = std::bit_cast(ctx.gpr[4]); ctx.gpr[21] = (ctx.gpr[29] + static_cast(304)); @@ -2993,11 +2944,7 @@ L_0899CF38: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(240)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3061,10 +3008,7 @@ L_0899CF9C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((!(std::isnan(ctx.fpr[26]) || std::isnan(ctx.fpr[13])) && ctx.fpr[26] == ctx.fpr[13])) ? 0x00800000u : 0u); @@ -3114,15 +3058,8 @@ L_0899D010: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); { const bool branch_taken = 0u == 0u; @@ -3153,11 +3090,7 @@ L_0899D034: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(368)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3173,10 +3106,7 @@ L_0899D034: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (ctx.gpr[20] << 5u); @@ -3205,10 +3135,7 @@ L_0899D090: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.fpr[12] = ctx.fpr[12] + ctx.fpr[13]; @@ -3545,19 +3472,9 @@ L_0899D364: ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[15] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[16] + static_cast(304))); @@ -3729,15 +3646,8 @@ L_0899D544: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (ctx.gpr[20] << 5u); @@ -3763,15 +3673,8 @@ L_0899D58C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (ctx.gpr[20] << 5u); @@ -3791,15 +3694,8 @@ L_0899D5D0: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.fpr[14] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[16] + static_cast(120))); @@ -3808,15 +3704,8 @@ L_0899D5D0: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[15] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[13]; const float ft = ctx.fpr[15]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[13] = std::bit_cast(0x7FC00000u); else ctx.fpr[13] = fs * ft; } @@ -3826,15 +3715,8 @@ L_0899D5D0: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[15] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[14])); @@ -3842,15 +3724,8 @@ L_0899D5D0: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[15]; const float ft = ctx.fpr[12]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[12] = std::bit_cast(0x7FC00000u); else ctx.fpr[12] = fs * ft; } @@ -3860,15 +3735,8 @@ L_0899D5D0: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[15] = std::bit_cast(ctx.gpr[4]); { const std::uint32_t aot_run_words[3]{std::bit_cast(ctx.fpr[13]), std::bit_cast(ctx.fpr[12]), std::bit_cast(ctx.fpr[15])}; @@ -3924,11 +3792,7 @@ L_0899D5D0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4016,11 +3880,7 @@ L_0899D72C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(432)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4036,10 +3896,7 @@ L_0899D72C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[4]); ctx.gpr[19] = (0u | 40u); @@ -4111,11 +3968,7 @@ L_0899D7C0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(496)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -4148,10 +4001,7 @@ L_0899D7C0: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -4195,11 +4045,7 @@ L_0899D7C0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4339,11 +4185,7 @@ L_0899D8E4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(544)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4430,11 +4272,7 @@ L_0899D9A0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(400)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4464,10 +4302,7 @@ L_0899D9A0: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (16256u << 16u); @@ -4509,7 +4344,7 @@ L_0899D9A0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(448)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -4533,7 +4368,7 @@ L_0899D9A0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(464)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4558,17 +4393,9 @@ L_0899D9A0: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - ctx.execute_vfpu_vrot(1u, 64u, 2u, 4u); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<33u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] / vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_vrot_ct<1u, 64u, 2u, 4u>(); + ctx.execute_vfpu_vec3_ct<0u, 33u, 1u, 1u, 3u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[5]); { const float fs = ctx.fpr[12]; const float ft = ctx.fpr[13]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[12] = std::bit_cast(0x7FC00000u); else ctx.fpr[12] = fs * ft; } @@ -4640,11 +4467,7 @@ L_0899DAC4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(608)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4660,10 +4483,7 @@ L_0899DAC4: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16128u << 16u); @@ -4700,11 +4520,7 @@ L_0899DB0C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(400)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4720,10 +4536,7 @@ L_0899DB0C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (0u | 40u); @@ -4786,11 +4599,7 @@ L_0899DB78: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(400)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4844,11 +4653,7 @@ L_0899DBBC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(624)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4864,10 +4669,7 @@ L_0899DBBC: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[20] = ctx.fpr[20] - ctx.fpr[12]; @@ -4955,11 +4757,7 @@ L_0899DCA0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(688)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -4975,10 +4773,7 @@ L_0899DCA0: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.fpr[13] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[29] + static_cast(96))); @@ -5144,10 +4939,7 @@ L_0899DE18: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -5235,17 +5027,9 @@ L_0899DEC4: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - ctx.execute_vfpu_vrot(1u, 64u, 2u, 4u); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<33u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] / vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_vrot_ct<1u, 64u, 2u, 4u>(); + ctx.execute_vfpu_vec3_ct<0u, 33u, 1u, 1u, 3u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (aot_mem.aot_direct_load32(ctx.gpr[28] + static_cast(5876))); @@ -5319,11 +5103,7 @@ L_0899DF68: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5541,10 +5321,7 @@ L_0899E134: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[12] <= ctx.fpr[28])) ? 0x00800000u : 0u); @@ -5636,7 +5413,7 @@ L_0899E1E4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[20] = (ctx.gpr[29] + static_cast(224)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); @@ -5655,10 +5432,7 @@ L_0899E1E4: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -5767,11 +5541,7 @@ L_0899E2CC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5786,10 +5556,7 @@ L_0899E2CC: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[22] = std::bit_cast(ctx.gpr[4]); ctx.gpr[31] = (0x0899E314u); @@ -6382,11 +6149,7 @@ L_0899E710: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6440,11 +6203,7 @@ L_0899E710: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6496,11 +6255,7 @@ L_0899E7A8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[23] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[23] + static_cast(0); @@ -6532,10 +6287,7 @@ L_0899E7A8: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -6560,11 +6312,7 @@ L_0899E7A8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[23] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6579,10 +6327,7 @@ L_0899E7A8: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[30] <= ctx.fpr[12])) ? 0x00800000u : 0u); @@ -7052,11 +6797,7 @@ L_0899EB6C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(336)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7072,10 +6813,7 @@ L_0899EB6C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[30] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[30]; const float ft = ctx.fpr[12]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[30] = std::bit_cast(0x7FC00000u); else ctx.fpr[30] = fs * ft; } @@ -7105,10 +6843,7 @@ L_0899EBB8: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[30] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[30]; const float ft = ctx.fpr[12]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[30] = std::bit_cast(0x7FC00000u); else ctx.fpr[30] = fs * ft; } @@ -7253,19 +6988,9 @@ L_0899ECD8: { const float vfpu_constant = std::bit_cast(0x3FC90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 64u, 32u, 1u, 2u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[24] = std::bit_cast(ctx.gpr[4]); ctx.fpr[12] = std::bit_cast(std::bit_cast(ctx.fpr[20]) & 0x7FFFFFFFu); @@ -7337,19 +7062,9 @@ L_0899ED74: ctx.gpr[5] = (std::bit_cast(ctx.fpr[17])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[20] < ctx.fpr[24])) ? 0x00800000u : 0u); @@ -7498,19 +7213,9 @@ L_0899EEC0: ctx.gpr[5] = (std::bit_cast(ctx.fpr[13])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[15] = std::bit_cast(ctx.gpr[4]); ctx.fpr[12] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[17] + static_cast(24))); @@ -7621,19 +7326,9 @@ L_0899EF6C: ctx.gpr[5] = (std::bit_cast(ctx.fpr[16])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[4]); ctx.fpr[22] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[17] + static_cast(36))); @@ -7985,11 +7680,7 @@ L_0899F264: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[23] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8078,11 +7769,7 @@ L_0899F2F0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8168,11 +7855,7 @@ L_0899F370: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8329,11 +8012,7 @@ L_0899F498: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(368)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8349,10 +8028,7 @@ L_0899F498: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[4]); ctx.fpr[22] = ctx.fpr[26] - ctx.fpr[20]; @@ -8437,11 +8113,7 @@ L_0899F574: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(480)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -8460,10 +8132,7 @@ L_0899F574: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -8494,11 +8163,7 @@ L_0899F574: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(464)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -8514,10 +8179,7 @@ L_0899F574: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[7] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[7]); ctx.fpr[12] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[29] + static_cast(96))); @@ -8654,11 +8316,7 @@ L_0899F6D4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(544)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -8677,10 +8335,7 @@ L_0899F6D4: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -8779,11 +8434,7 @@ L_0899F7B4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(176)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8802,10 +8453,7 @@ L_0899F7B4: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -8830,11 +8478,7 @@ L_0899F7B4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[16] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); @@ -8850,10 +8494,7 @@ L_0899F7B4: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[28] + static_cast(8068), 0u); @@ -8873,17 +8514,9 @@ L_0899F7B4: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - ctx.execute_vfpu_vrot(1u, 64u, 2u, 4u); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<33u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] / vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_vrot_ct<1u, 64u, 2u, 4u>(); + ctx.execute_vfpu_vec3_ct<0u, 33u, 1u, 1u, 3u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (aot_mem.aot_direct_load32(ctx.gpr[28] + static_cast(5876))); @@ -8950,11 +8583,7 @@ L_0899F8A0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8995,11 +8624,7 @@ L_0899F8C0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9014,10 +8639,7 @@ L_0899F8C0: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (15395u << 16u); @@ -9861,19 +9483,9 @@ L_0899FF78: ctx.gpr[6] = (std::bit_cast(ctx.fpr[14])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[6]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.fpr[14] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[17] + static_cast(296))); diff --git a/profiles/vcs/generated/generated_unit_0103.cpp b/profiles/vcs/generated/generated_unit_0103.cpp index 24cacc8..12d835b 100644 --- a/profiles/vcs/generated/generated_unit_0103.cpp +++ b/profiles/vcs/generated/generated_unit_0103.cpp @@ -882,15 +882,8 @@ L_089A00B8: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.fpr[14] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[17] + static_cast(120))); @@ -899,15 +892,8 @@ L_089A00B8: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[15] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[13]; const float ft = ctx.fpr[15]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[13] = std::bit_cast(0x7FC00000u); else ctx.fpr[13] = fs * ft; } @@ -917,15 +903,8 @@ L_089A00B8: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[15] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[14])); @@ -933,15 +912,8 @@ L_089A00B8: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[15]; const float ft = ctx.fpr[12]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[12] = std::bit_cast(0x7FC00000u); else ctx.fpr[12] = fs * ft; } @@ -951,15 +923,8 @@ L_089A00B8: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[15] = std::bit_cast(ctx.gpr[4]); { const std::uint32_t aot_run_words[3]{std::bit_cast(ctx.fpr[13]), std::bit_cast(ctx.fpr[12]), std::bit_cast(ctx.fpr[15])}; @@ -1015,11 +980,7 @@ L_089A00B8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1074,11 +1035,7 @@ L_089A00B8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1132,11 +1089,7 @@ L_089A0218: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[22] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1168,10 +1121,7 @@ L_089A0218: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -1197,11 +1147,7 @@ L_089A0218: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[22] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1216,10 +1162,7 @@ L_089A0218: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[13] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[17] + static_cast(16))); @@ -1349,10 +1292,7 @@ L_089A0350: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (15523u << 16u); @@ -1518,11 +1458,7 @@ L_089A0414: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(272)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1538,10 +1474,7 @@ L_089A0414: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[14]; const float ft = ctx.fpr[15]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[14] = std::bit_cast(0x7FC00000u); else ctx.fpr[14] = fs * ft; } @@ -1704,19 +1637,9 @@ L_089A05B8: { const float vfpu_constant = std::bit_cast(0x3FC90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 64u, 32u, 1u, 2u>(); ctx.gpr[6] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[6]); aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(924), std::bit_cast(ctx.fpr[12])); @@ -1845,7 +1768,7 @@ L_089A069C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(336)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1864,10 +1787,7 @@ L_089A069C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -1892,7 +1812,7 @@ L_089A069C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(352)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1911,10 +1831,7 @@ L_089A069C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -2052,15 +1969,8 @@ L_089A07E8: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.fpr[12] = std::bit_cast(std::bit_cast(ctx.fpr[13]) & 0x7FFFFFFFu); @@ -2069,19 +1979,9 @@ L_089A07E8: { const float vfpu_constant = std::bit_cast(0x3FC90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 64u, 32u, 1u, 2u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[12] <= ctx.fpr[20])) ? 0x00800000u : 0u); @@ -2108,15 +2008,8 @@ L_089A0850: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[13] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[20] + static_cast(32))); @@ -2136,15 +2029,8 @@ L_089A0884: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[13] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[20] + static_cast(20))); @@ -2253,15 +2139,8 @@ L_089A094C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[4]); ctx.fpr[12] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[17] + static_cast(8))); @@ -2275,10 +2154,7 @@ L_089A094C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((!(std::isnan(ctx.fpr[12]) || std::isnan(ctx.fpr[24])) && ctx.fpr[12] == ctx.fpr[24])) ? 0x00800000u : 0u); @@ -2423,15 +2299,8 @@ L_089A0A60: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[4]); ctx.fpr[12] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[16] + static_cast(8))); @@ -2445,10 +2314,7 @@ L_089A0A60: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((!(std::isnan(ctx.fpr[12]) || std::isnan(ctx.fpr[24])) && ctx.fpr[12] == ctx.fpr[24])) ? 0x00800000u : 0u); @@ -2540,19 +2406,9 @@ L_089A0B34: ctx.gpr[5] = (std::bit_cast(ctx.fpr[13])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); ctx.fpr[12] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[18] + static_cast(24))); @@ -2872,19 +2728,9 @@ L_089A0E10: ctx.gpr[5] = (std::bit_cast(ctx.fpr[14])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[26] = std::bit_cast(ctx.gpr[4]); ctx.fpr[22] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[18] + static_cast(36))); @@ -3293,15 +3139,8 @@ L_089A1184: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.fpr[14] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[17] + static_cast(120))); @@ -3310,15 +3149,8 @@ L_089A1184: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[15] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[13]; const float ft = ctx.fpr[15]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[13] = std::bit_cast(0x7FC00000u); else ctx.fpr[13] = fs * ft; } @@ -3328,15 +3160,8 @@ L_089A1184: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[15] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[14])); @@ -3344,15 +3169,8 @@ L_089A1184: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[15]; const float ft = ctx.fpr[12]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[12] = std::bit_cast(0x7FC00000u); else ctx.fpr[12] = fs * ft; } @@ -3362,15 +3180,8 @@ L_089A1184: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[15] = std::bit_cast(ctx.gpr[4]); { const std::uint32_t aot_run_words[3]{std::bit_cast(ctx.fpr[13]), std::bit_cast(ctx.fpr[12]), std::bit_cast(ctx.fpr[15])}; @@ -3433,11 +3244,7 @@ L_089A124C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[22] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3493,15 +3300,8 @@ L_089A124C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (std::bit_cast(ctx.fpr[20])); @@ -3509,15 +3309,8 @@ L_089A124C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[5]); { const float fs = ctx.fpr[12]; const float ft = ctx.fpr[14]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[12] = std::bit_cast(0x7FC00000u); else ctx.fpr[12] = fs * ft; } @@ -3527,15 +3320,8 @@ L_089A124C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (std::bit_cast(ctx.fpr[20])); @@ -3543,15 +3329,8 @@ L_089A124C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[5]); { const float fs = ctx.fpr[14]; const float ft = ctx.fpr[13]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[13] = std::bit_cast(0x7FC00000u); else ctx.fpr[13] = fs * ft; } @@ -3561,15 +3340,8 @@ L_089A124C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[5]); { const std::uint32_t aot_run_words[3]{std::bit_cast(ctx.fpr[12]), std::bit_cast(ctx.fpr[13]), std::bit_cast(ctx.fpr[14])}; @@ -3610,11 +3382,7 @@ L_089A124C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[30] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3638,15 +3406,8 @@ L_089A124C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[20])); @@ -3654,15 +3415,8 @@ L_089A124C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[13]; const float ft = ctx.fpr[14]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[13] = std::bit_cast(0x7FC00000u); else ctx.fpr[13] = fs * ft; } @@ -3870,11 +3624,7 @@ L_089A15AC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(416)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4171,11 +3921,7 @@ L_089A1864: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4886,11 +4632,7 @@ L_089A1EF0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4927,11 +4669,7 @@ L_089A1EF0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4962,10 +4700,7 @@ L_089A1EF0: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -5117,11 +4852,7 @@ L_089A2048: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(96)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5251,11 +4982,7 @@ L_089A212C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(160)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5533,28 +5260,7 @@ L_089A22C4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5792,10 +5498,7 @@ L_089A2494: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -5830,10 +5533,7 @@ L_089A2494: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -5858,7 +5558,7 @@ L_089A2494: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5890,10 +5590,7 @@ L_089A2494: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -5918,7 +5615,7 @@ L_089A2494: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5949,10 +5646,7 @@ L_089A2494: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -6119,11 +5813,7 @@ L_089A26B8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6390,11 +6080,7 @@ L_089A28BC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6478,11 +6164,7 @@ L_089A294C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6628,11 +6310,7 @@ L_089A2A6C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6663,10 +6341,7 @@ L_089A2A6C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7093,15 +6768,8 @@ L_089A2E90: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[18] + static_cast(96), std::bit_cast(ctx.fpr[13])); @@ -7110,15 +6778,8 @@ L_089A2E90: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[18] + static_cast(104), std::bit_cast(ctx.fpr[13])); @@ -7202,11 +6863,7 @@ L_089A2F80: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7238,10 +6895,7 @@ L_089A2F80: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7266,7 +6920,7 @@ L_089A2F80: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7302,7 +6956,7 @@ L_089A2F80: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7333,10 +6987,7 @@ L_089A2F80: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7593,15 +7244,8 @@ L_089A3294: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[18] + static_cast(96), std::bit_cast(ctx.fpr[13])); @@ -7610,15 +7254,8 @@ L_089A3294: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[18] + static_cast(104), std::bit_cast(ctx.fpr[13])); @@ -7665,11 +7302,7 @@ L_089A3324: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7701,10 +7334,7 @@ L_089A3324: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7729,7 +7359,7 @@ L_089A3324: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7765,7 +7395,7 @@ L_089A3324: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7796,10 +7426,7 @@ L_089A3324: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -8415,19 +8042,9 @@ L_089A38FC: ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[14] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[23] + static_cast(304))); @@ -9093,11 +8710,7 @@ L_089A3DC0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9313,11 +8926,7 @@ L_089A3F38: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(368)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -9354,11 +8963,7 @@ L_089A3F38: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9389,10 +8994,7 @@ L_089A3F38: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); diff --git a/profiles/vcs/generated/generated_unit_0104.cpp b/profiles/vcs/generated/generated_unit_0104.cpp index 9559d5c..f2e82fd 100644 --- a/profiles/vcs/generated/generated_unit_0104.cpp +++ b/profiles/vcs/generated/generated_unit_0104.cpp @@ -916,11 +916,7 @@ L_089A4164: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1064,11 +1060,7 @@ L_089A4280: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1101,10 +1093,7 @@ L_089A4280: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -1201,11 +1190,7 @@ L_089A4388: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1220,10 +1205,7 @@ L_089A4388: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16460u << 16u); @@ -1351,11 +1333,7 @@ L_089A446C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1387,10 +1365,7 @@ L_089A446C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -1444,11 +1419,7 @@ L_089A44F8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1480,10 +1451,7 @@ L_089A44F8: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -1533,11 +1501,7 @@ L_089A456C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[20] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); @@ -1569,10 +1533,7 @@ L_089A456C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -1631,11 +1592,7 @@ L_089A45F8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1712,10 +1669,7 @@ L_089A4654: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -1740,7 +1694,7 @@ L_089A4654: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1759,10 +1713,7 @@ L_089A4654: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -1787,7 +1738,7 @@ L_089A4654: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1903,28 +1854,7 @@ L_089A4754: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(176)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2214,7 +2144,7 @@ L_089A4944: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2245,10 +2175,7 @@ L_089A4944: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -2273,7 +2200,7 @@ L_089A4944: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2304,10 +2231,7 @@ L_089A4944: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -2590,7 +2514,7 @@ L_089A4B70: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2621,10 +2545,7 @@ L_089A4B70: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -2649,7 +2570,7 @@ L_089A4B70: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2680,10 +2601,7 @@ L_089A4B70: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -2823,7 +2741,7 @@ L_089A4CA4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2854,10 +2772,7 @@ L_089A4CA4: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -2882,7 +2797,7 @@ L_089A4CA4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2913,10 +2828,7 @@ L_089A4CA4: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -2952,15 +2864,8 @@ L_089A4D54: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16076u << 16u); @@ -2972,15 +2877,8 @@ L_089A4D54: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[13])); @@ -3005,15 +2903,8 @@ L_089A4D54: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (std::bit_cast(ctx.fpr[13])); @@ -3051,11 +2942,7 @@ L_089A4D54: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3359,7 +3246,7 @@ L_089A506C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -3392,10 +3279,7 @@ L_089A506C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -3420,11 +3304,7 @@ L_089A506C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3455,10 +3335,7 @@ L_089A506C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -3483,7 +3360,7 @@ L_089A506C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3514,10 +3391,7 @@ L_089A506C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -3588,11 +3462,7 @@ L_089A5140: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3653,7 +3523,7 @@ L_089A51EC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3685,10 +3555,7 @@ L_089A51EC: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -3719,15 +3586,8 @@ L_089A5260: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.fpr[14] = std::bit_cast(std::bit_cast(ctx.fpr[22])); @@ -3805,11 +3665,7 @@ L_089A529C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(208)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3828,10 +3684,7 @@ L_089A529C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -3869,10 +3722,7 @@ L_089A529C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -3939,7 +3789,7 @@ L_089A53CC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3970,10 +3820,7 @@ L_089A53CC: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -4096,11 +3943,7 @@ L_089A54F4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4375,11 +4218,7 @@ L_089A570C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4460,10 +4299,7 @@ L_089A57A0: ctx.fpr[12] = std::sqrt(ctx.fpr[12]); ctx.gpr[2] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<1u>(ctx.gpr[2]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<1u, 1u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -4535,11 +4371,7 @@ L_089A57A0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[11] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4972,11 +4804,7 @@ L_089A5B00: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5118,11 +4946,7 @@ L_089A5C0C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5289,11 +5113,7 @@ L_089A5D3C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6352,11 +6172,7 @@ L_089A64F0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6389,10 +6205,7 @@ L_089A64F0: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -6437,7 +6250,7 @@ L_089A64F0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[7] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); @@ -6456,10 +6269,7 @@ L_089A64F0: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -6484,7 +6294,7 @@ L_089A64F0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6764,11 +6574,7 @@ L_089A674C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6801,10 +6607,7 @@ L_089A674C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -6849,7 +6652,7 @@ L_089A674C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(160)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6868,10 +6671,7 @@ L_089A674C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -6958,11 +6758,7 @@ L_089A6838: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7146,11 +6942,7 @@ L_089A6988: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(224)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -7185,10 +6977,7 @@ L_089A6988: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7232,7 +7021,7 @@ L_089A6988: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[8] = (ctx.gpr[29] + static_cast(272)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[8] + static_cast(0); @@ -7251,10 +7040,7 @@ L_089A6988: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7279,11 +7065,7 @@ L_089A6988: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[9] = (ctx.gpr[29] + static_cast(240)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[9] + static_cast(0); @@ -7325,11 +7107,7 @@ L_089A6988: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7361,10 +7139,7 @@ L_089A6988: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7407,11 +7182,7 @@ L_089A6988: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7551,11 +7322,7 @@ L_089A6BA0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(320)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7588,10 +7355,7 @@ L_089A6BA0: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7636,7 +7400,7 @@ L_089A6BA0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[7] = (ctx.gpr[29] + static_cast(352)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); @@ -7655,10 +7419,7 @@ L_089A6BA0: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7683,7 +7444,7 @@ L_089A6BA0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7884,11 +7645,7 @@ L_089A6DB8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -7915,7 +7672,7 @@ L_089A6DB8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(96)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -7940,11 +7697,7 @@ L_089A6DB8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -7969,11 +7722,7 @@ L_089A6DB8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7990,10 +7739,7 @@ L_089A6DB8: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16968u << 16u); @@ -8053,11 +7799,7 @@ L_089A6EB0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8424,15 +8166,8 @@ L_089A7128: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[22]; const float ft = ctx.fpr[13]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[13] = std::bit_cast(0x7FC00000u); else ctx.fpr[13] = fs * ft; } @@ -8443,15 +8178,8 @@ L_089A7128: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[22]; const float ft = ctx.fpr[14]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[14] = std::bit_cast(0x7FC00000u); else ctx.fpr[14] = fs * ft; } @@ -8579,11 +8307,7 @@ L_089A725C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8713,10 +8437,7 @@ L_089A7320: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -8751,15 +8472,8 @@ L_089A7380: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[20]; const float ft = ctx.fpr[14]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[14] = std::bit_cast(0x7FC00000u); else ctx.fpr[14] = fs * ft; } @@ -8771,15 +8485,8 @@ L_089A7380: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[20]; const float ft = ctx.fpr[14]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[12] = std::bit_cast(0x7FC00000u); else ctx.fpr[12] = fs * ft; } @@ -8840,11 +8547,7 @@ L_089A7400: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8926,10 +8629,7 @@ L_089A7478: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -8999,10 +8699,7 @@ L_089A7518: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9106,11 +8803,7 @@ L_089A7588: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -9191,10 +8884,7 @@ L_089A7634: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9361,10 +9051,7 @@ L_089A7704: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9420,11 +9107,7 @@ L_089A7738: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[21] = (ctx.gpr[29] + static_cast(224)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); @@ -9522,11 +9205,7 @@ L_089A77BC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9590,11 +9269,7 @@ L_089A7834: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[22] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9856,10 +9531,7 @@ L_089A79CC: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -9911,15 +9583,8 @@ L_089A7A44: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[20]; const float ft = ctx.fpr[14]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[14] = std::bit_cast(0x7FC00000u); else ctx.fpr[14] = fs * ft; } @@ -9931,15 +9596,8 @@ L_089A7A44: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[20]; const float ft = ctx.fpr[14]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[12] = std::bit_cast(0x7FC00000u); else ctx.fpr[12] = fs * ft; } @@ -9971,15 +9629,8 @@ L_089A7AB4: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[20]; const float ft = ctx.fpr[14]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[14] = std::bit_cast(0x7FC00000u); else ctx.fpr[14] = fs * ft; } @@ -9991,15 +9642,8 @@ L_089A7AB4: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[20]; const float ft = ctx.fpr[14]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[12] = std::bit_cast(0x7FC00000u); else ctx.fpr[12] = fs * ft; } @@ -10252,11 +9896,7 @@ L_089A7C84: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -10667,10 +10307,7 @@ L_089A7F20: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -10706,10 +10343,7 @@ L_089A7F20: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -10768,7 +10402,7 @@ L_089A7FA8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(240)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -10815,7 +10449,7 @@ L_089A7FD4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(240)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -10830,10 +10464,7 @@ L_089A7FD4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(224)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); diff --git a/profiles/vcs/generated/generated_unit_0105.cpp b/profiles/vcs/generated/generated_unit_0105.cpp index 63b5261..618b8df 100644 --- a/profiles/vcs/generated/generated_unit_0105.cpp +++ b/profiles/vcs/generated/generated_unit_0105.cpp @@ -1041,10 +1041,7 @@ L_089A8008: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -1176,10 +1173,7 @@ L_089A8114: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (16025u << 16u); @@ -1265,24 +1259,10 @@ L_089A81A8: { const float vfpu_constant = std::bit_cast(0x3FC90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 64u, 32u, 1u, 2u>(); + ctx.execute_vfpu_vec3_ct<64u, 32u, 0u, 1u, 1u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<64u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[13] = std::bit_cast(std::bit_cast(ctx.fpr[12])); @@ -1343,7 +1323,7 @@ L_089A8214: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -1463,7 +1443,7 @@ L_089A82B4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -1495,10 +1475,7 @@ L_089A82B4: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -1527,10 +1504,7 @@ L_089A82B4: ctx.fpr[13] = std::bit_cast(ctx.gpr[6]); ctx.gpr[6] = (std::bit_cast(ctx.fpr[13])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[6]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -1555,10 +1529,7 @@ L_089A82B4: ctx.fpr[12] = std::bit_cast(std::bit_cast(ctx.fpr[12]) & 0x7FFFFFFFu); ctx.gpr[6] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[6]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -1591,10 +1562,7 @@ L_089A836C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -1616,10 +1584,7 @@ L_089A838C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -1673,7 +1638,7 @@ L_089A83E8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1704,10 +1669,7 @@ L_089A83E8: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -1732,7 +1694,7 @@ L_089A83E8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1794,17 +1756,9 @@ L_089A8464: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - ctx.execute_vfpu_vrot(1u, 64u, 2u, 4u); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<33u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] / vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_vrot_ct<1u, 64u, 2u, 4u>(); + ctx.execute_vfpu_vec3_ct<0u, 33u, 1u, 1u, 3u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[4]); ctx.gpr[31] = (0x089A8514u); @@ -1893,11 +1847,7 @@ L_089A8578: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(160)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2138,11 +2088,7 @@ L_089A871C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(256)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -2376,11 +2322,7 @@ L_089A8880: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(272)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2425,11 +2367,7 @@ L_089A88A4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(288)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2463,10 +2401,7 @@ L_089A88C0: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -2512,11 +2447,7 @@ L_089A88E8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -2566,11 +2497,7 @@ L_089A88E8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2599,10 +2526,7 @@ L_089A88E8: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16640u << 16u); @@ -2761,33 +2685,8 @@ L_089A8A60: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2827,30 +2726,10 @@ L_089A8A88: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -3380,11 +3259,7 @@ L_089A8E04: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3462,28 +3337,7 @@ L_089A8E6C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3589,11 +3443,7 @@ L_089A8EDC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3671,28 +3521,7 @@ L_089A8F44: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3775,11 +3604,7 @@ L_089A8F9C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -3795,10 +3620,7 @@ L_089A8F9C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (ctx.gpr[29] + static_cast(112)); @@ -3820,11 +3642,7 @@ L_089A8F9C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3840,10 +3658,7 @@ L_089A8F9C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[12] <= ctx.fpr[13])) ? 0x00800000u : 0u); @@ -4475,11 +4290,7 @@ L_089A93F0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5373,28 +5184,7 @@ L_089A99DC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5636,11 +5426,7 @@ L_089A9B5C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5695,28 +5481,7 @@ L_089A9B5C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5889,11 +5654,7 @@ L_089A9CE4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[23] = (ctx.gpr[29] + static_cast(224)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[23] + static_cast(0); @@ -5949,28 +5710,7 @@ L_089A9CE4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[23] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6013,11 +5753,7 @@ L_089A9D34: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[23] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6046,10 +5782,7 @@ L_089A9D34: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[22] = std::bit_cast(ctx.gpr[4]); { const bool branch_taken = 0u == 0u; @@ -6098,11 +5831,7 @@ L_089A9D90: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[23] = (ctx.gpr[29] + static_cast(240)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[23] + static_cast(0); @@ -6158,28 +5887,7 @@ L_089A9D90: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[23] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6222,11 +5930,7 @@ L_089A9DE0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[23] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6255,10 +5959,7 @@ L_089A9DE0: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[4]); { const bool branch_taken = 0u == 0u; @@ -6307,11 +6008,7 @@ L_089A9E3C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[21] = (ctx.gpr[29] + static_cast(256)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); @@ -6367,28 +6064,7 @@ L_089A9E3C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6431,11 +6107,7 @@ L_089A9E8C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6464,10 +6136,7 @@ L_089A9E8C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); { const bool branch_taken = 0u == 0u; @@ -6843,28 +6512,7 @@ L_089AA0C4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7256,11 +6904,7 @@ L_089AA3A0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[16] = (ctx.gpr[29] + static_cast(144)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); @@ -7579,28 +7223,7 @@ L_089AA520: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8253,28 +7876,7 @@ L_089AA968: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -11198,11 +10800,7 @@ L_089ABE00: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -11415,11 +11013,7 @@ L_089ABF54: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0106.cpp b/profiles/vcs/generated/generated_unit_0106.cpp index 1e32a11..2952c48 100644 --- a/profiles/vcs/generated/generated_unit_0106.cpp +++ b/profiles/vcs/generated/generated_unit_0106.cpp @@ -1216,11 +1216,7 @@ L_089AC0A8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1436,11 +1432,7 @@ L_089AC1FC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1646,11 +1638,7 @@ L_089AC340: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1815,28 +1803,7 @@ L_089AC448: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(144)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1873,11 +1840,7 @@ L_089AC448: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(160)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1916,10 +1879,7 @@ L_089AC4EC: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (16153u << 16u); @@ -1963,11 +1923,7 @@ L_089AC4EC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(176)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -3672,11 +3628,7 @@ L_089ACEEC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4783,11 +4735,7 @@ L_089AD690: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -5179,11 +5127,7 @@ L_089AD97C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5237,11 +5181,7 @@ L_089AD97C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(96)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5755,28 +5695,7 @@ L_089ADE28: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6614,11 +6533,7 @@ L_089AE47C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7678,11 +7593,7 @@ L_089AEC64: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8686,11 +8597,7 @@ L_089AF3C4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -10453,33 +10360,8 @@ L_089AFFC4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(32)); ctx.pc = 0x089B0000u; return; } diff --git a/profiles/vcs/generated/generated_unit_0107.cpp b/profiles/vcs/generated/generated_unit_0107.cpp index 7468330..3df58e8 100644 --- a/profiles/vcs/generated/generated_unit_0107.cpp +++ b/profiles/vcs/generated/generated_unit_0107.cpp @@ -4160,11 +4160,7 @@ L_089B1374: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(224)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4359,28 +4355,7 @@ L_089B14B4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4829,11 +4804,7 @@ L_089B17D4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[18] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); @@ -5535,11 +5506,7 @@ L_089B1B64: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[18] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); @@ -5594,28 +5561,7 @@ L_089B1B64: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5690,28 +5636,7 @@ L_089B1BC0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5918,11 +5843,7 @@ L_089B1CF8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6088,11 +6009,7 @@ L_089B1DE8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6613,11 +6530,7 @@ L_089B209C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6914,28 +6827,7 @@ L_089B2288: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7290,11 +7182,7 @@ L_089B250C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(144)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7310,10 +7198,7 @@ L_089B250C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16480u << 16u); @@ -7922,28 +7807,7 @@ L_089B2958: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8289,11 +8153,7 @@ L_089B2BFC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(144)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8309,10 +8169,7 @@ L_089B2BFC: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16480u << 16u); @@ -9043,10 +8900,7 @@ L_089B3108: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (15948u << 16u); diff --git a/profiles/vcs/generated/generated_unit_0108.cpp b/profiles/vcs/generated/generated_unit_0108.cpp index bf271c3..a1c3cc9 100644 --- a/profiles/vcs/generated/generated_unit_0108.cpp +++ b/profiles/vcs/generated/generated_unit_0108.cpp @@ -1302,11 +1302,7 @@ L_089B429C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2931,10 +2927,7 @@ L_089B4E24: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (15523u << 16u); @@ -3010,11 +3003,7 @@ L_089B4E98: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(176)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -3139,11 +3128,7 @@ L_089B4F64: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(208)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3153,10 +3138,7 @@ L_089B4F64: ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -3389,28 +3371,7 @@ L_089B50D0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(192)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3435,11 +3396,7 @@ L_089B50D0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(208)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -3496,28 +3453,7 @@ L_089B50D0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3540,11 +3476,7 @@ L_089B50D0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3583,11 +3515,7 @@ L_089B50D0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3626,11 +3554,7 @@ L_089B50D0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4950,11 +4874,7 @@ L_089B5BA8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5003,11 +4923,7 @@ L_089B5C14: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5061,11 +4977,7 @@ L_089B5C88: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5267,11 +5179,7 @@ L_089B5EEC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(176)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5338,11 +5246,7 @@ L_089B5F88: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(240)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5421,11 +5325,7 @@ L_089B6048: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(288)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5553,11 +5453,7 @@ L_089B6170: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(336)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5626,11 +5522,7 @@ L_089B6214: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(400)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5710,11 +5602,7 @@ L_089B62D8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(448)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5843,11 +5731,7 @@ L_089B63F4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(496)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5914,11 +5798,7 @@ L_089B6490: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(560)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5997,11 +5877,7 @@ L_089B6550: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(608)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6121,11 +5997,7 @@ L_089B6650: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(656)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6189,11 +6061,7 @@ L_089B66E0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(720)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6271,11 +6139,7 @@ L_089B679C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(768)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); diff --git a/profiles/vcs/generated/generated_unit_0109.cpp b/profiles/vcs/generated/generated_unit_0109.cpp index 6be8265..133325d 100644 --- a/profiles/vcs/generated/generated_unit_0109.cpp +++ b/profiles/vcs/generated/generated_unit_0109.cpp @@ -1314,11 +1314,7 @@ L_089B8234: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1898,11 +1894,7 @@ L_089B8678: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(96)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1918,10 +1910,7 @@ L_089B8678: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[16] + static_cast(21344), std::bit_cast(ctx.fpr[12])); @@ -2036,11 +2025,7 @@ L_089B875C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2056,10 +2041,7 @@ L_089B875C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[16] + static_cast(21348), std::bit_cast(ctx.fpr[12])); @@ -2174,11 +2156,7 @@ L_089B8840: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(160)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2194,10 +2172,7 @@ L_089B8840: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[16] + static_cast(21352), std::bit_cast(ctx.fpr[12])); @@ -2312,11 +2287,7 @@ L_089B8924: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(192)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2332,10 +2303,7 @@ L_089B8924: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[16] + static_cast(21356), std::bit_cast(ctx.fpr[12])); diff --git a/profiles/vcs/generated/generated_unit_0111.cpp b/profiles/vcs/generated/generated_unit_0111.cpp index 54334d2..20eb809 100644 --- a/profiles/vcs/generated/generated_unit_0111.cpp +++ b/profiles/vcs/generated/generated_unit_0111.cpp @@ -1855,15 +1855,8 @@ L_089C06EC: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16000u << 16u); diff --git a/profiles/vcs/generated/generated_unit_0112.cpp b/profiles/vcs/generated/generated_unit_0112.cpp index 5156eb8..e9fbdfa 100644 --- a/profiles/vcs/generated/generated_unit_0112.cpp +++ b/profiles/vcs/generated/generated_unit_0112.cpp @@ -7361,33 +7361,8 @@ L_089C71F8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7667,28 +7642,7 @@ L_089C7468: ctx.write_vfpu_vector_ct<39u, 4u>(vfpu_value); } ctx.gpr[6] = (std::bit_cast(ctx.fpr[20])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[6]); - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 4u); - ctx.read_vfpu_vector_ct<12u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 14u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<14u, 36u, 12u, 4u, 3u>(); ctx.vfpu_ctrl[1u] = 0x00000000u; ctx.execute_vfpu_vcmp_ct<14u, 0u, 4u, 3u>(); ctx.gpr[6] = (0u + static_cast(0)); @@ -7738,11 +7692,7 @@ L_089C74F4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(1072)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7758,10 +7708,7 @@ L_089C74F4: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[22] = std::bit_cast(ctx.gpr[4]); ctx.gpr[31] = (0x089C7528u); @@ -8648,11 +8595,7 @@ L_089C7C2C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8667,10 +8610,7 @@ L_089C7C2C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[12] < ctx.fpr[20])) ? 0x00800000u : 0u); diff --git a/profiles/vcs/generated/generated_unit_0113.cpp b/profiles/vcs/generated/generated_unit_0113.cpp index a78febb..5d7f57d 100644 --- a/profiles/vcs/generated/generated_unit_0113.cpp +++ b/profiles/vcs/generated/generated_unit_0113.cpp @@ -2669,11 +2669,7 @@ L_089C89A0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.fpr[13] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[4] + static_cast(8))); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); @@ -5503,11 +5499,7 @@ L_089C9DC4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(2208)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -6189,28 +6181,7 @@ L_089CA1AC: ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 4u); - ctx.read_vfpu_vector_ct<12u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 14u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<14u, 36u, 12u, 4u, 3u>(); ctx.vfpu_ctrl[1u] = 0x00000000u; ctx.execute_vfpu_vcmp_ct<14u, 0u, 4u, 3u>(); ctx.gpr[4] = (0u + static_cast(0)); @@ -6254,11 +6225,7 @@ L_089CA208: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(2240)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6274,10 +6241,7 @@ L_089CA208: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[4]); ctx.gpr[31] = (0x089CA240u); @@ -6412,11 +6376,7 @@ L_089CA2F4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(2256)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6788,30 +6748,10 @@ L_089CA5B8: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -8928,11 +8868,7 @@ L_089CB424: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8960,10 +8896,7 @@ L_089CB424: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (ctx.gpr[16] + static_cast(400)); @@ -8985,11 +8918,7 @@ L_089CB424: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9017,10 +8946,7 @@ L_089CB424: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (aot_mem.aot_direct_load8(ctx.gpr[16] + static_cast(476))); @@ -9364,11 +9290,7 @@ L_089CB6B0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9396,10 +9318,7 @@ L_089CB6B0: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[12] < ctx.fpr[20])) ? 0x00800000u : 0u); @@ -10227,15 +10146,8 @@ L_089CBCBC: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[12]; const float ft = ctx.fpr[24]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[12] = std::bit_cast(0x7FC00000u); else ctx.fpr[12] = fs * ft; } @@ -10244,15 +10156,8 @@ L_089CBCBC: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[13]; const float ft = ctx.fpr[24]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[13] = std::bit_cast(0x7FC00000u); else ctx.fpr[13] = fs * ft; } @@ -10261,15 +10166,8 @@ L_089CBCBC: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[14]; const float ft = ctx.fpr[24]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[14] = std::bit_cast(0x7FC00000u); else ctx.fpr[14] = fs * ft; } @@ -10278,15 +10176,8 @@ L_089CBCBC: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[15] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[15]; const float ft = ctx.fpr[30]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[15] = std::bit_cast(0x7FC00000u); else ctx.fpr[15] = fs * ft; } @@ -10341,15 +10232,8 @@ L_089CBD88: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[12]; const float ft = ctx.fpr[24]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[12] = std::bit_cast(0x7FC00000u); else ctx.fpr[12] = fs * ft; } @@ -10358,15 +10242,8 @@ L_089CBD88: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[13]; const float ft = ctx.fpr[24]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[13] = std::bit_cast(0x7FC00000u); else ctx.fpr[13] = fs * ft; } @@ -10375,15 +10252,8 @@ L_089CBD88: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[14]; const float ft = ctx.fpr[24]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[14] = std::bit_cast(0x7FC00000u); else ctx.fpr[14] = fs * ft; } @@ -10392,15 +10262,8 @@ L_089CBD88: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[15] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[15]; const float ft = ctx.fpr[30]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[15] = std::bit_cast(0x7FC00000u); else ctx.fpr[15] = fs * ft; } @@ -10431,15 +10294,8 @@ L_089CBE3C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[12]; const float ft = ctx.fpr[24]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[12] = std::bit_cast(0x7FC00000u); else ctx.fpr[12] = fs * ft; } @@ -10448,15 +10304,8 @@ L_089CBE3C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[13]; const float ft = ctx.fpr[24]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[13] = std::bit_cast(0x7FC00000u); else ctx.fpr[13] = fs * ft; } @@ -10465,15 +10314,8 @@ L_089CBE3C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[14]; const float ft = ctx.fpr[24]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[14] = std::bit_cast(0x7FC00000u); else ctx.fpr[14] = fs * ft; } @@ -10482,15 +10324,8 @@ L_089CBE3C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[15] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[15]; const float ft = ctx.fpr[30]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[15] = std::bit_cast(0x7FC00000u); else ctx.fpr[15] = fs * ft; } diff --git a/profiles/vcs/generated/generated_unit_0114.cpp b/profiles/vcs/generated/generated_unit_0114.cpp index 50a3c9a..835be10 100644 --- a/profiles/vcs/generated/generated_unit_0114.cpp +++ b/profiles/vcs/generated/generated_unit_0114.cpp @@ -1310,15 +1310,8 @@ L_089CC3C8: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16256u << 16u); @@ -1375,28 +1368,7 @@ L_089CC3C8: ctx.write_vfpu_vector_ct<39u, 4u>(vfpu_value); } ctx.gpr[4] = (std::bit_cast(ctx.fpr[28])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 4u); - ctx.read_vfpu_vector_ct<12u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 14u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<14u, 36u, 12u, 4u, 3u>(); ctx.vfpu_ctrl[1u] = 0x00000000u; ctx.execute_vfpu_vcmp_ct<14u, 0u, 4u, 3u>(); ctx.gpr[4] = (0u + static_cast(0)); @@ -1460,11 +1432,7 @@ L_089CC494: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1480,10 +1448,7 @@ L_089CC494: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[4]); ctx.gpr[31] = (0x089CC4C8u); @@ -1642,15 +1607,8 @@ L_089CC5F0: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[22] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[20]; const float ft = ctx.fpr[22]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[22] = std::bit_cast(0x7FC00000u); else ctx.fpr[22] = fs * ft; } @@ -1659,15 +1617,8 @@ L_089CC5F0: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[20]; const float ft = ctx.fpr[13]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[20] = std::bit_cast(0x7FC00000u); else ctx.fpr[20] = fs * ft; } @@ -2397,28 +2348,7 @@ L_089CCB54: ctx.write_vfpu_vector_ct<39u, 4u>(vfpu_value); } ctx.gpr[7] = (std::bit_cast(ctx.fpr[20])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[7]); - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 4u); - ctx.read_vfpu_vector_ct<12u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 14u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<14u, 36u, 12u, 4u, 3u>(); ctx.vfpu_ctrl[1u] = 0x00000000u; ctx.execute_vfpu_vcmp_ct<14u, 0u, 4u, 3u>(); ctx.gpr[7] = (0u + static_cast(0)); @@ -2469,11 +2399,7 @@ L_089CCBFC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2489,10 +2415,7 @@ L_089CCBFC: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[22] = std::bit_cast(ctx.gpr[4]); ctx.gpr[31] = (0x089CCC30u); @@ -3195,15 +3118,8 @@ L_089CD1A4: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16256u << 16u); @@ -3260,28 +3176,7 @@ L_089CD1A4: ctx.write_vfpu_vector_ct<39u, 4u>(vfpu_value); } ctx.gpr[4] = (std::bit_cast(ctx.fpr[30])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 4u); - ctx.read_vfpu_vector_ct<12u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 14u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<14u, 36u, 12u, 4u, 3u>(); ctx.vfpu_ctrl[1u] = 0x00000000u; ctx.execute_vfpu_vcmp_ct<14u, 0u, 4u, 3u>(); ctx.gpr[4] = (0u + static_cast(0)); @@ -3345,11 +3240,7 @@ L_089CD2A0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3365,10 +3256,7 @@ L_089CD2A0: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[4]); ctx.gpr[31] = (0x089CD2D4u); @@ -3479,15 +3367,8 @@ L_089CD380: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[22] = std::bit_cast(ctx.gpr[4]); ctx.fpr[20] = ctx.fpr[30] + ctx.fpr[13]; @@ -3497,15 +3378,8 @@ L_089CD380: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[20]; const float ft = ctx.fpr[13]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[20] = std::bit_cast(0x7FC00000u); else ctx.fpr[20] = fs * ft; } @@ -4181,28 +4055,7 @@ L_089CD8A4: ctx.write_vfpu_vector_ct<39u, 4u>(vfpu_value); } ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 4u); - ctx.read_vfpu_vector_ct<12u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 14u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<14u, 36u, 12u, 4u, 3u>(); ctx.vfpu_ctrl[1u] = 0x00000000u; ctx.execute_vfpu_vcmp_ct<14u, 0u, 4u, 3u>(); ctx.gpr[4] = (0u + static_cast(0)); @@ -4266,11 +4119,7 @@ L_089CD910: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4286,10 +4135,7 @@ L_089CD910: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[4]); ctx.gpr[31] = (0x089CD944u); @@ -6912,11 +6758,7 @@ L_089CEF94: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -6976,11 +6818,7 @@ L_089CF034: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[11] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7019,11 +6857,7 @@ L_089CF034: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[9] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7059,11 +6893,7 @@ L_089CF034: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[10] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7078,10 +6908,7 @@ L_089CF034: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[14] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[15] = std::bit_cast(ctx.gpr[14]); { const std::uint32_t vfpu_address = ctx.gpr[8] + static_cast(0); @@ -7102,11 +6929,7 @@ L_089CF034: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[9] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7147,11 +6970,7 @@ L_089CF034: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[10] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7187,11 +7006,7 @@ L_089CF034: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[11] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7214,11 +7029,7 @@ L_089CF034: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[10] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7254,11 +7065,7 @@ L_089CF034: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[11] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7310,11 +7117,7 @@ L_089CF034: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[11] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7824,11 +7627,7 @@ L_089CF4CC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[10] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7944,33 +7743,8 @@ L_089CF564: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0115.cpp b/profiles/vcs/generated/generated_unit_0115.cpp index b730f5d..e3b457d 100644 --- a/profiles/vcs/generated/generated_unit_0115.cpp +++ b/profiles/vcs/generated/generated_unit_0115.cpp @@ -2146,11 +2146,7 @@ L_089D0AA0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(96)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4538,19 +4534,9 @@ L_089D1ED8: { const float vfpu_constant = std::bit_cast(0x3FC90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 64u, 32u, 1u, 2u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; diff --git a/profiles/vcs/generated/generated_unit_0116.cpp b/profiles/vcs/generated/generated_unit_0116.cpp index 068775c..0bcb5d2 100644 --- a/profiles/vcs/generated/generated_unit_0116.cpp +++ b/profiles/vcs/generated/generated_unit_0116.cpp @@ -5762,10 +5762,7 @@ L_089D6C00: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.fpr[13] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[4] + static_cast(8))); diff --git a/profiles/vcs/generated/generated_unit_0117.cpp b/profiles/vcs/generated/generated_unit_0117.cpp index ae6cd48..98a1639 100644 --- a/profiles/vcs/generated/generated_unit_0117.cpp +++ b/profiles/vcs/generated/generated_unit_0117.cpp @@ -2185,15 +2185,8 @@ L_089D8A98: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); @@ -2201,15 +2194,8 @@ L_089D8A98: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[12] = ctx.fpr[13] - ctx.fpr[12]; @@ -2706,7 +2692,7 @@ L_089D8C74: goto L_089D8C7C; } L_089D8C7C: - ctx.execute_vfpu_matrix_init(0u, 4u, 3u); + ctx.execute_vfpu_matrix_init_ct<0u, 4u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3365,7 +3351,7 @@ L_089D9028: goto L_089D9030; } L_089D9030: - ctx.execute_vfpu_matrix_init(0u, 4u, 3u); + ctx.execute_vfpu_matrix_init_ct<0u, 4u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3800,10 +3786,7 @@ L_089D9334: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (15692u << 16u); @@ -5507,10 +5490,7 @@ L_089DA190: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (15820u << 16u); @@ -5619,10 +5599,7 @@ L_089DA244: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[24] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16015u << 16u); @@ -5833,10 +5810,7 @@ L_089DA430: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[28] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16448u << 16u); @@ -6149,15 +6123,8 @@ L_089DA6C8: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); ctx.fpr[13] = ctx.fpr[14] - ctx.fpr[24]; @@ -7012,19 +6979,9 @@ L_089DAE6C: { const float vfpu_constant = std::bit_cast(0x3FC90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 64u, 32u, 1u, 2u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[12] = std::bit_cast(std::bit_cast(ctx.fpr[12]) ^ 0x80000000u); @@ -8484,19 +8441,9 @@ L_089DB9F4: ctx.gpr[5] = (std::bit_cast(ctx.fpr[13])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[13] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[21] + static_cast(1620))); @@ -8903,33 +8850,8 @@ L_089DBCF0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[19] = (ctx.gpr[29] + static_cast(432)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); @@ -8980,33 +8902,8 @@ L_089DBCF0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[20] = (ctx.gpr[29] + static_cast(448)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); @@ -9102,33 +8999,8 @@ L_089DBDB0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[23] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9191,33 +9063,8 @@ L_089DBDB0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[23] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0118.cpp b/profiles/vcs/generated/generated_unit_0118.cpp index 4f24aeb..00b0c9e 100644 --- a/profiles/vcs/generated/generated_unit_0118.cpp +++ b/profiles/vcs/generated/generated_unit_0118.cpp @@ -1422,11 +1422,7 @@ L_089DC2D0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1468,28 +1464,7 @@ L_089DC2D0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<8u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 4u, 3u); - ctx.read_vfpu_vector_ct<8u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 4u, 8u, 3u, 3u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -1563,11 +1538,7 @@ L_089DC330: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[8] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3063,30 +3034,10 @@ L_089DD0F0: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -3124,30 +3075,10 @@ L_089DD0F0: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -3174,11 +3105,7 @@ L_089DD0F0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3411,11 +3338,7 @@ L_089DD31C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3452,11 +3375,7 @@ L_089DD31C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3491,10 +3410,7 @@ L_089DD31C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -3514,10 +3430,7 @@ L_089DD31C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -3572,11 +3485,7 @@ L_089DD3D8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -3592,10 +3501,7 @@ L_089DD3D8: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[6] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[6]); ctx.gpr[6] = (17096u << 16u); @@ -3758,11 +3664,7 @@ L_089DD524: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[7] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); @@ -3792,10 +3694,7 @@ L_089DD524: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[7] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[7]); ctx.gpr[7] = (16968u << 16u); @@ -4550,10 +4449,7 @@ L_089DDB3C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); diff --git a/profiles/vcs/generated/generated_unit_0119.cpp b/profiles/vcs/generated/generated_unit_0119.cpp index a86e4a6..5407bfb 100644 --- a/profiles/vcs/generated/generated_unit_0119.cpp +++ b/profiles/vcs/generated/generated_unit_0119.cpp @@ -4806,11 +4806,7 @@ L_089E1E9C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[9] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4838,10 +4834,7 @@ L_089E1E9C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[8] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[8]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[13] < ctx.fpr[12])) ? 0x00800000u : 0u); @@ -5173,11 +5166,7 @@ L_089E2178: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5214,10 +5203,7 @@ L_089E219C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[12] < ctx.fpr[20])) ? 0x00800000u : 0u); diff --git a/profiles/vcs/generated/generated_unit_0120.cpp b/profiles/vcs/generated/generated_unit_0120.cpp index 37bbb75..d9269c4 100644 --- a/profiles/vcs/generated/generated_unit_0120.cpp +++ b/profiles/vcs/generated/generated_unit_0120.cpp @@ -2121,11 +2121,7 @@ L_089E4784: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2831,11 +2827,7 @@ L_089E4CEC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3099,11 +3091,7 @@ L_089E4E94: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); diff --git a/profiles/vcs/generated/generated_unit_0121.cpp b/profiles/vcs/generated/generated_unit_0121.cpp index 153037c..759641f 100644 --- a/profiles/vcs/generated/generated_unit_0121.cpp +++ b/profiles/vcs/generated/generated_unit_0121.cpp @@ -4153,11 +4153,7 @@ L_089E9C28: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4173,10 +4169,7 @@ L_089E9C28: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16928u << 16u); @@ -4478,11 +4471,7 @@ L_089E9EA8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(208)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4498,10 +4487,7 @@ L_089E9EA8: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (17224u << 16u); @@ -4870,11 +4856,7 @@ L_089EA18C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(304)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4884,10 +4866,7 @@ L_089EA18C: ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -6976,15 +6955,8 @@ L_089EB344: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[20]; const float ft = ctx.fpr[13]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[13] = std::bit_cast(0x7FC00000u); else ctx.fpr[13] = fs * ft; } @@ -6996,15 +6968,8 @@ L_089EB344: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[20]; const float ft = ctx.fpr[14]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[12] = std::bit_cast(0x7FC00000u); else ctx.fpr[12] = fs * ft; } @@ -7305,11 +7270,7 @@ L_089EB658: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(416)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -7543,10 +7504,7 @@ L_089EB8DC: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); diff --git a/profiles/vcs/generated/generated_unit_0122.cpp b/profiles/vcs/generated/generated_unit_0122.cpp index 9eb5a93..db527f5 100644 --- a/profiles/vcs/generated/generated_unit_0122.cpp +++ b/profiles/vcs/generated/generated_unit_0122.cpp @@ -934,24 +934,10 @@ L_089EC124: { const float vfpu_constant = std::bit_cast(0x3FC90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 64u, 32u, 1u, 2u>(); + ctx.execute_vfpu_vec3_ct<64u, 32u, 0u, 1u, 1u>(); ctx.gpr[6] = (ctx.vfpu_scalar_bits_ct<64u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[6]); ctx.gpr[6] = (16384u << 16u); @@ -1621,11 +1607,7 @@ L_089EC66C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3448,33 +3430,8 @@ L_089ED8FC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4376,11 +4333,7 @@ L_089EE05C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4395,10 +4348,7 @@ L_089EE05C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[6] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[6]); ctx.fpr[13] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[5] + static_cast(8))); @@ -4506,33 +4456,8 @@ L_089EE154: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4765,11 +4690,7 @@ L_089EE37C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4875,33 +4796,8 @@ L_089EE404: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4925,11 +4821,7 @@ L_089EE404: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4980,33 +4872,8 @@ L_089EE404: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5051,11 +4918,7 @@ L_089EE474: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5108,11 +4971,7 @@ L_089EE4D0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5388,11 +5247,7 @@ L_089EE710: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5407,10 +5262,7 @@ L_089EE710: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); { const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); @@ -5431,11 +5283,7 @@ L_089EE710: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5450,10 +5298,7 @@ L_089EE710: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); { const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); @@ -5474,11 +5319,7 @@ L_089EE710: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[23] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5527,11 +5368,7 @@ L_089EE784: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6034,11 +5871,7 @@ L_089EEAFC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -6082,11 +5915,7 @@ L_089EEAFC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -6526,33 +6355,8 @@ L_089EEE54: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[19] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); @@ -6691,11 +6495,7 @@ L_089EEF78: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[20] = (ctx.gpr[29] + static_cast(144)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); @@ -6728,10 +6528,7 @@ L_089EEF78: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -6971,33 +6768,8 @@ L_089EF128: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7078,28 +6850,7 @@ L_089EF17C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[23] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7430,7 +7181,7 @@ L_089EF400: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7449,10 +7200,7 @@ L_089EF400: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7527,7 +7275,7 @@ L_089EF494: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7558,10 +7306,7 @@ L_089EF494: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7651,10 +7396,7 @@ L_089EF560: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (16015u << 16u); @@ -8792,11 +8534,7 @@ L_089EFDA4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[17] = (ctx.gpr[29] + static_cast(160)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); @@ -8938,28 +8676,7 @@ L_089EFE84: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(176)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8992,10 +8709,7 @@ L_089EFE84: L_089EFED0: ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<1u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<1u, 1u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); diff --git a/profiles/vcs/generated/generated_unit_0123.cpp b/profiles/vcs/generated/generated_unit_0123.cpp index be307a6..0467fbd 100644 --- a/profiles/vcs/generated/generated_unit_0123.cpp +++ b/profiles/vcs/generated/generated_unit_0123.cpp @@ -940,11 +940,7 @@ L_089F00C0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1153,33 +1149,8 @@ L_089F026C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(96)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2157,11 +2128,7 @@ L_089F09DC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2176,10 +2143,7 @@ L_089F09DC: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16968u << 16u); @@ -2205,10 +2169,7 @@ L_089F0A1C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (15820u << 16u); @@ -2276,28 +2237,7 @@ L_089F0A78: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3484,11 +3424,7 @@ L_089F135C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3503,10 +3439,7 @@ L_089F135C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[12] = std::bit_cast(std::bit_cast(ctx.fpr[12]) & 0x7FFFFFFFu); @@ -3943,15 +3876,8 @@ L_089F16B0: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[13]; const float ft = ctx.fpr[20]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[13] = std::bit_cast(0x7FC00000u); else ctx.fpr[13] = fs * ft; } @@ -3963,15 +3889,8 @@ L_089F16B0: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[14]; const float ft = ctx.fpr[20]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[14] = std::bit_cast(0x7FC00000u); else ctx.fpr[14] = fs * ft; } @@ -4049,15 +3968,8 @@ L_089F1794: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[13]; const float ft = ctx.fpr[20]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[13] = std::bit_cast(0x7FC00000u); else ctx.fpr[13] = fs * ft; } @@ -4069,15 +3981,8 @@ L_089F1794: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[14]; const float ft = ctx.fpr[20]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[12] = std::bit_cast(0x7FC00000u); else ctx.fpr[12] = fs * ft; } @@ -4461,11 +4366,7 @@ L_089F1AA0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4603,11 +4504,7 @@ L_089F1B64: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4762,11 +4659,7 @@ L_089F1C74: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(144)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4967,11 +4860,7 @@ L_089F1DE4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(272)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5805,11 +5694,7 @@ L_089F26F0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6822,11 +6707,7 @@ L_089F2F40: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7009,11 +6890,7 @@ L_089F3060: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7055,28 +6932,7 @@ L_089F3060: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<8u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 4u, 3u); - ctx.read_vfpu_vector_ct<8u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 4u, 8u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7737,10 +7593,7 @@ L_089F35D4: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7823,10 +7676,7 @@ L_089F3650: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7890,10 +7740,7 @@ L_089F36B0: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -8383,24 +8230,10 @@ L_089F3A6C: { const float vfpu_constant = std::bit_cast(0x3FC90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 64u, 32u, 1u, 2u>(); + ctx.execute_vfpu_vec3_ct<64u, 32u, 0u, 1u, 1u>(); ctx.gpr[8] = (ctx.vfpu_scalar_bits_ct<64u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[8]); aot_mem.aot_direct_store32(ctx.gpr[6] + static_cast(0), std::bit_cast(ctx.fpr[13])); @@ -8485,24 +8318,10 @@ L_089F3B14: { const float vfpu_constant = std::bit_cast(0x3FC90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 64u, 32u, 1u, 2u>(); + ctx.execute_vfpu_vec3_ct<64u, 32u, 0u, 1u, 1u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<64u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[7] + static_cast(0), std::bit_cast(ctx.fpr[12])); diff --git a/profiles/vcs/generated/generated_unit_0124.cpp b/profiles/vcs/generated/generated_unit_0124.cpp index 91ab954..a8edb5d 100644 --- a/profiles/vcs/generated/generated_unit_0124.cpp +++ b/profiles/vcs/generated/generated_unit_0124.cpp @@ -988,24 +988,10 @@ L_089F404C: { const float vfpu_constant = std::bit_cast(0x3FC90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 64u, 32u, 1u, 2u>(); + ctx.execute_vfpu_vec3_ct<64u, 32u, 0u, 1u, 1u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<64u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[5]); aot_mem.aot_direct_store32(ctx.gpr[17] + static_cast(0), std::bit_cast(ctx.fpr[13])); @@ -1090,24 +1076,10 @@ L_089F40F4: { const float vfpu_constant = std::bit_cast(0x3FC90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 64u, 32u, 1u, 2u>(); + ctx.execute_vfpu_vec3_ct<64u, 32u, 0u, 1u, 1u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<64u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[16] + static_cast(0), std::bit_cast(ctx.fpr[12])); @@ -1260,19 +1232,9 @@ L_089F4264: { const float vfpu_constant = std::bit_cast(0x3FC90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 64u, 32u, 1u, 2u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[16] + static_cast(8), std::bit_cast(ctx.fpr[12])); @@ -1342,19 +1304,9 @@ L_089F4300: { const float vfpu_constant = std::bit_cast(0x3FC90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 64u, 32u, 1u, 2u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[24] = std::bit_cast(ctx.gpr[4]); ctx.fpr[12] = ctx.fpr[20] - ctx.fpr[0]; @@ -1385,15 +1337,8 @@ L_089F435C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[12]; const float ft = ctx.fpr[13]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[13] = std::bit_cast(0x7FC00000u); else ctx.fpr[13] = fs * ft; } @@ -1906,15 +1851,8 @@ L_089F475C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.fpr[13] = std::bit_cast(std::bit_cast(ctx.fpr[13]) ^ 0x80000000u); @@ -1924,15 +1862,8 @@ L_089F475C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(16), std::bit_cast(ctx.fpr[13])); @@ -2096,19 +2027,9 @@ L_089F4964: { const float vfpu_constant = std::bit_cast(0x3FC90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 64u, 32u, 1u, 2u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[30] = std::bit_cast(ctx.gpr[4]); { const std::uint32_t aot_run_words[9]{std::bit_cast(ctx.fpr[24]), std::bit_cast(ctx.fpr[24]), std::bit_cast(ctx.fpr[26]), std::bit_cast(ctx.fpr[24]), std::bit_cast(ctx.fpr[26]), std::bit_cast(ctx.fpr[24]), std::bit_cast(ctx.fpr[26]), std::bit_cast(ctx.fpr[24]), std::bit_cast(ctx.fpr[24])}; @@ -2914,10 +2835,7 @@ L_089F50A0: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[9] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[9]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[12] < ctx.fpr[17])) ? 0x00800000u : 0u); @@ -6404,33 +6322,8 @@ L_089F6BE4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0125.cpp b/profiles/vcs/generated/generated_unit_0125.cpp index aace3f1..227fc98 100644 --- a/profiles/vcs/generated/generated_unit_0125.cpp +++ b/profiles/vcs/generated/generated_unit_0125.cpp @@ -2005,10 +2005,7 @@ L_089F8C4C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (ctx.gpr[18] + static_cast(32)); @@ -2022,10 +2019,7 @@ L_089F8C4C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[13] <= ctx.fpr[12])) ? 0x00800000u : 0u); @@ -4238,11 +4232,7 @@ L_089F9EE4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(208)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -4316,11 +4306,7 @@ L_089F9F94: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(256)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -4398,11 +4384,7 @@ L_089FA050: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(144)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -4647,11 +4629,7 @@ L_089FA26C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(288)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -4855,11 +4833,7 @@ L_089FA448: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(368)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -5496,11 +5470,7 @@ L_089FA9A4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(496)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6631,10 +6601,7 @@ L_089FB2A0: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16332u << 16u); @@ -6827,25 +6794,25 @@ L_089FB44C: ctx.gpr[7] = (aot_mem.aot_direct_load16(ctx.gpr[6] + static_cast(0))); ctx.gpr[7] = (ctx.gpr[7] & 65535u); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[7]); - ctx.execute_vfpu_vh2f(0u, 0u, 1u); + ctx.execute_vfpu_vh2f_ct<0u, 0u, 1u>(); ctx.gpr[7] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[7]); ctx.gpr[7] = (aot_mem.aot_direct_load16(ctx.gpr[6] + static_cast(2))); ctx.gpr[7] = (ctx.gpr[7] & 65535u); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[7]); - ctx.execute_vfpu_vh2f(0u, 0u, 1u); + ctx.execute_vfpu_vh2f_ct<0u, 0u, 1u>(); ctx.gpr[7] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[7]); ctx.gpr[7] = (aot_mem.aot_direct_load16(ctx.gpr[6] + static_cast(4))); ctx.gpr[7] = (ctx.gpr[7] & 65535u); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[7]); - ctx.execute_vfpu_vh2f(0u, 0u, 1u); + ctx.execute_vfpu_vh2f_ct<0u, 0u, 1u>(); ctx.gpr[7] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[7]); ctx.gpr[6] = (aot_mem.aot_direct_load16(ctx.gpr[6] + static_cast(6))); ctx.gpr[6] = (ctx.gpr[6] & 65535u); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[6]); - ctx.execute_vfpu_vh2f(0u, 0u, 1u); + ctx.execute_vfpu_vh2f_ct<0u, 0u, 1u>(); ctx.gpr[6] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[15] = std::bit_cast(ctx.gpr[6]); { const std::uint32_t aot_run_words[4]{std::bit_cast(ctx.fpr[12]), std::bit_cast(ctx.fpr[13]), std::bit_cast(ctx.fpr[14]), std::bit_cast(ctx.fpr[15])}; diff --git a/profiles/vcs/generated/generated_unit_0126.cpp b/profiles/vcs/generated/generated_unit_0126.cpp index 50d2b86..c6d0159 100644 --- a/profiles/vcs/generated/generated_unit_0126.cpp +++ b/profiles/vcs/generated/generated_unit_0126.cpp @@ -3654,33 +3654,8 @@ L_089FD46C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4580,11 +4555,7 @@ L_089FDAB4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[8] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[8] + static_cast(0); @@ -4864,33 +4835,8 @@ L_089FDCA8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[7] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); @@ -5028,33 +4974,8 @@ L_089FDDC0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5322,33 +5243,8 @@ L_089FDFC4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[7] = (ctx.gpr[29] + static_cast(208)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); @@ -5463,33 +5359,8 @@ L_089FE084: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[7] = (ctx.gpr[29] + static_cast(240)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); @@ -5646,33 +5517,8 @@ L_089FE1A8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[7] = (ctx.gpr[29] + static_cast(272)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); @@ -5810,33 +5656,8 @@ L_089FE2C0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(304)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6107,33 +5928,8 @@ L_089FE4D0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[7] = (ctx.gpr[29] + static_cast(400)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); @@ -6248,33 +6044,8 @@ L_089FE590: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[7] = (ctx.gpr[29] + static_cast(432)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); @@ -7182,10 +6953,7 @@ L_089FEDDC: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7222,10 +6990,7 @@ L_089FEE9C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7266,10 +7031,7 @@ L_089FEED8: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7294,7 +7056,7 @@ L_089FEED8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7333,7 +7095,7 @@ L_089FEED8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7389,10 +7151,7 @@ L_089FEF74: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7417,7 +7176,7 @@ L_089FEF74: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8017,10 +7776,7 @@ L_089FF4B4: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -8059,10 +7815,7 @@ L_089FF4B4: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -8087,7 +7840,7 @@ L_089FF4B4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8126,7 +7879,7 @@ L_089FF4B4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8163,10 +7916,7 @@ L_089FF4B4: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -8191,7 +7941,7 @@ L_089FF4B4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8652,11 +8402,7 @@ L_089FF9B8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8686,10 +8432,7 @@ L_089FF9B8: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[28] = std::bit_cast(ctx.gpr[5]); ctx.fpr[26] = std::bit_cast(0u); diff --git a/profiles/vcs/generated/generated_unit_0127.cpp b/profiles/vcs/generated/generated_unit_0127.cpp index fb9edaf..00caf66 100644 --- a/profiles/vcs/generated/generated_unit_0127.cpp +++ b/profiles/vcs/generated/generated_unit_0127.cpp @@ -1543,11 +1543,7 @@ L_08A0024C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(240)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1574,7 +1570,7 @@ L_08A0024C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(224)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1599,11 +1595,7 @@ L_08A0024C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(208)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1628,11 +1620,7 @@ L_08A0024C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1655,11 +1643,7 @@ L_08A0024C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[21] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); @@ -1708,28 +1692,7 @@ L_08A002E8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<8u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 4u, 3u); - ctx.read_vfpu_vector_ct<8u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 4u, 8u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4568,11 +4531,7 @@ L_08A0171C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4646,11 +4605,7 @@ L_08A01770: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -4726,11 +4681,7 @@ L_08A017CC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -5144,11 +5095,7 @@ L_08A01AAC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7051,11 +6998,7 @@ L_08A027B0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7498,15 +7441,8 @@ L_08A02ACC: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[28] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); @@ -7514,15 +7450,8 @@ L_08A02ACC: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[30] = std::bit_cast(ctx.gpr[4]); ctx.gpr[31] = (0x08A02B18u); @@ -7589,11 +7518,7 @@ L_08A02B6C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9508,28 +9433,7 @@ L_08A036C4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9605,28 +9509,7 @@ L_08A03704: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9695,28 +9578,7 @@ L_08A0374C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9936,11 +9798,7 @@ L_08A038D8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -9959,10 +9817,7 @@ L_08A038D8: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); diff --git a/profiles/vcs/generated/generated_unit_0129.cpp b/profiles/vcs/generated/generated_unit_0129.cpp index 8bc2a36..1107243 100644 --- a/profiles/vcs/generated/generated_unit_0129.cpp +++ b/profiles/vcs/generated/generated_unit_0129.cpp @@ -2222,11 +2222,7 @@ L_08A08D08: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2612,28 +2608,7 @@ L_08A09084: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<12u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 24u, 4u); - ctx.read_vfpu_vector_ct<12u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 4u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 3u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<3u, 24u, 12u, 4u, 4u>(); ctx.gpr[4] = (ctx.gpr[18] & 1u); ctx.gpr[5] = (ctx.gpr[4] << 9u); ctx.gpr[6] = (0u - ctx.gpr[5]); @@ -2661,7 +2636,7 @@ L_08A09084: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - ctx.execute_vfpu_vhdp(1u, 0u, 3u, 4u); + ctx.execute_vfpu_vhdp_ct<1u, 0u, 3u, 4u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.gpr[23] = (0u | 0u); @@ -2680,7 +2655,7 @@ L_08A090F0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - ctx.execute_vfpu_vhdp(1u, 0u, 3u, 4u); + ctx.execute_vfpu_vhdp_ct<1u, 0u, 3u, 4u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (0u | 0u); @@ -3210,7 +3185,7 @@ L_08A0958C: L_08A09594: ctx.gpr[8] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<124u>(ctx.gpr[8]); - ctx.execute_vfpu_vocp(125u, 124u, 1u); + ctx.execute_vfpu_vocp_ct<125u, 124u, 1u>(); ctx.set_vfpu_scalar_bits_ct<0u>(aot_mem.aot_direct_load32(ctx.gpr[4] + static_cast(0))); ctx.set_vfpu_scalar_bits_ct<32u>(aot_mem.aot_direct_load32(ctx.gpr[4] + static_cast(4))); ctx.set_vfpu_scalar_bits_ct<1u>(aot_mem.aot_direct_load32(ctx.gpr[4] + static_cast(8))); @@ -3225,16 +3200,8 @@ L_08A09594: ctx.execute_vfpu_vscl_ct<1u, 1u, 125u, 3u>(); ctx.execute_vfpu_vscl_ct<12u, 12u, 124u, 2u>(); ctx.execute_vfpu_vscl_ct<13u, 13u, 124u, 3u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 2u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<12u, 2u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 2u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<13u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 12u, 2u, 0u>(); + ctx.execute_vfpu_vec3_ct<1u, 1u, 13u, 3u, 0u>(); aot_mem.aot_direct_store32(ctx.gpr[6] + static_cast(0), ctx.vfpu_scalar_bits_ct<0u>()); aot_mem.aot_direct_store32(ctx.gpr[6] + static_cast(4), ctx.vfpu_scalar_bits_ct<32u>()); aot_mem.aot_direct_store32(ctx.gpr[6] + static_cast(8), ctx.vfpu_scalar_bits_ct<1u>()); @@ -3322,10 +3289,7 @@ L_08A096B0: goto L_08A096B8; } L_08A096B8: - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<52u, 4u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<3u, 4u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<3u, 52u, 4u, 0u>(); { const bool branch_taken = 0u == 0u; // nop if (branch_taken) { @@ -3358,10 +3322,7 @@ L_08A096D4: goto L_08A096DC; } L_08A096DC: - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<53u, 4u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<3u, 4u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<3u, 53u, 4u, 0u>(); { const bool branch_taken = 0u == 0u; // nop if (branch_taken) { @@ -3370,10 +3331,7 @@ L_08A096DC: goto L_08A096E8; } L_08A096E8: - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<54u, 4u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<3u, 4u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<3u, 54u, 4u, 0u>(); { const bool branch_taken = 0u == 0u; // nop if (branch_taken) { @@ -3382,10 +3340,7 @@ L_08A096E8: goto L_08A096F4; } L_08A096F4: - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<55u, 4u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<3u, 4u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<3u, 55u, 4u, 0u>(); goto L_08A096F8; L_08A096F8: ctx.gpr[4] = (ctx.gpr[17] & 1u); @@ -3409,7 +3364,7 @@ L_08A096F8: ctx.set_vfpu_scalar_bits_ct<0u>(aot_mem.aot_direct_load32(ctx.gpr[4] + static_cast(8))); ctx.set_vfpu_scalar_bits_ct<32u>(aot_mem.aot_direct_load32(ctx.gpr[4] + static_cast(12))); ctx.set_vfpu_scalar_bits_ct<64u>(aot_mem.aot_direct_load32(ctx.gpr[4] + static_cast(16))); - ctx.execute_vfpu_vhdp(1u, 0u, 3u, 4u); + ctx.execute_vfpu_vhdp_ct<1u, 0u, 3u, 4u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.gpr[21] = (0u | 0u); @@ -3422,7 +3377,7 @@ L_08A09764: ctx.set_vfpu_scalar_bits_ct<0u>(aot_mem.aot_direct_load32(ctx.gpr[20] + static_cast(8))); ctx.set_vfpu_scalar_bits_ct<32u>(aot_mem.aot_direct_load32(ctx.gpr[20] + static_cast(12))); ctx.set_vfpu_scalar_bits_ct<64u>(aot_mem.aot_direct_load32(ctx.gpr[20] + static_cast(16))); - ctx.execute_vfpu_vhdp(1u, 0u, 3u, 4u); + ctx.execute_vfpu_vhdp_ct<1u, 0u, 3u, 4u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (0u | 0u); @@ -3781,7 +3736,7 @@ L_08A09AC8: L_08A09AD0: ctx.gpr[8] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<124u>(ctx.gpr[8]); - ctx.execute_vfpu_vocp(125u, 124u, 1u); + ctx.execute_vfpu_vocp_ct<125u, 124u, 1u>(); { const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -3842,21 +3797,9 @@ L_08A09AD0: ctx.execute_vfpu_vscl_ct<12u, 12u, 124u, 3u>(); ctx.execute_vfpu_vscl_ct<13u, 13u, 124u, 4u>(); ctx.execute_vfpu_vscl_ct<14u, 14u, 124u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<12u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 4u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<13u, 4u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 4u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<2u, 2u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<14u, 2u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 2u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 12u, 3u, 0u>(); + ctx.execute_vfpu_vec3_ct<1u, 1u, 13u, 4u, 0u>(); + ctx.execute_vfpu_vec3_ct<2u, 2u, 14u, 2u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3901,37 +3844,8 @@ L_08A09B2C: ctx.set_vfpu_scalar_bits_ct<29u>(ctx.gpr[8]); ctx.set_vfpu_scalar_bits_ct<61u>(ctx.gpr[9]); ctx.set_vfpu_scalar_bits_ct<93u>(ctx.gpr[10]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<29u, 3u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<3u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(15u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 3u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<29u, 3u>(vfpu_d); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 48u, 4u); - ctx.read_vfpu_vector_ct<29u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 12u, vfpu_side); } + ctx.execute_vfpu_vi2f_ct<29u, 29u, 3u, 15u>(); + ctx.execute_vfpu_vtfm_ct<12u, 48u, 29u, 4u, 3u>(); ctx.execute_vfpu_vcmp_ct<12u, 31u, 4u, 7u>(); // vflush: architectural no-op that retains VFPU prefixes ctx.gpr[11] = (ctx.vfpu_scalar_bits_ct<131u>()); @@ -3943,37 +3857,8 @@ L_08A09B2C: ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[8]); ctx.set_vfpu_scalar_bits_ct<60u>(ctx.gpr[9]); ctx.set_vfpu_scalar_bits_ct<92u>(ctx.gpr[10]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<28u, 3u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<3u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(15u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 3u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 3u>(vfpu_d); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 48u, 4u); - ctx.read_vfpu_vector_ct<28u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 12u, vfpu_side); } + ctx.execute_vfpu_vi2f_ct<28u, 28u, 3u, 15u>(); + ctx.execute_vfpu_vtfm_ct<12u, 48u, 28u, 4u, 3u>(); ctx.execute_vfpu_vcmp_ct<12u, 31u, 4u, 7u>(); // vflush: architectural no-op that retains VFPU prefixes ctx.gpr[12] = (ctx.vfpu_scalar_bits_ct<131u>()); @@ -3995,15 +3880,7 @@ L_08A09BE4: ctx.set_vfpu_scalar_bits_ct<15u>(ctx.gpr[8]); ctx.set_vfpu_scalar_bits_ct<47u>(ctx.gpr[9]); ctx.set_vfpu_scalar_bits_ct<79u>(ctx.gpr[10]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<15u, 3u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<3u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(15u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 3u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<15u, 3u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<15u, 15u, 3u, 15u>(); ctx.execute_vfpu_vcmp_ct<15u, 28u, 3u, 1u>(); if (((ctx.vfpu_ctrl[3] >> 5u) & 1u) != 0u) { ctx.gpr[11] = ((ctx.gpr[16] >> 8u) & 0x000000FFu); @@ -4018,28 +3895,7 @@ L_08A09C10: } goto L_08A09C1C; L_08A09C1C: - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 48u, 4u); - ctx.read_vfpu_vector_ct<15u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 12u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<12u, 48u, 15u, 4u, 3u>(); ctx.execute_vfpu_vcmp_ct<12u, 31u, 4u, 7u>(); // vflush: architectural no-op that retains VFPU prefixes ctx.gpr[11] = (ctx.vfpu_scalar_bits_ct<131u>()); @@ -4088,15 +3944,9 @@ L_08A09C78: ctx.gpr[17] = (ctx.gpr[17] + static_cast(10)); ctx.gpr[5] = (0u + static_cast(0)); ctx.gpr[4] = (ctx.gpr[17] + static_cast(-20)); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<29u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<29u, 28u, 3u, 0u>(); if (ctx.gpr[17] != ctx.gpr[18]) { - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<15u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 15u, 3u, 0u>(); goto L_08A09BE4; } goto L_08A09C94; @@ -4116,15 +3966,9 @@ L_08A09CA4: ctx.gpr[19] = (ctx.gpr[19] ^ 1u); ctx.gpr[17] = (ctx.gpr[17] + static_cast(10)); ctx.gpr[5] = (ctx.gpr[5] + static_cast(1)); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<29u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<29u, 28u, 3u, 0u>(); if (ctx.gpr[17] != ctx.gpr[18]) { - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<15u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 15u, 3u, 0u>(); goto L_08A09BE4; } goto L_08A09CBC; @@ -4162,28 +4006,7 @@ L_08A09CF4: ctx.set_vfpu_scalar_bits_ct<29u>(aot_mem.aot_direct_load32(ctx.gpr[4] + static_cast(8))); ctx.set_vfpu_scalar_bits_ct<61u>(aot_mem.aot_direct_load32(ctx.gpr[4] + static_cast(12))); ctx.set_vfpu_scalar_bits_ct<93u>(aot_mem.aot_direct_load32(ctx.gpr[4] + static_cast(16))); - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 48u, 4u); - ctx.read_vfpu_vector_ct<29u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 12u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<12u, 48u, 29u, 4u, 3u>(); ctx.execute_vfpu_vcmp_ct<12u, 31u, 4u, 7u>(); // vflush: architectural no-op that retains VFPU prefixes ctx.gpr[11] = (ctx.vfpu_scalar_bits_ct<131u>()); @@ -4192,28 +4015,7 @@ L_08A09CF4: ctx.set_vfpu_scalar_bits_ct<28u>(aot_mem.aot_direct_load32(ctx.gpr[4] + static_cast(28))); ctx.set_vfpu_scalar_bits_ct<60u>(aot_mem.aot_direct_load32(ctx.gpr[4] + static_cast(32))); ctx.set_vfpu_scalar_bits_ct<92u>(aot_mem.aot_direct_load32(ctx.gpr[4] + static_cast(36))); - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 48u, 4u); - ctx.read_vfpu_vector_ct<28u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 12u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<12u, 48u, 28u, 4u, 3u>(); ctx.execute_vfpu_vcmp_ct<12u, 31u, 4u, 7u>(); // vflush: architectural no-op that retains VFPU prefixes ctx.gpr[12] = (ctx.vfpu_scalar_bits_ct<131u>()); @@ -4246,28 +4048,7 @@ L_08A09D98: } goto L_08A09DA4; L_08A09DA4: - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 48u, 4u); - ctx.read_vfpu_vector_ct<15u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 12u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<12u, 48u, 15u, 4u, 3u>(); ctx.execute_vfpu_vcmp_ct<12u, 31u, 4u, 7u>(); // vflush: architectural no-op that retains VFPU prefixes ctx.gpr[11] = (ctx.vfpu_scalar_bits_ct<131u>()); @@ -4303,72 +4084,9 @@ L_08A09DD4: goto L_08A09DEC; } L_08A09DEC: - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 52u, 4u); - ctx.read_vfpu_vector_ct<15u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 12u, vfpu_side); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 52u, 4u); - ctx.read_vfpu_vector_ct<28u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 13u, vfpu_side); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 52u, 4u); - ctx.read_vfpu_vector_ct<29u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 14u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<12u, 52u, 15u, 4u, 3u>(); + ctx.execute_vfpu_vtfm_ct<13u, 52u, 28u, 4u, 3u>(); + ctx.execute_vfpu_vtfm_ct<14u, 52u, 29u, 4u, 3u>(); ctx.execute_vfpu_vcmp_ct<12u, 31u, 4u, 7u>(); // vflush: architectural no-op that retains VFPU prefixes ctx.gpr[8] = (ctx.vfpu_scalar_bits_ct<131u>()); @@ -4400,15 +4118,9 @@ L_08A09E40: ctx.gpr[17] = (ctx.gpr[17] + static_cast(20)); ctx.gpr[5] = (0u + static_cast(0)); ctx.gpr[4] = (ctx.gpr[17] + static_cast(-40)); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<29u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<29u, 28u, 3u, 0u>(); if (ctx.gpr[17] != ctx.gpr[18]) { - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<15u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 15u, 3u, 0u>(); goto L_08A09D7C; } goto L_08A09E58; @@ -4427,15 +4139,9 @@ L_08A09E60: L_08A09E68: ctx.gpr[17] = (ctx.gpr[17] + static_cast(20)); ctx.gpr[5] = (ctx.gpr[5] + static_cast(1)); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<29u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<29u, 28u, 3u, 0u>(); if (ctx.gpr[17] != ctx.gpr[18]) { - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<15u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 15u, 3u, 0u>(); goto L_08A09D7C; } goto L_08A09E7C; @@ -5695,11 +5401,7 @@ L_08A0A8E8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7424,11 +7126,7 @@ L_08A0B7F0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); diff --git a/profiles/vcs/generated/generated_unit_0130.cpp b/profiles/vcs/generated/generated_unit_0130.cpp index ad784e4..3ddb851 100644 --- a/profiles/vcs/generated/generated_unit_0130.cpp +++ b/profiles/vcs/generated/generated_unit_0130.cpp @@ -1506,11 +1506,7 @@ L_08A0C578: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1584,11 +1580,7 @@ L_08A0C5D8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1930,11 +1922,7 @@ L_08A0C824: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1964,10 +1952,7 @@ L_08A0C824: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[6] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[6]); ctx.gpr[6] = (16128u << 16u); @@ -1977,10 +1962,7 @@ L_08A0C824: ctx.gpr[23] = (std::bit_cast(ctx.fpr[14])); ctx.gpr[6] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[6]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -2188,28 +2170,7 @@ L_08A0C98C: ctx.write_vfpu_vector_ct<39u, 4u>(vfpu_value); } ctx.gpr[4] = (std::bit_cast(ctx.fpr[24])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 4u); - ctx.read_vfpu_vector_ct<12u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 14u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<14u, 36u, 12u, 4u, 3u>(); ctx.vfpu_ctrl[1u] = 0x00000000u; ctx.execute_vfpu_vcmp_ct<14u, 0u, 4u, 3u>(); ctx.gpr[4] = (0u + static_cast(0)); @@ -2436,11 +2397,7 @@ L_08A0CB40: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -2470,10 +2427,7 @@ L_08A0CB40: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[7] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[7]); ctx.gpr[7] = (16128u << 16u); @@ -2483,10 +2437,7 @@ L_08A0CB40: ctx.gpr[23] = (std::bit_cast(ctx.fpr[14])); ctx.gpr[7] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[7]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -2694,28 +2645,7 @@ L_08A0CCA4: ctx.write_vfpu_vector_ct<39u, 4u>(vfpu_value); } ctx.gpr[4] = (std::bit_cast(ctx.fpr[24])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 4u); - ctx.read_vfpu_vector_ct<12u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 14u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<14u, 36u, 12u, 4u, 3u>(); ctx.vfpu_ctrl[1u] = 0x00000000u; ctx.execute_vfpu_vcmp_ct<14u, 0u, 4u, 3u>(); ctx.gpr[4] = (0u + static_cast(0)); @@ -3229,11 +3159,7 @@ L_08A0D068: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3296,11 +3222,7 @@ L_08A0D068: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3429,11 +3351,7 @@ L_08A0D164: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3476,11 +3394,7 @@ L_08A0D164: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3613,11 +3527,7 @@ L_08A0D248: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(208)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3730,11 +3640,7 @@ L_08A0D2EC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(240)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3844,11 +3750,7 @@ L_08A0D38C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(288)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3891,11 +3793,7 @@ L_08A0D38C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(272)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4005,11 +3903,7 @@ L_08A0D454: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(352)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4052,11 +3946,7 @@ L_08A0D454: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(336)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4165,11 +4055,7 @@ L_08A0D53C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(400)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4356,11 +4242,7 @@ L_08A0D660: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(432)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4611,11 +4493,7 @@ L_08A0D898: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -4645,10 +4523,7 @@ L_08A0D898: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[7] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[7]); ctx.gpr[7] = (16153u << 16u); @@ -4691,11 +4566,7 @@ L_08A0D898: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4793,11 +4664,7 @@ L_08A0D990: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(176)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5108,30 +4975,10 @@ L_08A0DB88: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -6710,11 +6557,7 @@ L_08A0E7D8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6724,10 +6567,7 @@ L_08A0E7D8: ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -7870,33 +7710,8 @@ L_08A0F150: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9516,33 +9331,8 @@ L_08A0FCB0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0131.cpp b/profiles/vcs/generated/generated_unit_0131.cpp index 55b73ce..da328af 100644 --- a/profiles/vcs/generated/generated_unit_0131.cpp +++ b/profiles/vcs/generated/generated_unit_0131.cpp @@ -993,10 +993,7 @@ L_08A10160: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1154,33 +1151,8 @@ L_08A10230: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(272)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1280,33 +1252,8 @@ L_08A1029C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(272)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1405,33 +1352,8 @@ L_08A10304: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(272)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1523,33 +1445,8 @@ L_08A10360: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(272)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1996,33 +1893,8 @@ L_08A106E8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(352)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -2106,33 +1978,8 @@ L_08A1073C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(352)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2232,33 +2079,8 @@ L_08A107A8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(352)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -2342,33 +2164,8 @@ L_08A107FC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(352)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2467,33 +2264,8 @@ L_08A10864: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(352)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -2577,33 +2349,8 @@ L_08A108B8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(352)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2697,33 +2444,8 @@ L_08A10914: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(352)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -2807,33 +2529,8 @@ L_08A10968: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(352)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2981,33 +2678,8 @@ L_08A10A04: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(368)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -3091,33 +2763,8 @@ L_08A10A58: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(368)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3217,33 +2864,8 @@ L_08A10AC4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(368)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -3327,33 +2949,8 @@ L_08A10B18: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(368)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3452,33 +3049,8 @@ L_08A10B80: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(368)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -3562,33 +3134,8 @@ L_08A10BD4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(368)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3682,33 +3229,8 @@ L_08A10C30: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(368)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -3792,33 +3314,8 @@ L_08A10C84: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(368)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3866,11 +3363,7 @@ L_08A10CBC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(336)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3880,10 +3373,7 @@ L_08A10CBC: ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -5318,33 +4808,8 @@ L_08A11878: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6429,33 +5894,8 @@ L_08A121D0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6516,28 +5956,7 @@ L_08A121D0: ctx.write_vfpu_vector_ct<39u, 4u>(vfpu_value); } ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[5]); - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 4u); - ctx.read_vfpu_vector_ct<12u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 14u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<14u, 36u, 12u, 4u, 3u>(); ctx.vfpu_ctrl[1u] = 0x00000000u; ctx.execute_vfpu_vcmp_ct<14u, 0u, 4u, 3u>(); ctx.gpr[5] = (0u + static_cast(0)); @@ -6942,11 +6361,7 @@ L_08A12610: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[8] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6974,10 +6389,7 @@ L_08A12610: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[10] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[10]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[28] <= ctx.fpr[12])) ? 0x00800000u : 0u); @@ -7671,28 +7083,7 @@ L_08A12B78: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9495,19 +8886,9 @@ L_08A13A5C: { const float vfpu_constant = std::bit_cast(0x3FC90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 64u, 32u, 1u, 2u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (17204u << 16u); diff --git a/profiles/vcs/generated/generated_unit_0132.cpp b/profiles/vcs/generated/generated_unit_0132.cpp index afc8c8c..9c9c405 100644 --- a/profiles/vcs/generated/generated_unit_0132.cpp +++ b/profiles/vcs/generated/generated_unit_0132.cpp @@ -6996,11 +6996,7 @@ L_08A17018: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[13] = (ctx.gpr[29] + static_cast(208)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[13] + static_cast(0); @@ -7060,11 +7056,7 @@ L_08A17018: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[9] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[9] + static_cast(0); @@ -7107,11 +7099,7 @@ L_08A17018: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[9] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[9] + static_cast(0); @@ -7152,11 +7140,7 @@ L_08A17018: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[12] = (ctx.gpr[29] + static_cast(224)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[12] + static_cast(0); @@ -7216,11 +7200,7 @@ L_08A17018: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7263,11 +7243,7 @@ L_08A17018: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7291,11 +7267,7 @@ L_08A17018: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[3] = (ctx.gpr[29] + static_cast(240)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[3] + static_cast(0); @@ -7366,11 +7338,7 @@ L_08A17208: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[10] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7409,11 +7377,7 @@ L_08A17208: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7452,11 +7416,7 @@ L_08A17208: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7521,11 +7481,7 @@ L_08A17208: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7643,11 +7599,7 @@ L_08A17304: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[12] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[12] + static_cast(0); @@ -7688,11 +7640,7 @@ L_08A17304: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[3] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[3] + static_cast(0); @@ -7752,11 +7700,7 @@ L_08A17304: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[12] = (ctx.gpr[29] + static_cast(160)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[12] + static_cast(0); @@ -7799,11 +7743,7 @@ L_08A17304: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[12] = (ctx.gpr[29] + static_cast(144)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[12] + static_cast(0); @@ -7844,11 +7784,7 @@ L_08A17304: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[12] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[12] + static_cast(0); @@ -7889,11 +7825,7 @@ L_08A17304: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[3] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[3] + static_cast(0); @@ -7953,11 +7885,7 @@ L_08A17304: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[12] = (ctx.gpr[29] + static_cast(288)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[12] + static_cast(0); @@ -8000,11 +7928,7 @@ L_08A17304: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[2] = (ctx.gpr[29] + static_cast(272)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[2] + static_cast(0); @@ -8028,11 +7952,7 @@ L_08A17304: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[7] = (ctx.gpr[29] + static_cast(256)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); @@ -8073,11 +7993,7 @@ L_08A17304: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8217,11 +8133,7 @@ L_08A175B4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(352)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -8240,10 +8152,7 @@ L_08A175B4: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -8345,11 +8254,7 @@ L_08A17634: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[13] = (ctx.gpr[29] + static_cast(208)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[13] + static_cast(0); @@ -8409,11 +8314,7 @@ L_08A17634: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[8] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[8] + static_cast(0); @@ -8456,11 +8357,7 @@ L_08A17634: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[8] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[8] + static_cast(0); @@ -8501,11 +8398,7 @@ L_08A17634: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[12] = (ctx.gpr[29] + static_cast(224)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[12] + static_cast(0); @@ -8565,11 +8458,7 @@ L_08A17634: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8612,11 +8501,7 @@ L_08A17634: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8640,11 +8525,7 @@ L_08A17634: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[3] = (ctx.gpr[29] + static_cast(240)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[3] + static_cast(0); @@ -8714,11 +8595,7 @@ L_08A17810: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[9] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8757,11 +8634,7 @@ L_08A17810: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8800,11 +8673,7 @@ L_08A17810: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8869,11 +8738,7 @@ L_08A17810: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8888,10 +8753,7 @@ L_08A17810: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[15] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[15] = std::bit_cast(ctx.gpr[15]); ctx.fpr[0] = ctx.fpr[0] + ctx.fpr[15]; @@ -9370,11 +9232,7 @@ L_08A17BD8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(96)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -9497,10 +9355,7 @@ L_08A17C9C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); diff --git a/profiles/vcs/generated/generated_unit_0133.cpp b/profiles/vcs/generated/generated_unit_0133.cpp index 86614bd..c9f3a7a 100644 --- a/profiles/vcs/generated/generated_unit_0133.cpp +++ b/profiles/vcs/generated/generated_unit_0133.cpp @@ -1909,7 +1909,7 @@ L_08A18A18: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[24] = (ctx.gpr[29] + static_cast(144)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[24] + static_cast(0); @@ -1933,7 +1933,7 @@ L_08A18A18: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1975,11 +1975,7 @@ L_08A18A18: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2019,11 +2015,7 @@ L_08A18A18: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[8] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2063,11 +2055,7 @@ L_08A18A18: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[25] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[25] + static_cast(0); @@ -2107,11 +2095,7 @@ L_08A18A18: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[10] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2150,11 +2134,7 @@ L_08A18A18: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[2] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2193,11 +2173,7 @@ L_08A18A18: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[12] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2466,11 +2442,7 @@ L_08A18F00: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[3] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2522,11 +2494,7 @@ L_08A18F00: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[3] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2578,11 +2546,7 @@ L_08A18F00: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[11] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2621,11 +2585,7 @@ L_08A18F00: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[3] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2677,11 +2637,7 @@ L_08A18F00: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[11] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2720,11 +2676,7 @@ L_08A18F00: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[3] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3145,11 +3097,7 @@ L_08A19470: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3253,11 +3201,7 @@ L_08A19524: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -3311,11 +3255,7 @@ L_08A19524: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3883,10 +3823,7 @@ L_08A199C0: L_08A199DC: ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -3918,10 +3855,7 @@ L_08A19A08: ctx.fpr[12] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[28] + static_cast(7676))); ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -3986,11 +3920,7 @@ L_08A19A5C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5412,7 +5342,7 @@ L_08A1A5DC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5469,33 +5399,8 @@ L_08A1A5F4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5553,11 +5458,7 @@ L_08A1A638: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5587,11 +5488,7 @@ L_08A1A650: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6203,15 +6100,8 @@ L_08A1AAA8: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -6240,15 +6130,8 @@ L_08A1AACC: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -6263,15 +6146,8 @@ L_08A1AB28: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -6286,17 +6162,9 @@ L_08A1AB4C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - ctx.execute_vfpu_vrot(1u, 64u, 2u, 4u); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<33u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] / vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_vrot_ct<1u, 64u, 2u, 4u>(); + ctx.execute_vfpu_vec3_ct<0u, 33u, 1u, 1u, 3u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -6449,10 +6317,7 @@ L_08A1AC54: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -6476,10 +6341,7 @@ L_08A1AC7C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -6499,10 +6361,7 @@ L_08A1AC98: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -6842,7 +6701,7 @@ L_08A1AE94: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6934,11 +6793,7 @@ L_08A1AF14: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6970,10 +6825,7 @@ L_08A1AF14: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7039,7 +6891,7 @@ L_08A1AFC0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[18] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); @@ -7058,10 +6910,7 @@ L_08A1AFC0: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7086,7 +6935,7 @@ L_08A1AFC0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7151,7 +7000,7 @@ L_08A1B064: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7182,10 +7031,7 @@ L_08A1B064: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7210,7 +7056,7 @@ L_08A1B064: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7288,28 +7134,7 @@ L_08A1B0E8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<8u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 4u, 3u); - ctx.read_vfpu_vector_ct<8u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 4u, 8u, 3u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7341,10 +7166,7 @@ L_08A1B0E8: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7406,7 +7228,7 @@ L_08A1B154: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7437,10 +7259,7 @@ L_08A1B154: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7465,7 +7284,7 @@ L_08A1B154: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7530,7 +7349,7 @@ L_08A1B1F4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7561,10 +7380,7 @@ L_08A1B1F4: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7589,7 +7405,7 @@ L_08A1B1F4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7846,11 +7662,7 @@ L_08A1B448: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(240)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7914,11 +7726,7 @@ L_08A1B448: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8217,10 +8025,7 @@ L_08A1B674: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[3] + static_cast(0), std::bit_cast(ctx.fpr[12])); @@ -8249,10 +8054,7 @@ L_08A1B674: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[13] + static_cast(0), std::bit_cast(ctx.fpr[12])); @@ -8320,33 +8122,8 @@ L_08A1B674: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8409,11 +8186,7 @@ L_08A1B7EC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[10] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[10] + static_cast(0); @@ -8429,10 +8202,7 @@ L_08A1B7EC: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[10] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[10]); ctx.fpr[13] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[9] + static_cast(0))); diff --git a/profiles/vcs/generated/generated_unit_0134.cpp b/profiles/vcs/generated/generated_unit_0134.cpp index 73bc62e..ddc9316 100644 --- a/profiles/vcs/generated/generated_unit_0134.cpp +++ b/profiles/vcs/generated/generated_unit_0134.cpp @@ -1067,11 +1067,7 @@ L_08A1C198: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1233,7 +1229,7 @@ L_08A1C314: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[18] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); @@ -1279,10 +1275,7 @@ L_08A1C378: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -1307,11 +1300,7 @@ L_08A1C378: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1393,11 +1382,7 @@ L_08A1C3DC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1494,19 +1479,9 @@ L_08A1C4D8: { const float vfpu_constant = std::bit_cast(0x3FC90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 64u, 32u, 1u, 2u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[0] = ctx.fpr[14] + ctx.fpr[12]; @@ -1956,11 +1931,7 @@ L_08A1C7BC: ctx.execute_vfpu_vdot_ct<32u, 41u, 4u, 4u>(); ctx.execute_vfpu_vdot_ct<0u, 40u, 4u, 4u>(); ctx.execute_vfpu_vdot_ct<64u, 42u, 4u, 4u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<12u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<4u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<4u, 32u, 12u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<4u, 44u, 1u, 2u>(); { const bool branch_taken = ((ctx.vfpu_ctrl[3] >> 5u) & 1u) != 0u; // nop @@ -1970,11 +1941,7 @@ L_08A1C7BC: goto L_08A1C80C; } L_08A1C80C: - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<12u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<4u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<4u, 32u, 12u, 1u, 1u>(); ctx.execute_vfpu_vcmp_ct<4u, 76u, 1u, 7u>(); { const bool branch_taken = ((ctx.vfpu_ctrl[3] >> 5u) & 1u) != 0u; // nop @@ -4367,11 +4334,7 @@ L_08A1DA3C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4525,15 +4488,8 @@ L_08A1DB98: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); @@ -4541,15 +4497,8 @@ L_08A1DB98: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); ctx.fpr[14] = std::bit_cast(std::bit_cast(ctx.fpr[14]) ^ 0x80000000u); @@ -4600,15 +4549,8 @@ L_08A1DB98: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); @@ -4616,15 +4558,8 @@ L_08A1DB98: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[5]); ctx.fpr[12] = std::bit_cast(std::bit_cast(ctx.fpr[14]) ^ 0x80000000u); @@ -5913,10 +5848,7 @@ L_08A1E73C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[22] = std::bit_cast(ctx.gpr[4]); ctx.fpr[12] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[28] + static_cast(-7652))); @@ -6638,11 +6570,7 @@ L_08A1EDD4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[7] = (ctx.gpr[29] + static_cast(272)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); @@ -6697,11 +6625,7 @@ L_08A1EDD4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7219,11 +7143,7 @@ L_08A1F130: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7251,11 +7171,7 @@ L_08A1F204: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(256)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7271,18 +7187,12 @@ L_08A1F204: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<1u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<1u, 1u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -7381,15 +7291,8 @@ L_08A1F2BC: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); ctx.fpr[12] = ctx.fpr[14] + ctx.fpr[13]; @@ -7424,11 +7327,7 @@ L_08A1F340: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7444,10 +7343,7 @@ L_08A1F340: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[13] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[17] + static_cast(128))); @@ -7530,15 +7426,8 @@ L_08A1F3FC: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16256u << 16u); @@ -7686,11 +7575,7 @@ L_08A1F520: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(416)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -7732,11 +7617,7 @@ L_08A1F520: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(384)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -8056,15 +7937,8 @@ L_08A1F7C0: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16256u << 16u); @@ -8313,11 +8187,7 @@ L_08A1FA58: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(400)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8413,11 +8283,7 @@ L_08A1FB3C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8432,10 +8298,7 @@ L_08A1FB3C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[24] = std::bit_cast(ctx.gpr[4]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[20] <= ctx.fpr[24])) ? 0x00800000u : 0u); @@ -8540,11 +8403,7 @@ L_08A1FBC4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8575,10 +8434,7 @@ L_08A1FBC4: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -8750,11 +8606,7 @@ L_08A1FD60: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(464)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8773,10 +8625,7 @@ L_08A1FD60: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -8824,11 +8673,7 @@ L_08A1FDC4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(208)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8844,10 +8689,7 @@ L_08A1FDC4: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[28] = std::bit_cast(ctx.gpr[4]); ctx.fpr[12] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[28] + static_cast(-7444))); @@ -8899,15 +8741,8 @@ L_08A1FE2C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[15] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16128u << 16u); @@ -8975,15 +8810,8 @@ L_08A1FED8: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.fpr[12] = ctx.fpr[13] + ctx.fpr[22]; @@ -9078,10 +8906,7 @@ L_08A1FFB8: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[13] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[28] + static_cast(-7428))); diff --git a/profiles/vcs/generated/generated_unit_0135.cpp b/profiles/vcs/generated/generated_unit_0135.cpp index b5d3a2e..ee30fef 100644 --- a/profiles/vcs/generated/generated_unit_0135.cpp +++ b/profiles/vcs/generated/generated_unit_0135.cpp @@ -987,11 +987,7 @@ L_08A20298: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(2480)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1064,11 +1060,7 @@ L_08A2031C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(2496)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1373,11 +1365,7 @@ L_08A205C8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1408,10 +1396,7 @@ L_08A205C8: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -1759,11 +1744,7 @@ L_08A20818: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[19] = (ctx.gpr[29] + static_cast(272)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); @@ -1782,10 +1763,7 @@ L_08A20818: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -2414,11 +2392,7 @@ L_08A20DE4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(208)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2434,10 +2408,7 @@ L_08A20DE4: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16544u << 16u); @@ -2500,11 +2471,7 @@ L_08A20E8C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(336)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2520,10 +2487,7 @@ L_08A20E8C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[22] = std::bit_cast(ctx.gpr[4]); ctx.fpr[12] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[28] + static_cast(-7348))); @@ -2586,15 +2550,8 @@ L_08A20F0C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16256u << 16u); @@ -2693,15 +2650,8 @@ L_08A20FFC: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16256u << 16u); @@ -3247,10 +3197,7 @@ L_08A214E0: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -3270,10 +3217,7 @@ L_08A214E0: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -3333,11 +3277,7 @@ L_08A214E0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4019,10 +3959,7 @@ L_08A21ABC: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -4042,10 +3979,7 @@ L_08A21ABC: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -4161,11 +4095,7 @@ L_08A21ABC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[17] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); @@ -5869,11 +5799,7 @@ L_08A22B00: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(240)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5905,10 +5831,7 @@ L_08A22B00: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -5933,7 +5856,7 @@ L_08A22B00: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(256)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -5952,10 +5875,7 @@ L_08A22B00: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -5980,7 +5900,7 @@ L_08A22B00: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6011,10 +5931,7 @@ L_08A22B00: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -6238,15 +6155,8 @@ L_08A22D5C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[6] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[6]); ctx.gpr[6] = (16128u << 16u); @@ -6290,11 +6200,7 @@ L_08A22D5C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(304)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -6378,15 +6284,8 @@ L_08A22E20: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[6] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[6]); ctx.gpr[6] = (16128u << 16u); @@ -6420,11 +6319,7 @@ L_08A22E20: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(368)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -6465,11 +6360,7 @@ L_08A22E20: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(336)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6536,15 +6427,8 @@ L_08A22F0C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16128u << 16u); @@ -6587,11 +6471,7 @@ L_08A22F0C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6636,11 +6516,7 @@ L_08A22F78: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(416)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6656,10 +6532,7 @@ L_08A22F78: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[13] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[28] + static_cast(-7248))); @@ -6701,15 +6574,8 @@ L_08A22FE0: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); ctx.fpr[15] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[28] + static_cast(-7248))); @@ -6722,15 +6588,8 @@ L_08A22FE0: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); ctx.fpr[12] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[28] + static_cast(-7248))); @@ -6776,11 +6635,7 @@ L_08A23044: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6835,11 +6690,7 @@ L_08A23044: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6879,10 +6730,7 @@ L_08A230C8: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -6955,10 +6803,7 @@ L_08A2312C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7014,10 +6859,7 @@ L_08A231AC: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7037,10 +6879,7 @@ L_08A231AC: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7065,7 +6904,7 @@ L_08A231AC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(448)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7097,10 +6936,7 @@ L_08A231AC: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7125,7 +6961,7 @@ L_08A231AC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7156,10 +6992,7 @@ L_08A231AC: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7187,10 +7020,7 @@ L_08A23268: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7233,7 +7063,7 @@ L_08A23268: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7264,10 +7094,7 @@ L_08A23268: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7342,15 +7169,8 @@ L_08A23354: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16128u << 16u); @@ -7400,11 +7220,7 @@ L_08A233A0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(528)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7445,11 +7261,7 @@ L_08A233A0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(496)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7528,11 +7340,7 @@ L_08A2344C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(544)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7548,10 +7356,7 @@ L_08A2344C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[13] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[28] + static_cast(-7248))); @@ -7593,15 +7398,8 @@ L_08A234B8: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); ctx.fpr[15] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[28] + static_cast(-7248))); @@ -7614,15 +7412,8 @@ L_08A234B8: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[15]; const float ft = ctx.fpr[14]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[12] = std::bit_cast(0x7FC00000u); else ctx.fpr[12] = fs * ft; } @@ -7717,11 +7508,7 @@ L_08A235AC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(592)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7762,11 +7549,7 @@ L_08A235AC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(560)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7853,11 +7636,7 @@ L_08A23684: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7896,10 +7675,7 @@ L_08A236B4: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7981,10 +7757,7 @@ L_08A23754: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -8004,10 +7777,7 @@ L_08A23754: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -8032,7 +7802,7 @@ L_08A23754: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(624)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8064,10 +7834,7 @@ L_08A23754: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -8092,7 +7859,7 @@ L_08A23754: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8123,10 +7890,7 @@ L_08A23754: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -8154,10 +7918,7 @@ L_08A23810: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -8200,7 +7961,7 @@ L_08A23810: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8231,10 +7992,7 @@ L_08A23810: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -8293,11 +8051,7 @@ L_08A238D0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(272)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8567,10 +8321,7 @@ L_08A23A74: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); diff --git a/profiles/vcs/generated/generated_unit_0136.cpp b/profiles/vcs/generated/generated_unit_0136.cpp index 5c47317..abcc733 100644 --- a/profiles/vcs/generated/generated_unit_0136.cpp +++ b/profiles/vcs/generated/generated_unit_0136.cpp @@ -1582,11 +1582,7 @@ L_08A2456C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[16] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); @@ -1619,10 +1615,7 @@ L_08A2456C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[28] = std::bit_cast(ctx.gpr[5]); { const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); @@ -1635,10 +1628,7 @@ L_08A2456C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[26] = std::bit_cast(ctx.gpr[5]); ctx.fpr[13] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[29] + static_cast(64))); @@ -1831,11 +1821,7 @@ L_08A24740: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1866,10 +1852,7 @@ L_08A24740: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -1946,11 +1929,7 @@ L_08A24818: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[22] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1978,10 +1957,7 @@ L_08A24818: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[4]); ctx.gpr[31] = (0x08A24850u); @@ -2043,17 +2019,9 @@ L_08A2489C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - ctx.execute_vfpu_vrot(1u, 64u, 2u, 4u); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<33u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] / vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_vrot_ct<1u, 64u, 2u, 4u>(); + ctx.execute_vfpu_vec3_ct<0u, 33u, 1u, 1u, 3u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (aot_mem.aot_direct_load32(ctx.gpr[28] + static_cast(5876))); @@ -2140,11 +2108,7 @@ L_08A2492C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[23] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2213,11 +2177,7 @@ L_08A24998: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[17] = (ctx.gpr[29] + static_cast(272)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); @@ -2248,11 +2208,7 @@ L_08A24998: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(288)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2361,10 +2317,7 @@ L_08A24A98: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[12] = ctx.fpr[20] - ctx.fpr[12]; @@ -2381,10 +2334,7 @@ L_08A24A98: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -2410,10 +2360,7 @@ L_08A24A98: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -2445,10 +2392,7 @@ L_08A24A98: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2493,10 +2437,7 @@ L_08A24B4C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(336)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2526,10 +2467,7 @@ L_08A24B64: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(256)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2749,19 +2687,9 @@ L_08A24CB0: ctx.gpr[5] = (std::bit_cast(ctx.fpr[13])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[13] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[20] + static_cast(84))); @@ -3121,11 +3049,7 @@ L_08A24F6C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3157,10 +3081,7 @@ L_08A24F6C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16928u << 16u); @@ -3248,10 +3169,7 @@ L_08A25010: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16416u << 16u); @@ -3397,11 +3315,7 @@ L_08A250F8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3433,10 +3347,7 @@ L_08A250F8: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16384u << 16u); @@ -3486,10 +3397,7 @@ L_08A25168: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16128u << 16u); @@ -3529,10 +3437,7 @@ L_08A251BC: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16872u << 16u); @@ -3596,10 +3501,7 @@ L_08A25220: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16384u << 16u); @@ -3691,11 +3593,7 @@ L_08A252B4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(96)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3727,10 +3625,7 @@ L_08A252B4: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16960u << 16u); @@ -3887,11 +3782,7 @@ L_08A253C8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3923,10 +3814,7 @@ L_08A253C8: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16920u << 16u); @@ -4277,10 +4165,7 @@ L_08A25618: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16800u << 16u); @@ -4410,11 +4295,7 @@ L_08A256F0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(192)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4446,10 +4327,7 @@ L_08A256F0: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16640u << 16u); @@ -4579,11 +4457,7 @@ L_08A257E0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(224)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4615,10 +4489,7 @@ L_08A257E0: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16840u << 16u); @@ -4748,11 +4619,7 @@ L_08A258D0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(256)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4784,10 +4651,7 @@ L_08A258D0: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16640u << 16u); @@ -5042,11 +4906,7 @@ L_08A25A94: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(288)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5078,10 +4938,7 @@ L_08A25A94: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16944u << 16u); @@ -5161,10 +5018,7 @@ L_08A25B30: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16448u << 16u); @@ -5269,11 +5123,7 @@ L_08A25BD4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(320)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5305,10 +5155,7 @@ L_08A25BD4: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16968u << 16u); @@ -5334,10 +5181,7 @@ L_08A25C2C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16448u << 16u); @@ -5450,11 +5294,7 @@ L_08A25CD8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(352)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5486,10 +5326,7 @@ L_08A25CD8: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16968u << 16u); @@ -5569,10 +5406,7 @@ L_08A25D74: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16384u << 16u); @@ -5677,10 +5511,7 @@ L_08A25E1C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16996u << 16u); @@ -5706,10 +5537,7 @@ L_08A25E54: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16256u << 16u); @@ -5809,11 +5637,7 @@ L_08A25EF0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(416)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5845,10 +5669,7 @@ L_08A25EF0: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16912u << 16u); @@ -5874,10 +5695,7 @@ L_08A25F48: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16384u << 16u); @@ -6329,28 +6147,7 @@ L_08A26228: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6549,10 +6346,7 @@ L_08A26378: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -6797,11 +6591,7 @@ L_08A2654C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6833,10 +6623,7 @@ L_08A2654C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16928u << 16u); @@ -6916,10 +6703,7 @@ L_08A265E8: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16416u << 16u); @@ -7113,10 +6897,7 @@ L_08A26718: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7307,11 +7088,7 @@ L_08A26898: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(96)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7343,10 +7120,7 @@ L_08A26898: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16872u << 16u); @@ -7426,10 +7200,7 @@ L_08A26934: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16384u << 16u); @@ -7569,10 +7340,7 @@ L_08A26A10: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7808,10 +7576,7 @@ L_08A26BD4: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -8364,11 +8129,7 @@ L_08A26FB0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8383,10 +8144,7 @@ L_08A26FB0: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16880u << 16u); @@ -8756,11 +8514,7 @@ L_08A27220: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8775,10 +8529,7 @@ L_08A27220: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16880u << 16u); @@ -8900,28 +8651,7 @@ L_08A272E4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(240)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -9104,10 +8834,7 @@ L_08A27424: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -9342,10 +9069,7 @@ L_08A275DC: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -9551,10 +9275,7 @@ L_08A27758: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -9758,10 +9479,7 @@ L_08A278CC: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -9980,28 +9698,7 @@ L_08A27A38: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(304)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -10216,10 +9913,7 @@ L_08A27BA4: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -10365,11 +10059,7 @@ L_08A27CB8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[18] = (ctx.gpr[29] + static_cast(416)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); @@ -10400,11 +10090,7 @@ L_08A27CE0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(384)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); diff --git a/profiles/vcs/generated/generated_unit_0137.cpp b/profiles/vcs/generated/generated_unit_0137.cpp index e7ea35f..11c4ab4 100644 --- a/profiles/vcs/generated/generated_unit_0137.cpp +++ b/profiles/vcs/generated/generated_unit_0137.cpp @@ -3736,11 +3736,7 @@ L_08A29300: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(192)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3756,10 +3752,7 @@ L_08A29300: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); { const bool branch_taken = 0u == 0u; @@ -3884,11 +3877,7 @@ L_08A293A0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3903,10 +3892,7 @@ L_08A293A0: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16752u << 16u); @@ -3943,11 +3929,7 @@ L_08A29478: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(272)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3968,10 +3950,7 @@ L_08A29478: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -4034,11 +4013,7 @@ L_08A294E8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(288)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -4108,11 +4083,7 @@ L_08A2956C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(304)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4295,11 +4266,7 @@ L_08A296C0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(368)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4347,11 +4314,7 @@ L_08A296F0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(384)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4436,10 +4399,7 @@ L_08A29770: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -4531,11 +4491,7 @@ L_08A297E0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(416)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4600,11 +4556,7 @@ L_08A2984C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(448)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4669,11 +4621,7 @@ L_08A29894: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(480)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4824,11 +4772,7 @@ L_08A298FC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4872,11 +4816,7 @@ L_08A298FC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4923,11 +4863,7 @@ L_08A298FC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4974,11 +4910,7 @@ L_08A298FC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5035,11 +4967,7 @@ L_08A29AE4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(768)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -5078,11 +5006,7 @@ L_08A29AE4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(784)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); diff --git a/profiles/vcs/generated/generated_unit_0138.cpp b/profiles/vcs/generated/generated_unit_0138.cpp index f9c980f..98efd62 100644 --- a/profiles/vcs/generated/generated_unit_0138.cpp +++ b/profiles/vcs/generated/generated_unit_0138.cpp @@ -6505,11 +6505,7 @@ L_08A2F20C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -6573,11 +6569,7 @@ L_08A2F250: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -7938,10 +7930,7 @@ L_08A2FCB0: aot_mem.aot_direct_store32_block(vfpu_address, vfpu_words); } ctx.gpr[2] = (std::bit_cast(ctx.fpr[15])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[2]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -8141,7 +8130,7 @@ L_08A2FDD8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8212,11 +8201,7 @@ L_08A2FE78: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8247,10 +8232,7 @@ L_08A2FE78: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[5]); aot_mem.aot_direct_store32(ctx.gpr[16] + static_cast(56), std::bit_cast(ctx.fpr[14])); @@ -8334,11 +8316,7 @@ L_08A2FF08: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); diff --git a/profiles/vcs/generated/generated_unit_0139.cpp b/profiles/vcs/generated/generated_unit_0139.cpp index 74ea126..61c6ef4 100644 --- a/profiles/vcs/generated/generated_unit_0139.cpp +++ b/profiles/vcs/generated/generated_unit_0139.cpp @@ -1105,10 +1105,7 @@ L_08A3005C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(192)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1197,11 +1194,7 @@ L_08A30110: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(208)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -1239,11 +1232,7 @@ L_08A30110: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1341,11 +1330,7 @@ L_08A301D8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(240)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -1404,10 +1389,7 @@ L_08A30248: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[13] < ctx.fpr[12])) ? 0x00800000u : 0u); @@ -1589,11 +1571,7 @@ L_08A30364: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[7] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); @@ -1665,11 +1643,7 @@ L_08A303FC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1694,10 +1668,7 @@ L_08A303FC: aot_mem.aot_direct_store32_block(vfpu_address, vfpu_words); } ctx.gpr[5] = (std::bit_cast(ctx.fpr[13])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -1850,10 +1821,7 @@ L_08A30508: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(160)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1989,11 +1957,7 @@ L_08A30614: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(240)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -2016,10 +1980,7 @@ L_08A30614: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4470,11 +4431,7 @@ L_08A318F0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5990,11 +5947,7 @@ L_08A324FC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6010,10 +5963,7 @@ L_08A324FC: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[16] + static_cast(2292), std::bit_cast(ctx.fpr[12])); @@ -6383,11 +6333,7 @@ L_08A327C0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7417,11 +7363,7 @@ L_08A32E48: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(96)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -7532,11 +7474,7 @@ L_08A32F2C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(208)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8059,11 +7997,7 @@ L_08A33334: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8566,11 +8500,7 @@ L_08A33750: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8619,10 +8549,7 @@ L_08A337D8: L_08A337E0: ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 17u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[15] = std::bit_cast(ctx.gpr[4]); ctx.fpr[16] = std::bit_cast(ctx.gpr[4]); @@ -8660,10 +8587,7 @@ L_08A33838: L_08A33840: ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 17u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); ctx.fpr[15] = std::bit_cast(ctx.gpr[4]); @@ -9279,10 +9203,7 @@ L_08A33CA0: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[13] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[19] + static_cast(2292))); @@ -9623,11 +9544,7 @@ L_08A33F54: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[18] = (ctx.gpr[29] + static_cast(160)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); diff --git a/profiles/vcs/generated/generated_unit_0140.cpp b/profiles/vcs/generated/generated_unit_0140.cpp index 094a095..fcc9ea9 100644 --- a/profiles/vcs/generated/generated_unit_0140.cpp +++ b/profiles/vcs/generated/generated_unit_0140.cpp @@ -1991,11 +1991,7 @@ L_08A345C0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2675,11 +2671,7 @@ L_08A34ABC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2785,10 +2777,7 @@ L_08A34BEC: L_08A34BF4: ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 17u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); @@ -2826,10 +2815,7 @@ L_08A34C4C: L_08A34C54: ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 17u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); @@ -2951,10 +2937,7 @@ L_08A34D20: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[12] <= ctx.fpr[22])) ? 0x00800000u : 0u); @@ -2993,11 +2976,7 @@ L_08A34D88: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3022,7 +3001,7 @@ L_08A34D88: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(144)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); diff --git a/profiles/vcs/generated/generated_unit_0141.cpp b/profiles/vcs/generated/generated_unit_0141.cpp index 5050568..002d98a 100644 --- a/profiles/vcs/generated/generated_unit_0141.cpp +++ b/profiles/vcs/generated/generated_unit_0141.cpp @@ -4890,15 +4890,8 @@ L_08A39E7C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[22] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (21352u << 16u); @@ -4971,11 +4964,7 @@ L_08A39F70: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5046,10 +5035,7 @@ L_08A39FDC: L_08A39FE8: ctx.gpr[4] = (std::bit_cast(ctx.fpr[28])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 17u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); @@ -5189,11 +5175,7 @@ L_08A3A0B0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[11] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5244,10 +5226,7 @@ L_08A3A0F0: L_08A3A0FC: ctx.gpr[3] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[3]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 17u>(); ctx.gpr[3] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[3]); ctx.fpr[15] = std::bit_cast(ctx.gpr[3]); @@ -7568,10 +7547,7 @@ L_08A3B2AC: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7591,10 +7567,7 @@ L_08A3B2AC: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7610,10 +7583,7 @@ L_08A3B2AC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -7645,10 +7615,7 @@ L_08A3B2AC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7689,11 +7656,7 @@ L_08A3B2AC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[18] = (ctx.gpr[29] + static_cast(144)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); @@ -7747,28 +7710,7 @@ L_08A3B3A0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(240)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); diff --git a/profiles/vcs/generated/generated_unit_0142.cpp b/profiles/vcs/generated/generated_unit_0142.cpp index cc25daa..7a49df3 100644 --- a/profiles/vcs/generated/generated_unit_0142.cpp +++ b/profiles/vcs/generated/generated_unit_0142.cpp @@ -2147,33 +2147,8 @@ L_08A3CA6C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3862,11 +3837,7 @@ L_08A3D774: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[19] = (ctx.gpr[29] + static_cast(224)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); @@ -3885,10 +3856,7 @@ L_08A3D774: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -3916,11 +3884,7 @@ L_08A3D774: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(240)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -3936,10 +3900,7 @@ L_08A3D774: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16384u << 16u); @@ -3979,10 +3940,7 @@ L_08A3D828: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -4394,10 +4352,7 @@ L_08A3DBD0: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -4445,10 +4400,7 @@ L_08A3DBD0: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[6] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[6]); ctx.gpr[6] = (16512u << 16u); @@ -4547,10 +4499,7 @@ L_08A3DD14: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -4640,11 +4589,7 @@ L_08A3DE04: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4667,10 +4612,7 @@ L_08A3DE68: ctx.gpr[5] = (std::bit_cast(ctx.fpr[30])); aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(272), std::bit_cast(ctx.fpr[30])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -5057,10 +4999,7 @@ L_08A3E1E8: ctx.gpr[5] = (ctx.gpr[16] + static_cast(96)); ctx.gpr[6] = (std::bit_cast(ctx.fpr[22])); ctx.set_vfpu_scalar_bits_ct<1u>(ctx.gpr[6]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<1u, 1u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -5419,10 +5358,7 @@ L_08A3E4AC: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -5489,10 +5425,7 @@ L_08A3E50C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -5548,11 +5481,7 @@ L_08A3E50C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[10] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5605,11 +5534,7 @@ L_08A3E50C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[9] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5662,11 +5587,7 @@ L_08A3E50C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[10] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5705,11 +5626,7 @@ L_08A3E50C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[9] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7980,33 +7897,8 @@ L_08A3FB50: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8036,11 +7928,7 @@ L_08A3FB78: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0143.cpp b/profiles/vcs/generated/generated_unit_0143.cpp index 39429f2..24463e9 100644 --- a/profiles/vcs/generated/generated_unit_0143.cpp +++ b/profiles/vcs/generated/generated_unit_0143.cpp @@ -1293,11 +1293,7 @@ L_08A40280: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2183,10 +2179,7 @@ L_08A40A5C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -3134,11 +3127,7 @@ L_08A410A4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -3179,11 +3168,7 @@ L_08A410A4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3341,11 +3326,7 @@ L_08A41254: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3830,11 +3811,7 @@ L_08A41688: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[9] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[9] + static_cast(0); @@ -4844,11 +4821,7 @@ L_08A41DBC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -4903,10 +4876,7 @@ L_08A41DBC: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -5381,11 +5351,7 @@ L_08A4221C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5409,11 +5375,7 @@ L_08A4221C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -5577,15 +5539,8 @@ L_08A42398: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (aot_mem.aot_direct_load32(ctx.gpr[28] + static_cast(8816))); @@ -5643,11 +5598,7 @@ L_08A42438: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -5680,10 +5631,7 @@ L_08A42484: L_08A4248C: ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 17u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[5]); ctx.fpr[14] = std::bit_cast(ctx.gpr[5]); @@ -5701,15 +5649,8 @@ L_08A424C0: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.fpr[12] = std::bit_cast(std::bit_cast(ctx.fpr[12]) ^ 0x80000000u); @@ -5718,15 +5659,8 @@ L_08A424C0: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[5]); aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(40), std::bit_cast(ctx.fpr[12])); @@ -5754,10 +5688,7 @@ L_08A42520: L_08A42528: ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 17u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[5]); ctx.fpr[14] = std::bit_cast(ctx.gpr[5]); @@ -5816,11 +5747,7 @@ L_08A4259C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6217,11 +6144,7 @@ L_08A42880: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6239,10 +6162,7 @@ L_08A42880: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -6280,10 +6200,7 @@ L_08A42920: L_08A42928: ctx.gpr[6] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[6]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 17u>(); ctx.gpr[6] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[6]); ctx.fpr[14] = std::bit_cast(ctx.gpr[6]); @@ -6512,7 +6429,7 @@ L_08A42A54: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6545,10 +6462,7 @@ L_08A42AF0: L_08A42AF8: ctx.gpr[4] = (std::bit_cast(ctx.fpr[24])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 17u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); @@ -6696,7 +6610,7 @@ L_08A42BC4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(288)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6730,10 +6644,7 @@ L_08A42C68: L_08A42C70: ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 17u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[5]); ctx.fpr[14] = std::bit_cast(ctx.gpr[5]); @@ -6890,7 +6801,7 @@ L_08A42D48: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(400)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6924,10 +6835,7 @@ L_08A42DEC: L_08A42DF4: ctx.gpr[5] = (std::bit_cast(ctx.fpr[24])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 17u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.fpr[13] = std::bit_cast(ctx.gpr[5]); @@ -6996,11 +6904,7 @@ L_08A42E70: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(96)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -7016,10 +6920,7 @@ L_08A42E70: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (15948u << 16u); @@ -7064,10 +6965,7 @@ L_08A42F04: aot_mem.aot_direct_store32_block(vfpu_address, vfpu_words); } ctx.gpr[5] = (std::bit_cast(ctx.fpr[14])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -7102,11 +7000,7 @@ L_08A42F04: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7524,11 +7418,7 @@ L_08A43404: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0144.cpp b/profiles/vcs/generated/generated_unit_0144.cpp index 64b00ad..d765576 100644 --- a/profiles/vcs/generated/generated_unit_0144.cpp +++ b/profiles/vcs/generated/generated_unit_0144.cpp @@ -968,10 +968,7 @@ L_08A44028: L_08A44030: ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 17u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); ctx.fpr[15] = std::bit_cast(ctx.gpr[4]); @@ -1506,11 +1503,7 @@ L_08A4442C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1640,11 +1633,7 @@ L_08A44510: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1685,11 +1674,7 @@ L_08A44580: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2006,10 +1991,7 @@ L_08A44874: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16672u << 16u); @@ -2017,10 +1999,7 @@ L_08A44874: { const float fs = ctx.fpr[12]; const float ft = ctx.fpr[13]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[12] = std::bit_cast(0x7FC00000u); else ctx.fpr[12] = fs * ft; } ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -2245,11 +2224,7 @@ L_08A44A9C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[21] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); @@ -2288,11 +2263,7 @@ L_08A44AE8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2334,11 +2305,7 @@ L_08A44B2C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2446,11 +2413,7 @@ L_08A44BCC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -2632,10 +2595,7 @@ L_08A44DA0: L_08A44DA8: ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 17u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[5]); ctx.fpr[14] = std::bit_cast(ctx.gpr[5]); @@ -2802,10 +2762,7 @@ L_08A44F08: L_08A44F10: ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 17u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[5]); ctx.fpr[14] = std::bit_cast(ctx.gpr[5]); @@ -3572,15 +3529,8 @@ L_08A454F4: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (48665u << 16u); @@ -3595,15 +3545,8 @@ L_08A454F4: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (15897u << 16u); @@ -4061,11 +4004,7 @@ L_08A45860: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4095,10 +4034,7 @@ L_08A45860: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16640u << 16u); @@ -4254,11 +4190,7 @@ L_08A45974: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4288,10 +4220,7 @@ L_08A45974: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16672u << 16u); @@ -6202,11 +6131,7 @@ L_08A46778: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6401,11 +6326,7 @@ L_08A4688C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[7] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); @@ -6462,11 +6383,7 @@ L_08A468DC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[7] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); @@ -6504,11 +6421,7 @@ L_08A468DC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6690,11 +6603,7 @@ L_08A469FC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(160)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7340,10 +7249,7 @@ L_08A46E34: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (16672u << 16u); @@ -7351,10 +7257,7 @@ L_08A46E34: { const float fs = ctx.fpr[12]; const float ft = ctx.fpr[13]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[12] = std::bit_cast(0x7FC00000u); else ctx.fpr[12] = fs * ft; } ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -7600,11 +7503,7 @@ L_08A4708C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[19] = (ctx.gpr[29] + static_cast(208)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); @@ -7643,11 +7542,7 @@ L_08A470F8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[30] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7689,11 +7584,7 @@ L_08A4713C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[30] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7869,11 +7760,7 @@ L_08A47278: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(256)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -8055,10 +7942,7 @@ L_08A47428: L_08A47434: ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 17u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); @@ -8232,10 +8116,7 @@ L_08A475B0: L_08A475BC: ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 17u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); diff --git a/profiles/vcs/generated/generated_unit_0145.cpp b/profiles/vcs/generated/generated_unit_0145.cpp index bf4d939..7f360da 100644 --- a/profiles/vcs/generated/generated_unit_0145.cpp +++ b/profiles/vcs/generated/generated_unit_0145.cpp @@ -1526,15 +1526,8 @@ L_08A48518: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.fpr[13] = std::bit_cast(std::bit_cast(ctx.fpr[13]) ^ 0x80000000u); @@ -1544,15 +1537,8 @@ L_08A48518: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(148), std::bit_cast(ctx.fpr[14])); @@ -1581,10 +1567,7 @@ L_08A485A8: L_08A485B4: ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 17u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); @@ -2882,10 +2865,7 @@ L_08A48FC4: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -3009,10 +2989,7 @@ L_08A490C4: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -3209,15 +3186,8 @@ L_08A49244: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.fpr[13] = std::bit_cast(std::bit_cast(ctx.fpr[13]) ^ 0x80000000u); @@ -3227,15 +3197,8 @@ L_08A49244: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(436), std::bit_cast(ctx.fpr[14])); @@ -3264,10 +3227,7 @@ L_08A492D4: L_08A492E0: ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 17u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); @@ -4237,11 +4197,7 @@ L_08A49AF8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(96)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -4272,11 +4228,7 @@ L_08A49AF8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -4292,10 +4244,7 @@ L_08A49AF8: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[6] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[6]); { const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -4308,10 +4257,7 @@ L_08A49AF8: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (15820u << 16u); @@ -4341,10 +4287,7 @@ L_08A49B8C: L_08A49BA4: ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<1u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<1u, 1u, 1u, 16u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(96)); { const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); std::uint32_t vfpu_words[4]{}; @@ -4362,10 +4305,7 @@ L_08A49BA4: aot_mem.aot_direct_store32_block(vfpu_address, vfpu_words); } ctx.gpr[6] = (std::bit_cast(ctx.fpr[13])); ctx.set_vfpu_scalar_bits_ct<1u>(ctx.gpr[6]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<1u, 1u, 1u, 16u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(112)); { const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); std::uint32_t vfpu_words[4]{}; @@ -4420,24 +4360,10 @@ L_08A49C0C: { const float vfpu_constant = std::bit_cast(0x3FC90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 64u, 32u, 1u, 2u>(); + ctx.execute_vfpu_vec3_ct<64u, 32u, 0u, 1u, 1u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<64u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[5]); { const bool branch_taken = 0u == 0u; @@ -5200,10 +5126,7 @@ L_08A4A168: L_08A4A174: ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 17u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); @@ -5838,11 +5761,7 @@ L_08A4A684: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(224)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5990,11 +5909,7 @@ L_08A4A7BC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[22] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6041,11 +5956,7 @@ L_08A4A844: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[30] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6108,11 +6019,7 @@ L_08A4A8C0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(352)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6195,11 +6102,7 @@ L_08A4A93C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(400)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -6701,11 +6604,7 @@ L_08A4AD3C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(528)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6727,10 +6626,7 @@ L_08A4AD3C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (15948u << 16u); @@ -6796,11 +6692,7 @@ L_08A4ADF4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(560)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -8102,11 +7994,7 @@ L_08A4B81C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[16] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); @@ -8241,11 +8129,7 @@ L_08A4B8E4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8319,11 +8203,7 @@ L_08A4B8E4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8436,11 +8316,7 @@ L_08A4B9C0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8822,11 +8698,7 @@ L_08A4BC80: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[30] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8847,10 +8719,7 @@ L_08A4BC80: ctx.fpr[12] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[18] + static_cast(8))); ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[22] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -8976,11 +8845,7 @@ L_08A4BD4C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -9014,10 +8879,7 @@ L_08A4BD70: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -9057,15 +8919,8 @@ L_08A4BDAC: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (48896u << 16u); @@ -9077,15 +8932,8 @@ L_08A4BDAC: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[13]; const float ft = ctx.fpr[24]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[12] = std::bit_cast(0x7FC00000u); else ctx.fpr[12] = fs * ft; } @@ -9111,11 +8959,7 @@ L_08A4BE10: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); diff --git a/profiles/vcs/generated/generated_unit_0146.cpp b/profiles/vcs/generated/generated_unit_0146.cpp index de6de4f..9568bbd 100644 --- a/profiles/vcs/generated/generated_unit_0146.cpp +++ b/profiles/vcs/generated/generated_unit_0146.cpp @@ -925,15 +925,8 @@ L_08A4C1A8: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.fpr[13] = std::bit_cast(std::bit_cast(ctx.fpr[13]) ^ 0x80000000u); @@ -943,15 +936,8 @@ L_08A4C1A8: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(164), std::bit_cast(ctx.fpr[14])); @@ -981,10 +967,7 @@ L_08A4C250: L_08A4C260: ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 17u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); @@ -1075,11 +1058,7 @@ L_08A4C304: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(192)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1124,7 +1103,7 @@ L_08A4C304: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(224)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -1165,11 +1144,7 @@ L_08A4C304: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1425,11 +1400,7 @@ L_08A4C580: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(256)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1472,11 +1443,7 @@ L_08A4C580: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(320)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1641,11 +1608,7 @@ L_08A4C6F0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[16] = (ctx.gpr[29] + static_cast(400)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); @@ -1667,10 +1630,7 @@ L_08A4C714: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (15948u << 16u); @@ -1690,10 +1650,7 @@ L_08A4C714: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -1735,11 +1692,7 @@ L_08A4C714: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(352)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1913,11 +1866,7 @@ L_08A4C8B8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(496)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -1948,11 +1897,7 @@ L_08A4C8B8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(512)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -1968,10 +1913,7 @@ L_08A4C8B8: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[6] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[6]); { const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -1984,10 +1926,7 @@ L_08A4C8B8: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (15820u << 16u); @@ -2017,10 +1956,7 @@ L_08A4C94C: L_08A4C964: ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<1u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<1u, 1u, 1u, 16u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(496)); { const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); std::uint32_t vfpu_words[4]{}; @@ -2038,10 +1974,7 @@ L_08A4C964: aot_mem.aot_direct_store32_block(vfpu_address, vfpu_words); } ctx.gpr[6] = (std::bit_cast(ctx.fpr[13])); ctx.set_vfpu_scalar_bits_ct<1u>(ctx.gpr[6]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<1u, 1u, 1u, 16u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(512)); { const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); std::uint32_t vfpu_words[4]{}; @@ -2096,24 +2029,10 @@ L_08A4C9CC: { const float vfpu_constant = std::bit_cast(0x3FC90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 64u, 32u, 1u, 2u>(); + ctx.execute_vfpu_vec3_ct<64u, 32u, 0u, 1u, 1u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<64u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[5]); { const bool branch_taken = 0u == 0u; @@ -2370,10 +2289,7 @@ L_08A4CBE0: L_08A4CBEC: ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 17u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); @@ -2634,11 +2550,7 @@ L_08A4CDDC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(576)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2761,11 +2673,7 @@ L_08A4CEC4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(640)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2848,11 +2756,7 @@ L_08A4CF40: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(688)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -3225,11 +3129,7 @@ L_08A4D218: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(768)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3251,10 +3151,7 @@ L_08A4D218: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (15948u << 16u); @@ -3320,11 +3217,7 @@ L_08A4D2D0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(800)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -3942,11 +3835,7 @@ L_08A4D74C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(976)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -4112,15 +4001,8 @@ L_08A4D8C8: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.fpr[13] = std::bit_cast(std::bit_cast(ctx.fpr[13]) ^ 0x80000000u); @@ -4130,15 +4012,8 @@ L_08A4D8C8: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(100), std::bit_cast(ctx.fpr[14])); @@ -4165,10 +4040,7 @@ L_08A4D93C: L_08A4D944: ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 17u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); @@ -5036,10 +4908,7 @@ L_08A4E150: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -5475,11 +5344,7 @@ L_08A4E53C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[30] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5494,10 +5359,7 @@ L_08A4E53C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[12] < ctx.fpr[28])) ? 0x00800000u : 0u); @@ -5639,11 +5501,7 @@ L_08A4E63C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[22] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5813,11 +5671,7 @@ L_08A4E7A8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[30] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5832,10 +5686,7 @@ L_08A4E7A8: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[12] < ctx.fpr[20])) ? 0x00800000u : 0u); @@ -5977,11 +5828,7 @@ L_08A4E8A8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[22] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6281,11 +6128,7 @@ L_08A4EB64: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[16] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); @@ -6442,10 +6285,7 @@ L_08A4ED24: L_08A4ED2C: ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 17u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); @@ -7105,11 +6945,7 @@ L_08A4F2B0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(192)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7221,11 +7057,7 @@ L_08A4F33C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(256)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7371,33 +7203,8 @@ L_08A4F41C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(288)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7562,33 +7369,8 @@ L_08A4F4F8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(304)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7746,33 +7528,8 @@ L_08A4F5F8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(320)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7909,33 +7666,8 @@ L_08A4F6AC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(336)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8087,33 +7819,8 @@ L_08A4F7A8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(352)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8243,33 +7950,8 @@ L_08A4F850: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(368)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8499,11 +8181,7 @@ L_08A4F9AC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(448)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8568,11 +8246,7 @@ L_08A4F9F4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(480)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8736,11 +8410,7 @@ L_08A4FB38: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(544)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -8759,10 +8429,7 @@ L_08A4FB38: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -8830,11 +8497,7 @@ L_08A4FB9C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(592)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -8853,10 +8516,7 @@ L_08A4FB9C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); diff --git a/profiles/vcs/generated/generated_unit_0147.cpp b/profiles/vcs/generated/generated_unit_0147.cpp index f26c300..74f02ce 100644 --- a/profiles/vcs/generated/generated_unit_0147.cpp +++ b/profiles/vcs/generated/generated_unit_0147.cpp @@ -1601,11 +1601,7 @@ L_08A5050C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(784)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -1825,11 +1821,7 @@ L_08A5071C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1845,10 +1837,7 @@ L_08A5071C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[6] = (ctx.gpr[29] + static_cast(112)); @@ -2250,11 +2239,7 @@ L_08A50B70: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2270,10 +2255,7 @@ L_08A50B70: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[5]); { const std::uint32_t vfpu_address = ctx.gpr[23] + static_cast(0); @@ -2294,11 +2276,7 @@ L_08A50B70: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2313,10 +2291,7 @@ L_08A50B70: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[6] = (ctx.gpr[29] + static_cast(96)); @@ -2583,11 +2558,7 @@ L_08A50E00: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2605,10 +2576,7 @@ L_08A50E00: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -2635,11 +2603,7 @@ L_08A50E00: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2657,10 +2621,7 @@ L_08A50E00: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -2813,11 +2774,7 @@ L_08A50F48: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2833,10 +2790,7 @@ L_08A50F48: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[2] = (0u | 1u); @@ -2994,11 +2948,7 @@ L_08A510F0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[23] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3013,10 +2963,7 @@ L_08A510F0: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[22] <= ctx.fpr[13])) ? 0x00800000u : 0u); @@ -3057,11 +3004,7 @@ L_08A5113C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[22] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3076,10 +3019,7 @@ L_08A5113C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[13]; const float ft = ctx.fpr[24]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[13] = std::bit_cast(0x7FC00000u); else ctx.fpr[13] = fs * ft; } @@ -3104,11 +3044,7 @@ L_08A51174: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3133,11 +3069,7 @@ L_08A51174: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3225,17 +3157,9 @@ L_08A5120C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - ctx.execute_vfpu_vrot(1u, 64u, 2u, 4u); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<33u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] / vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_vrot_ct<1u, 64u, 2u, 4u>(); + ctx.execute_vfpu_vec3_ct<0u, 33u, 1u, 1u, 3u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); { const bool branch_taken = 0u == 0u; @@ -3269,17 +3193,9 @@ L_08A51264: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - ctx.execute_vfpu_vrot(1u, 64u, 2u, 4u); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<33u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] / vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_vrot_ct<1u, 64u, 2u, 4u>(); + ctx.execute_vfpu_vec3_ct<0u, 33u, 1u, 1u, 3u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); goto L_08A512A4; @@ -3312,11 +3228,7 @@ L_08A512B4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[16] = (ctx.gpr[29] + static_cast(192)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); @@ -3332,10 +3244,7 @@ L_08A512B4: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (static_cast(static_cast(static_cast(aot_mem.aot_direct_load16(ctx.gpr[29] + static_cast(460)))))); @@ -3361,11 +3270,7 @@ L_08A512B4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3380,10 +3285,7 @@ L_08A512B4: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[22] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (aot_mem.aot_direct_load32(ctx.gpr[4] + static_cast(32))); @@ -3450,11 +3352,7 @@ L_08A5135C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(224)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3496,11 +3394,7 @@ L_08A5135C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3938,11 +3832,7 @@ L_08A5173C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(144)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3999,11 +3889,7 @@ L_08A5177C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(160)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4156,26 +4042,10 @@ L_08A51868: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_vx2i(0u, 2u, 2u, 3u); - ctx.execute_vfpu_vx2i(1u, 66u, 2u, 3u); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<0u, 3u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<3u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(23u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 3u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<3u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(23u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 3u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 3u>(vfpu_d); } + ctx.execute_vfpu_vx2i_ct<0u, 2u, 2u, 3u>(); + ctx.execute_vfpu_vx2i_ct<1u, 66u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<0u, 0u, 3u, 23u>(); + ctx.execute_vfpu_vi2f_ct<1u, 1u, 3u, 23u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4600,11 +4470,7 @@ L_08A51B90: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4744,11 +4610,7 @@ L_08A51CC4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0148.cpp b/profiles/vcs/generated/generated_unit_0148.cpp index a938c2c..39a5ec2 100644 --- a/profiles/vcs/generated/generated_unit_0148.cpp +++ b/profiles/vcs/generated/generated_unit_0148.cpp @@ -8813,7 +8813,7 @@ L_08A57E84: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8861,28 +8861,7 @@ L_08A57E9C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8930,28 +8909,7 @@ L_08A57EBC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<8u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 4u, 3u); - ctx.read_vfpu_vector_ct<8u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 4u, 8u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9008,33 +8966,8 @@ L_08A57EDC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9083,10 +9016,7 @@ L_08A57F20: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9116,11 +9046,7 @@ L_08A57F34: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9150,11 +9076,7 @@ L_08A57F4C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9214,10 +9136,7 @@ L_08A57F80: L_08A57F9C: ctx.gpr[6] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[6]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); diff --git a/profiles/vcs/generated/generated_unit_0149.cpp b/profiles/vcs/generated/generated_unit_0149.cpp index 4aa32ad..2d52370 100644 --- a/profiles/vcs/generated/generated_unit_0149.cpp +++ b/profiles/vcs/generated/generated_unit_0149.cpp @@ -984,15 +984,8 @@ L_08A5800C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -1007,15 +1000,8 @@ L_08A58030: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -1030,19 +1016,9 @@ L_08A58054: { const float vfpu_constant = std::bit_cast(0x3FC90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 64u, 32u, 1u, 2u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -1071,19 +1047,9 @@ L_08A58090: ctx.gpr[5] = (std::bit_cast(ctx.fpr[13])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -1156,10 +1122,7 @@ L_08A5810C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -1183,10 +1146,7 @@ L_08A58134: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -1922,11 +1882,7 @@ L_08A58658: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -1978,11 +1934,7 @@ L_08A58658: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2006,11 +1958,7 @@ L_08A58658: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3407,11 +3355,7 @@ L_08A59050: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3591,11 +3535,7 @@ L_08A59204: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5176,10 +5116,7 @@ L_08A59CF4: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (15948u << 16u); @@ -5519,10 +5456,7 @@ L_08A59F50: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (15948u << 16u); @@ -6349,10 +6283,7 @@ L_08A5A638: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16204u << 16u); @@ -6767,17 +6698,9 @@ L_08A5A9F0: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); ctx.execute_vfpu_vrot_ct<1u, 64u, 2u, 4u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<33u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] / vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 33u, 1u, 1u, 3u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[16] + static_cast(1368), std::bit_cast(ctx.fpr[12])); @@ -7743,10 +7666,7 @@ L_08A5B1E0: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (ctx.gpr[30] + static_cast(32)); @@ -7760,10 +7680,7 @@ L_08A5B1E0: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[13] <= ctx.fpr[12])) ? 0x00800000u : 0u); @@ -7913,19 +7830,9 @@ L_08A5B33C: ctx.gpr[5] = (std::bit_cast(ctx.fpr[13])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[24] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[24])); @@ -8892,10 +8799,7 @@ L_08A5BB58: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16025u << 16u); @@ -8940,10 +8844,7 @@ L_08A5BBC4: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16128u << 16u); @@ -9007,11 +8908,7 @@ L_08A5BC1C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -9082,10 +8979,7 @@ L_08A5BCB4: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (ctx.gpr[23] + static_cast(32)); @@ -9144,11 +9038,7 @@ L_08A5BCF4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -9363,11 +9253,7 @@ L_08A5BE48: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[17] = (ctx.gpr[29] + static_cast(144)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); diff --git a/profiles/vcs/generated/generated_unit_0151.cpp b/profiles/vcs/generated/generated_unit_0151.cpp index 350ed3c..c592056 100644 --- a/profiles/vcs/generated/generated_unit_0151.cpp +++ b/profiles/vcs/generated/generated_unit_0151.cpp @@ -1985,10 +1985,7 @@ L_08A609A0: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[12] <= ctx.fpr[26])) ? 0x00800000u : 0u); @@ -2103,15 +2100,8 @@ L_08A60AD0: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (static_cast(static_cast(static_cast(aot_mem.aot_direct_load16(ctx.gpr[23] + static_cast(86)))))); @@ -2171,11 +2161,7 @@ L_08A60B58: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(368)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2257,11 +2243,7 @@ L_08A60BCC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2347,11 +2329,7 @@ L_08A60C68: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[21] = (ctx.gpr[29] + static_cast(400)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); @@ -2393,11 +2371,7 @@ L_08A60CC4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[22] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2453,15 +2427,8 @@ L_08A60D40: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (static_cast(static_cast(static_cast(aot_mem.aot_direct_load16(ctx.gpr[23] + static_cast(86)))))); @@ -2519,11 +2486,7 @@ L_08A60D40: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(448)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -2902,11 +2865,7 @@ L_08A61088: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[20] = (ctx.gpr[29] + static_cast(512)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); @@ -2936,10 +2895,7 @@ L_08A61088: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.gpr[16] = (ctx.gpr[29] + static_cast(640)); @@ -3027,24 +2983,10 @@ L_08A61124: { const float vfpu_constant = std::bit_cast(0x3FC90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 64u, 32u, 1u, 2u>(); + ctx.execute_vfpu_vec3_ct<64u, 32u, 0u, 1u, 1u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<64u>()); ctx.fpr[28] = std::bit_cast(ctx.gpr[4]); ctx.fpr[28] = std::bit_cast(std::bit_cast(ctx.fpr[28]) & 0x7FFFFFFFu); @@ -3203,33 +3145,8 @@ L_08A6122C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3679,33 +3596,8 @@ L_08A615C8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3968,10 +3860,7 @@ L_08A617B8: L_08A617EC: ctx.gpr[4] = (std::bit_cast(ctx.fpr[14])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 17u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); ctx.fpr[15] = std::bit_cast(ctx.gpr[4]); @@ -4240,11 +4129,7 @@ L_08A61A10: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(912)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4507,33 +4392,8 @@ L_08A61BA4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(1088)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -4594,11 +4454,7 @@ L_08A61C08: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(1120)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -4631,10 +4487,7 @@ L_08A61C08: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -5059,11 +4912,7 @@ L_08A61EE8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5386,15 +5235,8 @@ L_08A62204: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); @@ -5402,15 +5244,8 @@ L_08A62204: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); ctx.fpr[12] = std::bit_cast(std::bit_cast(ctx.fpr[13]) ^ 0x80000000u); @@ -5453,28 +5288,7 @@ L_08A62204: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[8] = (ctx.gpr[29] + static_cast(176)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[8] + static_cast(0); @@ -5576,11 +5390,7 @@ L_08A62300: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(2384)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -5607,7 +5417,7 @@ L_08A62300: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(2368)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -5631,11 +5441,7 @@ L_08A62300: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(2352)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -5659,11 +5465,7 @@ L_08A62300: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(1376)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -5778,11 +5580,7 @@ L_08A62424: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(2512)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5809,7 +5607,7 @@ L_08A62424: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(2496)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5833,11 +5631,7 @@ L_08A62424: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(2480)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5861,11 +5655,7 @@ L_08A62424: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(1392)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -7770,11 +7560,7 @@ L_08A633B8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7880,11 +7666,7 @@ L_08A634AC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -7973,11 +7755,7 @@ L_08A63508: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8083,11 +7861,7 @@ L_08A635F4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(144)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -8319,11 +8093,7 @@ L_08A6373C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(176)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -8437,11 +8207,7 @@ L_08A637CC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8761,10 +8527,7 @@ L_08A63A74: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.fpr[13] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[28] + static_cast(-5844))); @@ -9177,28 +8940,7 @@ L_08A63D14: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(208)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -9260,11 +9002,7 @@ L_08A63D70: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9296,10 +9034,7 @@ L_08A63D70: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.fpr[13] = ctx.fpr[12] + ctx.fpr[24]; @@ -9338,11 +9073,7 @@ L_08A63D70: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0152.cpp b/profiles/vcs/generated/generated_unit_0152.cpp index 1d866e0..0f0f56b 100644 --- a/profiles/vcs/generated/generated_unit_0152.cpp +++ b/profiles/vcs/generated/generated_unit_0152.cpp @@ -1877,19 +1877,9 @@ L_08A646FC: ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[13] = ctx.fpr[28] - ctx.fpr[12]; @@ -2366,19 +2356,9 @@ L_08A64AAC: ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[13] = ctx.fpr[28] - ctx.fpr[12]; @@ -3097,10 +3077,7 @@ L_08A6502C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[26] = std::bit_cast(ctx.gpr[4]); ctx.gpr[31] = (0x08A6504Cu); @@ -3119,10 +3096,7 @@ L_08A6504C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[26] <= ctx.fpr[12])) ? 0x00800000u : 0u); @@ -3151,10 +3125,7 @@ L_08A6507C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (15820u << 16u); @@ -4167,11 +4138,7 @@ L_08A657D8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -4862,33 +4829,8 @@ L_08A65C94: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4918,11 +4860,7 @@ L_08A65CBC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5011,19 +4949,9 @@ L_08A65D34: ctx.gpr[5] = (std::bit_cast(ctx.fpr[13])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -6025,33 +5953,8 @@ L_08A6639C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6542,11 +6445,7 @@ L_08A6689C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -6556,10 +6455,7 @@ L_08A6689C: ctx.fpr[12] = std::bit_cast(ctx.gpr[6]); ctx.gpr[6] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[6]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -6607,21 +6503,14 @@ L_08A6689C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; aot_mem.aot_direct_store32_block(vfpu_address, vfpu_words); } ctx.gpr[6] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[6]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -7186,10 +7075,7 @@ L_08A66CD8: ctx.fpr[12] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[16] + static_cast(208))); ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -7224,11 +7110,7 @@ L_08A66CD8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -7384,28 +7266,7 @@ L_08A66E08: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7432,11 +7293,7 @@ L_08A66E08: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7471,10 +7328,7 @@ L_08A66E08: ctx.fpr[12] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[16] + static_cast(212))); ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -7616,28 +7470,7 @@ L_08A66F58: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7661,11 +7494,7 @@ L_08A66F58: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7697,10 +7526,7 @@ L_08A66F58: ctx.fpr[12] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[16] + static_cast(212))); ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -7917,11 +7743,7 @@ L_08A67154: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -7973,11 +7795,7 @@ L_08A67154: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8001,11 +7819,7 @@ L_08A67154: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8227,11 +8041,7 @@ L_08A6734C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8275,11 +8085,7 @@ L_08A6734C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8856,11 +8662,7 @@ L_08A67710: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9013,11 +8815,7 @@ L_08A67814: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9098,10 +8896,7 @@ L_08A678A8: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -9161,11 +8956,7 @@ L_08A678C8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9207,11 +8998,7 @@ L_08A67918: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(192)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -9265,11 +9052,7 @@ L_08A67980: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9355,19 +9138,9 @@ L_08A67A48: ctx.gpr[5] = (std::bit_cast(ctx.fpr[13])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[20])); @@ -9425,10 +9198,7 @@ L_08A67AB0: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[13] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[16] + static_cast(220))); @@ -9769,10 +9539,7 @@ L_08A67CE8: ctx.fpr[13] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[16] + static_cast(208))); ctx.gpr[5] = (std::bit_cast(ctx.fpr[13])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -9807,11 +9574,7 @@ L_08A67CE8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -9863,11 +9626,7 @@ L_08A67D7C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(96)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -9899,11 +9658,7 @@ L_08A67D7C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(368)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -9955,11 +9710,7 @@ L_08A67D7C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(336)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -9984,11 +9735,7 @@ L_08A67D7C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -10070,28 +9817,7 @@ L_08A67E70: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(176)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -10120,11 +9846,7 @@ L_08A67E70: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(160)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); diff --git a/profiles/vcs/generated/generated_unit_0153.cpp b/profiles/vcs/generated/generated_unit_0153.cpp index f7e17ab..b3defe6 100644 --- a/profiles/vcs/generated/generated_unit_0153.cpp +++ b/profiles/vcs/generated/generated_unit_0153.cpp @@ -1406,28 +1406,7 @@ L_08A68404: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1455,11 +1434,7 @@ L_08A68404: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1482,7 +1457,7 @@ L_08A68404: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1608,10 +1583,7 @@ L_08A68528: ctx.fpr[12] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[4] + static_cast(208))); ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -1679,28 +1651,7 @@ L_08A68570: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -1727,11 +1678,7 @@ L_08A68570: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -1758,7 +1705,7 @@ L_08A68570: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1766,10 +1713,7 @@ L_08A68570: ctx.fpr[12] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[4] + static_cast(212))); ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -1884,11 +1828,7 @@ L_08A68654: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1904,10 +1844,7 @@ L_08A68654: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[13] = std::bit_cast(0u); @@ -1923,10 +1860,7 @@ L_08A68654: L_08A686F0: ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); std::uint32_t vfpu_words[4]{}; @@ -2029,11 +1963,7 @@ L_08A68778: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2065,11 +1995,7 @@ L_08A68778: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(464)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2096,7 +2022,7 @@ L_08A68778: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(448)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2121,11 +2047,7 @@ L_08A68778: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(432)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2150,11 +2072,7 @@ L_08A68778: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(144)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2229,11 +2147,7 @@ L_08A68778: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(160)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2249,10 +2163,7 @@ L_08A68778: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[24] = std::bit_cast(ctx.gpr[4]); ctx.fpr[12] = std::bit_cast(0u); @@ -2268,10 +2179,7 @@ L_08A68778: L_08A688C4: ctx.gpr[4] = (std::bit_cast(ctx.fpr[24])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(160)); { const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); std::uint32_t vfpu_words[4]{}; @@ -4897,10 +4805,7 @@ L_08A69D70: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[13] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[28] + static_cast(7676))); @@ -5238,28 +5143,7 @@ L_08A69F78: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[23] = (ctx.gpr[29] + static_cast(144)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[23] + static_cast(0); @@ -5314,28 +5198,7 @@ L_08A69F78: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[23] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6339,11 +6202,7 @@ L_08A6A770: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6358,10 +6217,7 @@ L_08A6A770: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[16] + static_cast(264), std::bit_cast(ctx.fpr[12])); @@ -6990,11 +6846,7 @@ L_08A6AC00: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7009,10 +6861,7 @@ L_08A6AC00: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[16] + static_cast(264), std::bit_cast(ctx.fpr[12])); @@ -7491,28 +7340,7 @@ L_08A6AFA0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(96)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7555,28 +7383,7 @@ L_08A6AFA0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8104,10 +7911,7 @@ L_08A6B37C: ctx.fpr[12] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[16] + static_cast(208))); ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -8153,11 +7957,7 @@ L_08A6B40C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(256)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -8251,10 +8051,7 @@ L_08A6B470: ctx.fpr[12] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[16] + static_cast(208))); ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -8289,11 +8086,7 @@ L_08A6B470: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(272)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -8346,11 +8139,7 @@ L_08A6B520: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[30] = (ctx.gpr[29] + static_cast(352)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[30] + static_cast(0); @@ -8516,11 +8305,7 @@ L_08A6B664: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(400)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8551,7 +8336,7 @@ L_08A6B664: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(384)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); diff --git a/profiles/vcs/generated/generated_unit_0154.cpp b/profiles/vcs/generated/generated_unit_0154.cpp index c6fb432..9b9393f 100644 --- a/profiles/vcs/generated/generated_unit_0154.cpp +++ b/profiles/vcs/generated/generated_unit_0154.cpp @@ -744,11 +744,7 @@ L_08A6C03C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(1264)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -801,11 +797,7 @@ L_08A6C03C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(1232)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -830,11 +822,7 @@ L_08A6C03C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[22] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -934,11 +922,7 @@ L_08A6C03C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(688)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1302,11 +1286,7 @@ L_08A6C3D4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(736)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -1348,11 +1328,7 @@ L_08A6C3D4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(1472)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -1405,11 +1381,7 @@ L_08A6C3D4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(1440)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -1434,11 +1406,7 @@ L_08A6C3D4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1536,11 +1504,7 @@ L_08A6C3D4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(768)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -1933,11 +1897,7 @@ L_08A6C7DC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(816)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -1981,11 +1941,7 @@ L_08A6C7DC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2027,11 +1983,7 @@ L_08A6C7DC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[7] = (ctx.gpr[29] + static_cast(1696)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); @@ -2084,11 +2036,7 @@ L_08A6C7DC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[7] = (ctx.gpr[29] + static_cast(1664)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); @@ -2113,11 +2061,7 @@ L_08A6C7DC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2159,11 +2103,7 @@ L_08A6C7DC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[8] = (ctx.gpr[29] + static_cast(1824)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[8] + static_cast(0); @@ -2216,11 +2156,7 @@ L_08A6C7DC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[8] = (ctx.gpr[29] + static_cast(1792)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[8] + static_cast(0); @@ -2245,11 +2181,7 @@ L_08A6C7DC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2346,11 +2278,7 @@ L_08A6C7DC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[7] = (ctx.gpr[29] + static_cast(848)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); @@ -2424,11 +2352,7 @@ L_08A6C7DC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3252,10 +3176,7 @@ L_08A6D08C: ctx.fpr[13] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[21] + static_cast(208))); ctx.gpr[5] = (std::bit_cast(ctx.fpr[13])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -3290,11 +3211,7 @@ L_08A6D08C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -3347,11 +3264,7 @@ L_08A6D120: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(96)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3380,11 +3293,7 @@ L_08A6D120: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(496)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3437,11 +3346,7 @@ L_08A6D120: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(464)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3466,11 +3371,7 @@ L_08A6D120: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3552,28 +3453,7 @@ L_08A6D214: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(176)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3602,11 +3482,7 @@ L_08A6D214: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(160)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4180,10 +4056,7 @@ L_08A6D6A0: ctx.fpr[13] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[21] + static_cast(208))); ctx.gpr[5] = (std::bit_cast(ctx.fpr[13])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -4367,11 +4240,7 @@ L_08A6D7EC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(144)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4405,10 +4274,7 @@ L_08A6D7EC: ctx.fpr[12] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[21] + static_cast(212))); ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -4594,11 +4460,7 @@ L_08A6D8E4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4646,11 +4508,7 @@ L_08A6D8E4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -4666,10 +4524,7 @@ L_08A6D8E4: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[6] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[6]); { const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -4682,10 +4537,7 @@ L_08A6D8E4: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[5]); { const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4715,10 +4567,7 @@ L_08A6D8E4: L_08A6DA18: ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<1u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<1u, 1u, 1u, 16u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(96)); { const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); std::uint32_t vfpu_words[4]{}; @@ -4959,11 +4808,7 @@ L_08A6DBA8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(208)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -4992,11 +4837,7 @@ L_08A6DBA8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(1184)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -5049,11 +4890,7 @@ L_08A6DBA8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(1152)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -5078,11 +4915,7 @@ L_08A6DBA8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(224)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -5185,11 +5018,7 @@ L_08A6DBA8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(240)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -5236,11 +5065,7 @@ L_08A6DBA8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(256)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5256,10 +5081,7 @@ L_08A6DBA8: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); { const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5272,10 +5094,7 @@ L_08A6DBA8: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); { const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -5321,10 +5140,7 @@ L_08A6DD90: L_08A6DD98: ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<1u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<1u, 1u, 1u, 16u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(272)); { const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); std::uint32_t vfpu_words[4]{}; @@ -5568,11 +5384,7 @@ L_08A6DF34: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(384)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5601,11 +5413,7 @@ L_08A6DF34: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(1312)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5658,11 +5466,7 @@ L_08A6DF34: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(1280)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5687,11 +5491,7 @@ L_08A6DF34: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[20] = (ctx.gpr[29] + static_cast(400)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); @@ -5783,11 +5583,7 @@ L_08A6E068: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(416)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5834,11 +5630,7 @@ L_08A6E068: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(432)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -5854,10 +5646,7 @@ L_08A6E068: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[6] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[6]); { const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -5870,10 +5659,7 @@ L_08A6E068: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[5]); { const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5919,10 +5705,7 @@ L_08A6E110: L_08A6E118: ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<1u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<1u, 1u, 1u, 16u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(448)); { const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); std::uint32_t vfpu_words[4]{}; @@ -6141,11 +5924,7 @@ L_08A6E280: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(576)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6177,11 +5956,7 @@ L_08A6E280: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(592)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -6210,11 +5985,7 @@ L_08A6E280: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(1440)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -6267,11 +6038,7 @@ L_08A6E280: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(1408)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -6296,11 +6063,7 @@ L_08A6E280: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(608)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -6329,11 +6092,7 @@ L_08A6E280: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(1568)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -6386,11 +6145,7 @@ L_08A6E280: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(1536)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -6415,11 +6170,7 @@ L_08A6E280: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(624)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -6521,11 +6272,7 @@ L_08A6E280: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(640)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -6572,11 +6319,7 @@ L_08A6E280: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(656)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6592,10 +6335,7 @@ L_08A6E280: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[6] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[6]); { const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6608,10 +6348,7 @@ L_08A6E280: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); { const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -6641,10 +6378,7 @@ L_08A6E280: L_08A6E50C: ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<1u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<1u, 1u, 1u, 16u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(672)); { const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); std::uint32_t vfpu_words[4]{}; @@ -7194,43 +6928,10 @@ L_08A6E930: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<8u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<32u, 4u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<4u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<5u, 4u>(vfpu_result); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<100u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<104u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<100u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<100u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<100u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<100u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<8u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<5u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<5u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<5u, 32u, 4u, 4u, 3u>(); + ctx.execute_vfpu_vec3_ct<100u, 100u, 104u, 1u, 0u>(); + ctx.execute_vfpu_vec3_ct<100u, 100u, 100u, 1u, 2u>(); + ctx.execute_vfpu_vec3_ct<5u, 8u, 5u, 3u, 1u>(); ctx.execute_vfpu_vdot_ct<4u, 5u, 5u, 3u>(); ctx.execute_vfpu_vcmp_ct<4u, 100u, 1u, 7u>(); ctx.gpr[4] = (0u | 0u); @@ -7670,11 +7371,7 @@ L_08A6ECA0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[21] = (ctx.gpr[29] + static_cast(144)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); @@ -7725,33 +7422,8 @@ L_08A6ECA0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7850,33 +7522,8 @@ L_08A6ED2C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(160)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8113,11 +7760,7 @@ L_08A6EEC0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[16] = (ctx.gpr[29] + static_cast(272)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); @@ -8168,33 +7811,8 @@ L_08A6EEC0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(176)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8291,33 +7909,8 @@ L_08A6EF44: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(288)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8708,11 +8301,7 @@ L_08A6F1EC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(304)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8731,10 +8320,7 @@ L_08A6F1EC: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -8769,10 +8355,7 @@ L_08A6F1EC: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -8807,10 +8390,7 @@ L_08A6F1EC: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -8849,10 +8429,7 @@ L_08A6F1EC: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16128u << 16u); diff --git a/profiles/vcs/generated/generated_unit_0155.cpp b/profiles/vcs/generated/generated_unit_0155.cpp index e577db1..c906174 100644 --- a/profiles/vcs/generated/generated_unit_0155.cpp +++ b/profiles/vcs/generated/generated_unit_0155.cpp @@ -1004,10 +1004,7 @@ L_08A700B0: ctx.fpr[20] = static_cast(static_cast(std::bit_cast(ctx.fpr[12]))); ctx.gpr[4] = (std::bit_cast(ctx.fpr[20])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(352)); { const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); std::uint32_t vfpu_words[4]{}; @@ -1043,11 +1040,7 @@ L_08A700B0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(560)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -1060,10 +1053,7 @@ L_08A700B0: L_08A700F8: ctx.gpr[4] = (std::bit_cast(ctx.fpr[20])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); ctx.gpr[16] = (ctx.gpr[29] + static_cast(368)); { const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); std::uint32_t vfpu_words[4]{}; @@ -1238,10 +1228,7 @@ L_08A701E4: aot_mem.aot_direct_store32_block(vfpu_address, vfpu_words); } ctx.gpr[5] = (std::bit_cast(ctx.fpr[15])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -1560,11 +1547,7 @@ L_08A70470: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(624)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -1601,11 +1584,7 @@ L_08A70470: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(656)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -1914,11 +1893,7 @@ L_08A706B8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(720)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -1955,11 +1930,7 @@ L_08A706B8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(752)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -2349,11 +2320,7 @@ L_08A70994: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(816)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -2390,11 +2357,7 @@ L_08A70994: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(848)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -4060,43 +4023,10 @@ L_08A714BC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<8u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<32u, 4u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<4u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<5u, 4u>(vfpu_result); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<100u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<104u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<100u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<100u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<100u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<100u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<8u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<5u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<5u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<5u, 32u, 4u, 4u, 3u>(); + ctx.execute_vfpu_vec3_ct<100u, 100u, 104u, 1u, 0u>(); + ctx.execute_vfpu_vec3_ct<100u, 100u, 100u, 1u, 2u>(); + ctx.execute_vfpu_vec3_ct<5u, 8u, 5u, 3u, 1u>(); ctx.execute_vfpu_vdot_ct<4u, 5u, 5u, 3u>(); ctx.execute_vfpu_vcmp_ct<4u, 100u, 1u, 7u>(); ctx.gpr[4] = (0u | 0u); @@ -4503,11 +4433,7 @@ L_08A717E4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[30] = (ctx.gpr[29] + static_cast(1152)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[30] + static_cast(0); @@ -4616,33 +4542,8 @@ L_08A7186C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(1168)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4877,11 +4778,7 @@ L_08A71A08: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[30] = (ctx.gpr[29] + static_cast(1280)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[30] + static_cast(0); @@ -4988,33 +4885,8 @@ L_08A71A84: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(1296)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5367,10 +5239,7 @@ L_08A71C9C: ctx.fpr[13] = static_cast(static_cast(std::bit_cast(ctx.fpr[13]))); ctx.gpr[6] = (std::bit_cast(ctx.fpr[13])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[6]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -5472,10 +5341,7 @@ L_08A71D58: ctx.fpr[13] = static_cast(static_cast(std::bit_cast(ctx.fpr[13]))); ctx.gpr[6] = (std::bit_cast(ctx.fpr[13])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[6]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -5553,11 +5419,7 @@ L_08A71E38: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(1472)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5576,10 +5438,7 @@ L_08A71E38: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -5795,10 +5654,7 @@ L_08A71FE8: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -5828,10 +5684,7 @@ L_08A71FE8: aot_mem.aot_direct_store32_block(vfpu_address, vfpu_words); } ctx.gpr[6] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[6]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -5967,10 +5820,7 @@ L_08A72138: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -6014,10 +5864,7 @@ L_08A72138: aot_mem.aot_direct_store32_block(vfpu_address, vfpu_words); } ctx.gpr[6] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[6]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -6250,11 +6097,7 @@ L_08A7233C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(1744)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6648,43 +6491,10 @@ L_08A725C4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<8u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<32u, 4u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<4u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<5u, 4u>(vfpu_result); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<100u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<104u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<100u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<100u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<100u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<100u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<8u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<5u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<5u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<5u, 32u, 4u, 4u, 3u>(); + ctx.execute_vfpu_vec3_ct<100u, 100u, 104u, 1u, 0u>(); + ctx.execute_vfpu_vec3_ct<100u, 100u, 100u, 1u, 2u>(); + ctx.execute_vfpu_vec3_ct<5u, 8u, 5u, 3u, 1u>(); ctx.execute_vfpu_vdot_ct<4u, 5u, 5u, 3u>(); ctx.execute_vfpu_vcmp_ct<4u, 100u, 1u, 7u>(); ctx.gpr[4] = (0u | 0u); @@ -6925,11 +6735,7 @@ L_08A727CC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[7] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); @@ -6967,11 +6773,7 @@ L_08A727CC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[7] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); @@ -7263,11 +7065,7 @@ L_08A72A1C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[7] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); @@ -7305,11 +7103,7 @@ L_08A72A1C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[7] = (ctx.gpr[29] + static_cast(160)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); @@ -7696,11 +7490,7 @@ L_08A72D38: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(224)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8012,11 +7802,7 @@ L_08A72FC0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[7] = (ctx.gpr[29] + static_cast(272)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); @@ -8054,11 +7840,7 @@ L_08A72FC0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[7] = (ctx.gpr[29] + static_cast(304)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); diff --git a/profiles/vcs/generated/generated_unit_0156.cpp b/profiles/vcs/generated/generated_unit_0156.cpp index ffd0f64..bb8c5c0 100644 --- a/profiles/vcs/generated/generated_unit_0156.cpp +++ b/profiles/vcs/generated/generated_unit_0156.cpp @@ -4795,10 +4795,7 @@ L_08A75F08: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(40), std::bit_cast(ctx.fpr[12])); @@ -7214,11 +7211,7 @@ L_08A77440: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7465,15 +7458,8 @@ L_08A77670: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[31] = (0x08A7769Cu); @@ -7524,15 +7510,8 @@ L_08A776EC: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[31] = (0x08A77728u); diff --git a/profiles/vcs/generated/generated_unit_0157.cpp b/profiles/vcs/generated/generated_unit_0157.cpp index 1270a73..0592c9c 100644 --- a/profiles/vcs/generated/generated_unit_0157.cpp +++ b/profiles/vcs/generated/generated_unit_0157.cpp @@ -3964,11 +3964,7 @@ L_08A79B48: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -4009,11 +4005,7 @@ L_08A79B48: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4943,11 +4935,7 @@ L_08A7A238: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4962,10 +4950,7 @@ L_08A7A238: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[31] = (0x08A7A268u); @@ -5315,11 +5300,7 @@ L_08A7A4E4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6330,11 +6311,7 @@ L_08A7ACB8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6349,10 +6326,7 @@ L_08A7ACB8: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[31] = (0x08A7ACECu); @@ -6877,10 +6851,7 @@ L_08A7B0F0: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (aot_mem.aot_direct_load16(ctx.gpr[18] + static_cast(58))); diff --git a/profiles/vcs/generated/generated_unit_0158.cpp b/profiles/vcs/generated/generated_unit_0158.cpp index 4c80a6a..11876c3 100644 --- a/profiles/vcs/generated/generated_unit_0158.cpp +++ b/profiles/vcs/generated/generated_unit_0158.cpp @@ -1510,10 +1510,7 @@ L_08A7C81C: aot_mem.aot_direct_store32_block(vfpu_address, vfpu_words); } ctx.gpr[6] = (std::bit_cast(ctx.fpr[20])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[6]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -1562,10 +1559,7 @@ L_08A7C81C: aot_mem.aot_direct_store32_block(vfpu_address, vfpu_words); } ctx.gpr[4] = (std::bit_cast(ctx.fpr[20])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -1623,10 +1617,7 @@ L_08A7C900: aot_mem.aot_direct_store32_block(vfpu_address, vfpu_words); } ctx.gpr[6] = (std::bit_cast(ctx.fpr[20])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[6]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -1675,10 +1666,7 @@ L_08A7C900: aot_mem.aot_direct_store32_block(vfpu_address, vfpu_words); } ctx.gpr[4] = (std::bit_cast(ctx.fpr[20])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -3415,11 +3403,7 @@ L_08A7D674: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3435,10 +3419,7 @@ L_08A7D674: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[31] = (0x08A7D6ACu); @@ -3568,11 +3549,7 @@ L_08A7D73C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3588,10 +3565,7 @@ L_08A7D73C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[31] = (0x08A7D774u); @@ -3641,10 +3615,7 @@ L_08A7D7A4: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (17317u << 16u); @@ -5293,11 +5264,7 @@ L_08A7E3EC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5428,10 +5395,7 @@ L_08A7E488: aot_mem.aot_direct_store32_block(vfpu_address, vfpu_words); } ctx.gpr[5] = (std::bit_cast(ctx.fpr[20])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -5479,10 +5443,7 @@ L_08A7E488: aot_mem.aot_direct_store32_block(vfpu_address, vfpu_words); } ctx.gpr[6] = (std::bit_cast(ctx.fpr[20])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[6]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -5538,10 +5499,7 @@ L_08A7E504: aot_mem.aot_direct_store32_block(vfpu_address, vfpu_words); } ctx.gpr[5] = (std::bit_cast(ctx.fpr[20])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -5589,10 +5547,7 @@ L_08A7E504: aot_mem.aot_direct_store32_block(vfpu_address, vfpu_words); } ctx.gpr[6] = (std::bit_cast(ctx.fpr[20])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[6]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -7555,33 +7510,8 @@ L_08A7F4E8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0160.cpp b/profiles/vcs/generated/generated_unit_0160.cpp index 995cc64..cf69966 100644 --- a/profiles/vcs/generated/generated_unit_0160.cpp +++ b/profiles/vcs/generated/generated_unit_0160.cpp @@ -2609,10 +2609,7 @@ L_08A850D0: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16384u << 16u); @@ -3971,33 +3968,8 @@ L_08A85D30: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4014,15 +3986,8 @@ L_08A85D58: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -4037,15 +4002,8 @@ L_08A85D7C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -4908,11 +4866,7 @@ L_08A864C8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4949,7 +4903,7 @@ L_08A864C8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4977,10 +4931,7 @@ L_08A864C8: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[12] = ctx.fpr[22] / ctx.fpr[12]; @@ -5593,10 +5544,7 @@ L_08A86B10: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[12] <= ctx.fpr[24])) ? 0x00800000u : 0u); @@ -5703,11 +5651,7 @@ L_08A86BF0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6097,28 +6041,7 @@ L_08A86F28: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6377,15 +6300,8 @@ L_08A87124: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (15948u << 16u); @@ -6573,28 +6489,7 @@ L_08A8727C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0162.cpp b/profiles/vcs/generated/generated_unit_0162.cpp index 7bda754..07edf34 100644 --- a/profiles/vcs/generated/generated_unit_0162.cpp +++ b/profiles/vcs/generated/generated_unit_0162.cpp @@ -6766,11 +6766,7 @@ L_08A8F0B4: { const float vfpu_constant = std::bit_cast(0x3EA2F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<33u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<33u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<1u, 1u, 33u, 1u, 2u>(); { const std::uint16_t vfpu_half = 20032u; const std::uint32_t vfpu_sign = static_cast(vfpu_half & 0x8000u) << 16u; std::uint32_t vfpu_exponent = (vfpu_half >> 10u) & 0x1Fu; @@ -6817,13 +6813,9 @@ L_08A8F0B4: { const float vfpu_constant = std::bit_cast(0x40C90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<97u, 1u>(vfpu_value); } - ctx.execute_vfpu_vminmax(0u, 0u, 2u, 1u, false); - ctx.execute_vfpu_vminmax(0u, 0u, 34u, 1u, true); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_vminmax_ct<0u, 0u, 2u, 1u, false>(); + ctx.execute_vfpu_vminmax_ct<0u, 0u, 34u, 1u, true>(); + ctx.execute_vfpu_vec3_ct<32u, 0u, 1u, 1u, 2u>(); { float vfpu_s[4]{}; std::int32_t vfpu_d[4]{}; ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); @@ -6849,45 +6841,18 @@ L_08A8F0B4: ctx.vfpu[psprecomp::AllegrexContext::vfpu_vector_lane_index(32u, 1u, vfpu_i)] = std::bit_cast(static_cast(vfpu_d[vfpu_i])); } ctx.eat_vfpu_prefixes(); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<97u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<32u, 32u, 1u, 0u>(); + ctx.execute_vfpu_vec3_ct<32u, 32u, 97u, 1u, 2u>(); { float vfpu_value[4]{}; ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 32u, 1u, 1u>(); ctx.execute_vfpu_vcmp_ct<0u, 65u, 1u, 6u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<65u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<33u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<33u, 65u, 1u, 2u>(); ctx.execute_vfpu_vcmov_ct<2u, 97u, 1u, 0u, false>(); ctx.execute_vfpu_vcmp_ct<0u, 33u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<2u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 2u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<34u, 97u, 1u, 0u, false>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<34u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 34u, 1u, 0u>(); ctx.gpr[8] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[8]); jump_target = ctx.gpr[31]; @@ -7474,11 +7439,7 @@ L_08A8F5CC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -7505,7 +7466,7 @@ L_08A8F5CC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -7530,11 +7491,7 @@ L_08A8F5CC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7558,11 +7515,7 @@ L_08A8F5CC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7759,28 +7712,7 @@ L_08A8F764: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7901,28 +7833,7 @@ L_08A8F80C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8283,11 +8194,7 @@ L_08A8FAB4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8367,28 +8274,7 @@ L_08A8FBAC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8606,11 +8492,7 @@ L_08A8FDA4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -8774,11 +8656,7 @@ L_08A8FEF4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(144)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8788,10 +8666,7 @@ L_08A8FEF4: ctx.fpr[12] = ctx.fpr[12] + ctx.fpr[28]; ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -8906,28 +8781,7 @@ L_08A8FFB4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[30] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0163.cpp b/profiles/vcs/generated/generated_unit_0163.cpp index 52931a4..7495a2f 100644 --- a/profiles/vcs/generated/generated_unit_0163.cpp +++ b/profiles/vcs/generated/generated_unit_0163.cpp @@ -941,28 +941,7 @@ L_08A90088: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(256)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1048,11 +1027,7 @@ L_08A90160: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1107,11 +1082,7 @@ L_08A901AC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(272)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1213,11 +1184,7 @@ L_08A90258: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(304)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1274,28 +1241,7 @@ L_08A902A0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1707,35 +1653,11 @@ L_08A9061C: ctx.set_vfpu_scalar_bits_ct<10u>(ctx.gpr[14]); ctx.set_vfpu_scalar_bits_ct<42u>(ctx.gpr[11]); ctx.execute_vfpu_vx2i_ct<4u, 8u, 2u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<4u, 4u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(23u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 4u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<4u, 4u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<4u, 4u, 4u, 23u>(); ctx.execute_vfpu_vx2i_ct<5u, 9u, 2u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<5u, 4u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(23u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 4u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<5u, 4u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<5u, 5u, 4u, 23u>(); ctx.execute_vfpu_vx2i_ct<6u, 10u, 2u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<6u, 4u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(23u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 4u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<6u, 4u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<6u, 6u, 4u, 23u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<4u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2101,15 +2023,8 @@ L_08A909C4: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[5]); { const float fs = ctx.fpr[22]; const float ft = ctx.fpr[14]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[14] = std::bit_cast(0x7FC00000u); else ctx.fpr[14] = fs * ft; } @@ -2121,15 +2036,8 @@ L_08A909C4: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[5]); { const float fs = ctx.fpr[22]; const float ft = ctx.fpr[14]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[12] = std::bit_cast(0x7FC00000u); else ctx.fpr[12] = fs * ft; } @@ -3149,15 +3057,8 @@ L_08A91380: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[6] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[6]); ctx.gpr[6] = (16076u << 16u); @@ -3169,15 +3070,8 @@ L_08A91380: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[6] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[6]); { const float fs = ctx.fpr[12]; const float ft = ctx.fpr[13]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[0] = std::bit_cast(0x7FC00000u); else ctx.fpr[0] = fs * ft; } @@ -3186,15 +3080,8 @@ L_08A91380: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[6] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[6]); { const float fs = ctx.fpr[12]; const float ft = ctx.fpr[13]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[2] = std::bit_cast(0x7FC00000u); else ctx.fpr[2] = fs * ft; } @@ -3203,15 +3090,8 @@ L_08A91380: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[6] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[6]); ctx.gpr[6] = (48844u << 16u); @@ -3250,15 +3130,8 @@ L_08A9145C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[6] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[6]); { const float fs = ctx.fpr[12]; const float ft = ctx.fpr[30]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[19] = std::bit_cast(0x7FC00000u); else ctx.fpr[19] = fs * ft; } @@ -3267,15 +3140,8 @@ L_08A9145C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[6] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[6]); { const float fs = ctx.fpr[12]; const float ft = ctx.fpr[30]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[0] = std::bit_cast(0x7FC00000u); else ctx.fpr[0] = fs * ft; } @@ -3284,15 +3150,8 @@ L_08A9145C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[6] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[6]); { const float fs = ctx.fpr[12]; const float ft = ctx.fpr[30]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[2] = std::bit_cast(0x7FC00000u); else ctx.fpr[2] = fs * ft; } @@ -3301,15 +3160,8 @@ L_08A9145C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[6] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[6]); ctx.gpr[6] = (48972u << 16u); @@ -3341,15 +3193,8 @@ L_08A91524: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[6] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[6]); { const float fs = ctx.fpr[12]; const float ft = ctx.fpr[28]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[19] = std::bit_cast(0x7FC00000u); else ctx.fpr[19] = fs * ft; } @@ -3358,15 +3203,8 @@ L_08A91524: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[6] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[6]); { const float fs = ctx.fpr[12]; const float ft = ctx.fpr[28]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[0] = std::bit_cast(0x7FC00000u); else ctx.fpr[0] = fs * ft; } @@ -3375,15 +3213,8 @@ L_08A91524: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[6] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[6]); { const float fs = ctx.fpr[12]; const float ft = ctx.fpr[28]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[2] = std::bit_cast(0x7FC00000u); else ctx.fpr[2] = fs * ft; } @@ -3392,15 +3223,8 @@ L_08A91524: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[6] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[6]); ctx.gpr[6] = (48793u << 16u); @@ -4444,30 +4268,10 @@ L_08A91F8C: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -4956,30 +4760,10 @@ L_08A923CC: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -6645,11 +6429,7 @@ L_08A931B8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6679,11 +6459,7 @@ L_08A931D0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0164.cpp b/profiles/vcs/generated/generated_unit_0164.cpp index 5ff6540..750a72d 100644 --- a/profiles/vcs/generated/generated_unit_0164.cpp +++ b/profiles/vcs/generated/generated_unit_0164.cpp @@ -1261,30 +1261,10 @@ L_08A94388: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -1462,11 +1442,7 @@ L_08A944D4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1576,11 +1552,7 @@ L_08A9455C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1919,11 +1891,7 @@ L_08A947AC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2033,11 +2001,7 @@ L_08A94834: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3766,11 +3730,7 @@ L_08A95544: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -4347,11 +4307,7 @@ L_08A95ACC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<15u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<3u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<15u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<3u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<3u, 3u, 15u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.write_vfpu_vector_with_destination_prefix_ct<31u, 4u>(vfpu_value); } { float vfpu_s[16]{}, vfpu_t[16]{}, vfpu_d[16]{}; @@ -4715,15 +4671,8 @@ L_08A95FFC: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -4738,15 +4687,8 @@ L_08A96020: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -5596,28 +5538,7 @@ L_08A967EC: ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 4u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<12u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<14u, 4u>(vfpu_result); } + ctx.execute_vfpu_vtfm_ct<14u, 36u, 12u, 4u, 3u>(); ctx.vfpu_ctrl[1u] = 0x00000000u; ctx.execute_vfpu_vcmp_ct<14u, 0u, 4u, 3u>(); ctx.gpr[4] = (0u + static_cast(0)); @@ -5811,11 +5732,7 @@ L_08A9697C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); diff --git a/profiles/vcs/generated/generated_unit_0165.cpp b/profiles/vcs/generated/generated_unit_0165.cpp index 33b3994..863a33b 100644 --- a/profiles/vcs/generated/generated_unit_0165.cpp +++ b/profiles/vcs/generated/generated_unit_0165.cpp @@ -798,7 +798,7 @@ L_08A980C0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(96)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2103,26 +2103,10 @@ L_08A98DE0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_vx2i(0u, 2u, 2u, 3u); - ctx.execute_vfpu_vx2i(1u, 66u, 2u, 3u); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<0u, 3u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<3u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(23u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 3u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<3u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(23u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 3u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 3u>(vfpu_d); } + ctx.execute_vfpu_vx2i_ct<0u, 2u, 2u, 3u>(); + ctx.execute_vfpu_vx2i_ct<1u, 66u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<0u, 0u, 3u, 23u>(); + ctx.execute_vfpu_vi2f_ct<1u, 1u, 3u, 23u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2217,11 +2201,7 @@ L_08A98EB8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(96)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2245,11 +2225,7 @@ L_08A98EB8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -2273,7 +2249,7 @@ L_08A98EB8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2306,10 +2282,7 @@ L_08A98EB8: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -4391,11 +4364,7 @@ L_08A9A340: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4410,10 +4379,7 @@ L_08A9A340: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[13] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[18] + static_cast(32))); diff --git a/profiles/vcs/generated/generated_unit_0166.cpp b/profiles/vcs/generated/generated_unit_0166.cpp index 67bbca7..7309f48 100644 --- a/profiles/vcs/generated/generated_unit_0166.cpp +++ b/profiles/vcs/generated/generated_unit_0166.cpp @@ -4114,15 +4114,8 @@ L_08A9E0F4: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.fpr[13] = std::bit_cast(std::bit_cast(ctx.fpr[13]) ^ 0x80000000u); @@ -4132,15 +4125,8 @@ L_08A9E0F4: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(24), std::bit_cast(ctx.fpr[13])); @@ -4373,28 +4359,7 @@ L_08A9E2F4: ctx.write_vfpu_vector_ct<39u, 4u>(vfpu_value); } ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[5]); - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 4u); - ctx.read_vfpu_vector_ct<12u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 14u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<14u, 36u, 12u, 4u, 3u>(); ctx.vfpu_ctrl[1u] = 0x00000000u; ctx.execute_vfpu_vcmp_ct<14u, 0u, 4u, 3u>(); ctx.gpr[5] = (0u + static_cast(0)); diff --git a/profiles/vcs/generated/generated_unit_0168.cpp b/profiles/vcs/generated/generated_unit_0168.cpp index f42fd64..b59dd12 100644 --- a/profiles/vcs/generated/generated_unit_0168.cpp +++ b/profiles/vcs/generated/generated_unit_0168.cpp @@ -805,15 +805,8 @@ L_08AA401C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -828,15 +821,8 @@ L_08AA4040: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -3274,17 +3260,9 @@ L_08AA57AC: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - ctx.execute_vfpu_vrot(1u, 64u, 2u, 4u); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<33u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] / vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_vrot_ct<1u, 64u, 2u, 4u>(); + ctx.execute_vfpu_vec3_ct<0u, 33u, 1u, 1u, 3u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[20] + static_cast(252), std::bit_cast(ctx.fpr[12])); @@ -3824,10 +3802,7 @@ L_08AA5B0C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (ctx.gpr[4] + static_cast(32)); @@ -3841,10 +3816,7 @@ L_08AA5B0C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[13] <= ctx.fpr[12])) ? 0x00800000u : 0u); @@ -4724,15 +4696,8 @@ L_08AA639C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (static_cast(static_cast(static_cast(aot_mem.aot_direct_load16(ctx.gpr[20] + static_cast(86)))))); @@ -4793,11 +4758,7 @@ L_08AA6428: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(288)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4880,11 +4841,7 @@ L_08AA64A0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4973,11 +4930,7 @@ L_08AA654C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(320)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -5018,11 +4971,7 @@ L_08AA65A8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5093,15 +5042,8 @@ L_08AA6650: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (static_cast(static_cast(static_cast(aot_mem.aot_direct_load16(ctx.gpr[20] + static_cast(86)))))); @@ -5159,11 +5101,7 @@ L_08AA6650: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(368)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -5207,15 +5145,8 @@ L_08AA6738: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); @@ -5223,15 +5154,8 @@ L_08AA6738: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); ctx.fpr[12] = std::bit_cast(std::bit_cast(ctx.fpr[13]) ^ 0x80000000u); @@ -5274,28 +5198,7 @@ L_08AA6738: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(160)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5427,10 +5330,7 @@ L_08AA6878: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -5653,19 +5553,9 @@ L_08AA6A00: { const float vfpu_constant = std::bit_cast(0x3FC90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 64u, 32u, 1u, 2u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[24] = std::bit_cast(ctx.gpr[4]); ctx.fpr[24] = std::bit_cast(std::bit_cast(ctx.fpr[24]) ^ 0x80000000u); @@ -5827,15 +5717,8 @@ L_08AA6B58: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (15692u << 16u); @@ -5990,15 +5873,8 @@ L_08AA6C7C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (15759u << 16u); @@ -6111,15 +5987,8 @@ L_08AA6D34: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16256u << 16u); @@ -6268,15 +6137,8 @@ L_08AA6E8C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); @@ -6284,15 +6146,8 @@ L_08AA6E8C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); ctx.fpr[12] = std::bit_cast(std::bit_cast(ctx.fpr[13]) ^ 0x80000000u); @@ -6335,28 +6190,7 @@ L_08AA6E8C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(704)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6442,33 +6276,8 @@ L_08AA6F54: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(624)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -6562,33 +6371,8 @@ L_08AA6F9C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(720)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7276,15 +7060,8 @@ L_08AA7590: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[18] = std::bit_cast(ctx.gpr[5]); { const float fs = ctx.fpr[18]; const float ft = ctx.fpr[16]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[17] = std::bit_cast(0x7FC00000u); else ctx.fpr[17] = fs * ft; } @@ -8212,11 +7989,7 @@ L_08AA7C78: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -8232,10 +8005,7 @@ L_08AA7C78: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[8] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[8]); ctx.gpr[8] = (15820u << 16u); @@ -8277,11 +8047,7 @@ L_08AA7D54: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8594,11 +8360,7 @@ L_08AA7F94: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[30] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0169.cpp b/profiles/vcs/generated/generated_unit_0169.cpp index 17ce91a..b54f48c 100644 --- a/profiles/vcs/generated/generated_unit_0169.cpp +++ b/profiles/vcs/generated/generated_unit_0169.cpp @@ -1021,33 +1021,8 @@ L_08AA8250: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2280,15 +2255,8 @@ L_08AA9170: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[17] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[18])); @@ -2296,15 +2264,8 @@ L_08AA9170: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[19] = std::bit_cast(ctx.gpr[4]); ctx.fpr[18] = std::bit_cast(std::bit_cast(ctx.fpr[19]) ^ 0x80000000u); @@ -3527,15 +3488,8 @@ L_08AA9F5C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[18])); @@ -3543,15 +3497,8 @@ L_08AA9F5C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[19] = std::bit_cast(ctx.gpr[4]); ctx.fpr[18] = std::bit_cast(std::bit_cast(ctx.fpr[19]) ^ 0x80000000u); @@ -4515,15 +4462,8 @@ L_08AAA9F8: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[24] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[19])); @@ -4531,15 +4471,8 @@ L_08AAA9F8: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[26] = std::bit_cast(ctx.gpr[4]); ctx.fpr[19] = std::bit_cast(std::bit_cast(ctx.fpr[26]) ^ 0x80000000u); diff --git a/profiles/vcs/generated/generated_unit_0172.cpp b/profiles/vcs/generated/generated_unit_0172.cpp index af5a281..9c455ca 100644 --- a/profiles/vcs/generated/generated_unit_0172.cpp +++ b/profiles/vcs/generated/generated_unit_0172.cpp @@ -1968,11 +1968,7 @@ L_08AB44B0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[12] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1987,10 +1983,7 @@ L_08AB44B0: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[12] < ctx.fpr[20])) ? 0x00800000u : 0u); @@ -2193,11 +2186,7 @@ L_08AB45FC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2212,10 +2201,7 @@ L_08AB45FC: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[12] < ctx.fpr[20])) ? 0x00800000u : 0u); diff --git a/profiles/vcs/generated/generated_unit_0176.cpp b/profiles/vcs/generated/generated_unit_0176.cpp index 4629706..d95c854 100644 --- a/profiles/vcs/generated/generated_unit_0176.cpp +++ b/profiles/vcs/generated/generated_unit_0176.cpp @@ -4011,28 +4011,7 @@ L_08AC524C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4057,11 +4036,7 @@ L_08AC524C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8539,33 +8514,8 @@ L_08AC7434: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8595,11 +8545,7 @@ L_08AC745C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8629,11 +8575,7 @@ L_08AC7474: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8726,15 +8668,8 @@ L_08AC74F4: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -8749,15 +8684,8 @@ L_08AC7518: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -8772,19 +8700,9 @@ L_08AC753C: { const float vfpu_constant = std::bit_cast(0x3FC90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 64u, 32u, 1u, 2u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -8799,24 +8717,10 @@ L_08AC7564: { const float vfpu_constant = std::bit_cast(0x3FC90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 64u, 32u, 1u, 2u>(); + ctx.execute_vfpu_vec3_ct<64u, 32u, 0u, 1u, 1u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<64u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -8873,10 +8777,7 @@ L_08AC75C0: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -8896,10 +8797,7 @@ L_08AC75DC: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -9332,30 +9230,10 @@ L_08AC78A8: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -9398,30 +9276,10 @@ L_08AC78EC: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } diff --git a/profiles/vcs/generated/generated_unit_0178.cpp b/profiles/vcs/generated/generated_unit_0178.cpp index f22ef4a..5cbe947 100644 --- a/profiles/vcs/generated/generated_unit_0178.cpp +++ b/profiles/vcs/generated/generated_unit_0178.cpp @@ -6460,11 +6460,7 @@ L_08ACE0F8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6733,11 +6729,7 @@ L_08ACE2C8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8023,11 +8015,7 @@ L_08ACEAD8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8311,11 +8299,7 @@ L_08ACED08: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(176)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8439,11 +8423,7 @@ L_08ACEE00: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(224)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8522,11 +8502,7 @@ L_08ACEE7C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(272)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -9866,33 +9842,8 @@ L_08ACF85C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0179.cpp b/profiles/vcs/generated/generated_unit_0179.cpp index 60dbf99..8e03b5f 100644 --- a/profiles/vcs/generated/generated_unit_0179.cpp +++ b/profiles/vcs/generated/generated_unit_0179.cpp @@ -1495,65 +1495,11 @@ L_08AD03B8: ctx.write_vfpu_vector_with_destination_prefix_ct<35u, 4u>(vfpu_value); } { float vfpu_value[4]{}; vfpu_value[3u] = 1.0f; ctx.write_vfpu_vector_with_destination_prefix_ct<43u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<32u, 4u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<4u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<5u, 4u>(vfpu_result); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<40u, 4u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<12u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<13u, 4u>(vfpu_result); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<100u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<108u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<100u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<100u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<100u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<100u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<13u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<5u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<5u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<5u, 32u, 4u, 4u, 3u>(); + ctx.execute_vfpu_vtfm_ct<13u, 40u, 12u, 4u, 3u>(); + ctx.execute_vfpu_vec3_ct<100u, 100u, 108u, 1u, 0u>(); + ctx.execute_vfpu_vec3_ct<100u, 100u, 100u, 1u, 2u>(); + ctx.execute_vfpu_vec3_ct<5u, 13u, 5u, 3u, 1u>(); ctx.execute_vfpu_vdot_ct<4u, 5u, 5u, 3u>(); ctx.execute_vfpu_vcmp_ct<4u, 100u, 1u, 7u>(); ctx.gpr[4] = (0u | 0u); @@ -1756,50 +1702,8 @@ L_08AD0530: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<39u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<32u, 4u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<12u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<13u, 4u>(vfpu_result); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 4u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<13u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<14u, 4u>(vfpu_result); } + ctx.execute_vfpu_vtfm_ct<13u, 32u, 12u, 4u, 3u>(); + ctx.execute_vfpu_vtfm_ct<14u, 36u, 13u, 4u, 3u>(); ctx.vfpu_ctrl[1u] = 0x000000FFu; ctx.execute_vfpu_vcmp_ct<14u, 12u, 4u, 3u>(); // vflush: architectural no-op that retains VFPU prefixes @@ -2848,15 +2752,8 @@ L_08AD0BDC: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (48035u << 16u); @@ -2894,15 +2791,8 @@ L_08AD0C38: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (48131u << 16u); @@ -3716,28 +3606,7 @@ L_08AD12A0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3760,11 +3629,7 @@ L_08AD12A0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5775,11 +5640,7 @@ L_08AD23BC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[10] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[10] + static_cast(0); @@ -5833,11 +5694,7 @@ L_08AD23BC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[9] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5951,11 +5808,7 @@ L_08AD2480: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[10] = (ctx.gpr[29] + static_cast(144)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[10] + static_cast(0); @@ -6009,11 +5862,7 @@ L_08AD2480: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[9] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6126,11 +5975,7 @@ L_08AD2540: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[10] = (ctx.gpr[29] + static_cast(208)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[10] + static_cast(0); @@ -6184,11 +6029,7 @@ L_08AD2540: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[9] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6283,11 +6124,7 @@ L_08AD2604: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(272)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -6303,10 +6140,7 @@ L_08AD2604: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (16968u << 16u); diff --git a/profiles/vcs/generated/generated_unit_0183.cpp b/profiles/vcs/generated/generated_unit_0183.cpp index 838e130..2b0fb0e 100644 --- a/profiles/vcs/generated/generated_unit_0183.cpp +++ b/profiles/vcs/generated/generated_unit_0183.cpp @@ -8438,33 +8438,8 @@ L_08AE34D8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9475,28 +9450,7 @@ L_08AE3C4C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9744,33 +9698,8 @@ L_08AE3E58: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -10035,33 +9964,8 @@ L_08AE3F8C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0184.cpp b/profiles/vcs/generated/generated_unit_0184.cpp index 65dc7d2..40ce7df 100644 --- a/profiles/vcs/generated/generated_unit_0184.cpp +++ b/profiles/vcs/generated/generated_unit_0184.cpp @@ -1119,11 +1119,7 @@ L_08AE417C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5180,11 +5176,7 @@ L_08AE62CC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5374,33 +5366,8 @@ L_08AE63F0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5461,28 +5428,7 @@ L_08AE63F0: ctx.write_vfpu_vector_ct<39u, 4u>(vfpu_value); } ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[5]); - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 4u); - ctx.read_vfpu_vector_ct<12u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 14u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<14u, 36u, 12u, 4u, 3u>(); ctx.vfpu_ctrl[1u] = 0x00000000u; ctx.execute_vfpu_vcmp_ct<14u, 0u, 4u, 3u>(); ctx.gpr[5] = (0u + static_cast(0)); @@ -5794,10 +5740,7 @@ L_08AE676C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (16968u << 16u); @@ -6857,11 +6800,7 @@ L_08AE6FBC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7713,28 +7652,7 @@ L_08AE76C4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0185.cpp b/profiles/vcs/generated/generated_unit_0185.cpp index 8c68b20..dd19ac4 100644 --- a/profiles/vcs/generated/generated_unit_0185.cpp +++ b/profiles/vcs/generated/generated_unit_0185.cpp @@ -2507,11 +2507,7 @@ L_08AE8CC8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4537,19 +4533,9 @@ L_08AE9E28: { const float vfpu_constant = std::bit_cast(0x3FC90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 64u, 32u, 1u, 2u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (17204u << 16u); @@ -6919,33 +6905,8 @@ L_08AEB308: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0186.cpp b/profiles/vcs/generated/generated_unit_0186.cpp index 91b378b..0b31645 100644 --- a/profiles/vcs/generated/generated_unit_0186.cpp +++ b/profiles/vcs/generated/generated_unit_0186.cpp @@ -3436,28 +3436,7 @@ L_08AED10C: ctx.fpr[12] = std::bit_cast(0u); ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 4u); - ctx.read_vfpu_vector_ct<12u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 14u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<14u, 36u, 12u, 4u, 3u>(); ctx.vfpu_ctrl[1u] = 0x00000000u; ctx.execute_vfpu_vcmp_ct<14u, 0u, 4u, 3u>(); ctx.gpr[4] = (0u + static_cast(0)); diff --git a/profiles/vcs/generated/generated_unit_0188.cpp b/profiles/vcs/generated/generated_unit_0188.cpp index 881bd62..ab7b462 100644 --- a/profiles/vcs/generated/generated_unit_0188.cpp +++ b/profiles/vcs/generated/generated_unit_0188.cpp @@ -5001,30 +5001,10 @@ L_08AF629C: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -5195,11 +5175,7 @@ L_08AF63EC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); diff --git a/profiles/vcs/generated/generated_unit_0189.cpp b/profiles/vcs/generated/generated_unit_0189.cpp index f073eb3..fa3df2b 100644 --- a/profiles/vcs/generated/generated_unit_0189.cpp +++ b/profiles/vcs/generated/generated_unit_0189.cpp @@ -2136,11 +2136,7 @@ L_08AF89E0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2164,11 +2160,7 @@ L_08AF89E0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2192,11 +2184,7 @@ L_08AF89E0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2293,15 +2281,8 @@ L_08AF8C14: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[31] = (0x08AF8C40u); @@ -2375,15 +2356,8 @@ L_08AF8CC8: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[31] = (0x08AF8D04u); @@ -3311,11 +3285,7 @@ L_08AF9588: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -3600,11 +3570,7 @@ L_08AF97FC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5460,28 +5426,7 @@ L_08AFA988: ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[13])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 4u); - ctx.read_vfpu_vector_ct<12u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 14u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<14u, 36u, 12u, 4u, 3u>(); ctx.vfpu_ctrl[1u] = 0x00000000u; ctx.execute_vfpu_vcmp_ct<14u, 0u, 4u, 3u>(); ctx.gpr[4] = (0u + static_cast(0)); @@ -5654,28 +5599,7 @@ L_08AFAAF0: ctx.fpr[15] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[15])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 4u); - ctx.read_vfpu_vector_ct<12u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 14u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<14u, 36u, 12u, 4u, 3u>(); ctx.vfpu_ctrl[1u] = 0x00000000u; ctx.execute_vfpu_vcmp_ct<14u, 0u, 4u, 3u>(); ctx.gpr[4] = (0u + static_cast(0)); @@ -6513,10 +6437,7 @@ L_08AFB484: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -6772,15 +6693,8 @@ L_08AFB6D0: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[20]; const float ft = ctx.fpr[14]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[14] = std::bit_cast(0x7FC00000u); else ctx.fpr[14] = fs * ft; } @@ -6792,15 +6706,8 @@ L_08AFB6D0: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[20]; const float ft = ctx.fpr[14]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[12] = std::bit_cast(0x7FC00000u); else ctx.fpr[12] = fs * ft; } @@ -7054,15 +6961,8 @@ L_08AFB978: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[20]; const float ft = ctx.fpr[14]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[14] = std::bit_cast(0x7FC00000u); else ctx.fpr[14] = fs * ft; } @@ -7074,15 +6974,8 @@ L_08AFB978: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[20]; const float ft = ctx.fpr[14]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[12] = std::bit_cast(0x7FC00000u); else ctx.fpr[12] = fs * ft; } diff --git a/profiles/vcs/generated/generated_unit_0190.cpp b/profiles/vcs/generated/generated_unit_0190.cpp index 6bef3f5..16ad932 100644 --- a/profiles/vcs/generated/generated_unit_0190.cpp +++ b/profiles/vcs/generated/generated_unit_0190.cpp @@ -3393,10 +3393,7 @@ L_08AFD290: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -3436,11 +3433,7 @@ L_08AFD290: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3455,10 +3448,7 @@ L_08AFD290: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); { const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); @@ -6529,28 +6519,7 @@ L_08AFE8E8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6598,28 +6567,7 @@ L_08AFE908: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<8u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<4u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<8u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } + ctx.execute_vfpu_vtfm_ct<0u, 4u, 8u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6676,33 +6624,8 @@ L_08AFE928: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6751,10 +6674,7 @@ L_08AFE96C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6784,11 +6704,7 @@ L_08AFE980: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6818,11 +6734,7 @@ L_08AFE998: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7190,15 +7102,8 @@ L_08AFEBF0: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -7213,15 +7118,8 @@ L_08AFEC14: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -7236,19 +7134,9 @@ L_08AFEC38: { const float vfpu_constant = std::bit_cast(0x3FC90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 64u, 32u, 1u, 2u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -7263,24 +7151,10 @@ L_08AFEC60: { const float vfpu_constant = std::bit_cast(0x3FC90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 64u, 32u, 1u, 2u>(); + ctx.execute_vfpu_vec3_ct<64u, 32u, 0u, 1u, 1u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<64u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -7294,19 +7168,9 @@ L_08AFEC8C: ctx.gpr[5] = (std::bit_cast(ctx.fpr[13])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -7398,10 +7262,7 @@ L_08AFED20: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7425,10 +7286,7 @@ L_08AFED48: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -7650,11 +7508,7 @@ L_08AFEE90: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -7706,11 +7560,7 @@ L_08AFEE90: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7734,11 +7584,7 @@ L_08AFEE90: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9164,10 +9010,7 @@ L_08AFF838: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(96)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -9428,11 +9271,7 @@ L_08AFF954: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); diff --git a/profiles/vcs/generated/generated_unit_0191.cpp b/profiles/vcs/generated/generated_unit_0191.cpp index 09fc044..2027052 100644 --- a/profiles/vcs/generated/generated_unit_0191.cpp +++ b/profiles/vcs/generated/generated_unit_0191.cpp @@ -5040,11 +5040,7 @@ L_08B01844: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5210,7 +5206,7 @@ L_08B01924: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -5234,7 +5230,7 @@ L_08B01924: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7173,10 +7169,7 @@ L_08B026B0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7260,10 +7253,7 @@ L_08B02724: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7297,10 +7287,7 @@ L_08B02724: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7383,10 +7370,7 @@ L_08B0279C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); { const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); @@ -7399,10 +7383,7 @@ L_08B0279C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16256u << 16u); @@ -7517,10 +7498,7 @@ L_08B028B4: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -7609,11 +7587,7 @@ L_08B02950: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); diff --git a/profiles/vcs/generated/generated_unit_0192.cpp b/profiles/vcs/generated/generated_unit_0192.cpp index 7853cbe..b0c69b4 100644 --- a/profiles/vcs/generated/generated_unit_0192.cpp +++ b/profiles/vcs/generated/generated_unit_0192.cpp @@ -3947,11 +3947,7 @@ L_08B05488: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4164,11 +4160,7 @@ L_08B055C0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(192)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4450,11 +4442,7 @@ L_08B057C4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4540,11 +4528,7 @@ L_08B0583C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(384)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4707,11 +4691,7 @@ L_08B0598C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4781,11 +4761,7 @@ L_08B05A00: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(432)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4854,11 +4830,7 @@ L_08B05A84: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(496)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4902,11 +4874,7 @@ L_08B05A84: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(480)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4975,11 +4943,7 @@ L_08B05B30: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(576)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5026,11 +4990,7 @@ L_08B05B30: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(560)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5202,11 +5162,7 @@ L_08B05CC8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(688)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5257,11 +5213,7 @@ L_08B05D34: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(720)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6103,28 +6055,7 @@ L_08B06338: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6159,10 +6090,7 @@ L_08B06378: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); { const std::uint32_t vfpu_address = ctx.gpr[23] + static_cast(0); @@ -6486,10 +6414,7 @@ L_08B065A4: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[26] = std::bit_cast(ctx.gpr[4]); { const std::uint32_t vfpu_address = ctx.gpr[23] + static_cast(0); @@ -6836,11 +6761,7 @@ L_08B067F4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6894,19 +6815,9 @@ L_08B06874: ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[13] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[17] + static_cast(116))); @@ -7083,15 +6994,8 @@ L_08B069F8: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[12] = ctx.fpr[30] / ctx.fpr[12]; @@ -7133,15 +7037,8 @@ L_08B06A44: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[12] = ctx.fpr[30] / ctx.fpr[12]; @@ -7172,15 +7069,8 @@ L_08B06A8C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[12] = ctx.fpr[30] / ctx.fpr[12]; @@ -7214,15 +7104,8 @@ L_08B06AE0: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[12] = ctx.fpr[30] / ctx.fpr[12]; @@ -7263,15 +7146,8 @@ L_08B06B38: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[12] = ctx.fpr[30] / ctx.fpr[12]; @@ -7920,15 +7796,8 @@ L_08B06FDC: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[13])); @@ -7953,15 +7822,8 @@ L_08B06FDC: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (std::bit_cast(ctx.fpr[13])); @@ -8000,11 +7862,7 @@ L_08B06FDC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(480)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8247,11 +8105,7 @@ L_08B0719C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(1520)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8278,7 +8132,7 @@ L_08B0719C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(1504)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8302,11 +8156,7 @@ L_08B0719C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(1488)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8331,11 +8181,7 @@ L_08B0719C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0193.cpp b/profiles/vcs/generated/generated_unit_0193.cpp index 17bc842..9ee3d12 100644 --- a/profiles/vcs/generated/generated_unit_0193.cpp +++ b/profiles/vcs/generated/generated_unit_0193.cpp @@ -2718,33 +2718,8 @@ L_08B08ED0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3269,10 +3244,7 @@ L_08B09364: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3338,10 +3310,7 @@ L_08B093B4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3407,10 +3376,7 @@ L_08B09408: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3469,28 +3435,7 @@ L_08B09434: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3712,11 +3657,7 @@ L_08B0959C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3751,10 +3692,7 @@ L_08B095E8: L_08B095F4: ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 17u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); @@ -3951,11 +3889,7 @@ L_08B097B8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4064,11 +3998,7 @@ L_08B09868: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[8] = (ctx.gpr[29] + static_cast(272)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[8] + static_cast(0); @@ -4114,11 +4044,7 @@ L_08B09868: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[8] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4233,11 +4159,7 @@ L_08B0998C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[22] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4275,7 +4197,7 @@ L_08B0998C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[22] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4326,11 +4248,7 @@ L_08B099D8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[21] = (ctx.gpr[29] + static_cast(320)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); @@ -4497,11 +4415,7 @@ L_08B09B24: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(368)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4898,15 +4812,8 @@ L_08B09F0C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[28]; const float ft = ctx.fpr[12]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[12] = std::bit_cast(0x7FC00000u); else ctx.fpr[12] = fs * ft; } @@ -4916,15 +4823,8 @@ L_08B09F0C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[28]; const float ft = ctx.fpr[14]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[26] = std::bit_cast(0x7FC00000u); else ctx.fpr[26] = fs * ft; } @@ -4995,11 +4895,7 @@ L_08B09FD4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[30] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6767,11 +6663,7 @@ L_08B0AD28: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[17] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); @@ -6958,10 +6850,7 @@ L_08B0AE58: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.fpr[13] = std::bit_cast(std::bit_cast(ctx.fpr[12])); @@ -6983,10 +6872,7 @@ L_08B0AE88: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[5]); goto L_08B0AE9C; @@ -7011,10 +6897,7 @@ L_08B0AEAC: L_08B0AEB4: ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -7094,10 +6977,7 @@ L_08B0AF00: L_08B0AF08: ctx.gpr[4] = (std::bit_cast(ctx.fpr[13])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -7166,28 +7046,7 @@ L_08B0AF2C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7213,11 +7072,7 @@ L_08B0AF2C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[22] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7240,7 +7095,7 @@ L_08B0AF2C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8075,11 +7930,7 @@ L_08B0B56C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8112,10 +7963,7 @@ L_08B0B56C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -8132,10 +7980,7 @@ L_08B0B56C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[13] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[16] + static_cast(208))); @@ -8158,7 +8003,7 @@ L_08B0B56C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8345,11 +8190,7 @@ L_08B0B6D4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8408,11 +8249,7 @@ L_08B0B6D4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(160)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8523,10 +8360,7 @@ L_08B0B7F4: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[7] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[7]); ctx.gpr[7] = (15692u << 16u); @@ -8876,11 +8710,7 @@ L_08B0BA5C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[7] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); @@ -8904,11 +8734,7 @@ L_08B0BA5C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8949,11 +8775,7 @@ L_08B0BA5C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8976,11 +8798,7 @@ L_08B0BA5C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9470,10 +9288,7 @@ L_08B0BF04: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); ctx.gpr[7] = (ctx.gpr[29] + static_cast(320)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[7] + static_cast(0); diff --git a/profiles/vcs/generated/generated_unit_0194.cpp b/profiles/vcs/generated/generated_unit_0194.cpp index 2c3ecaa..a3e2671 100644 --- a/profiles/vcs/generated/generated_unit_0194.cpp +++ b/profiles/vcs/generated/generated_unit_0194.cpp @@ -6205,11 +6205,7 @@ L_08B0E7D4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6224,10 +6220,7 @@ L_08B0E7D4: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[4]); ctx.gpr[31] = (0x08B0E804u); @@ -6390,11 +6383,7 @@ L_08B0E8F0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6410,10 +6399,7 @@ L_08B0E8F0: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[4]); ctx.gpr[31] = (0x08B0E924u); @@ -6555,10 +6541,7 @@ L_08B0E9E8: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (15759u << 16u); @@ -6599,11 +6582,7 @@ L_08B0EA28: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6619,10 +6598,7 @@ L_08B0EA28: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16880u << 16u); @@ -7264,10 +7240,7 @@ L_08B0EF3C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (15692u << 16u); @@ -7331,10 +7304,7 @@ L_08B0EFB4: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (15692u << 16u); @@ -7415,10 +7385,7 @@ L_08B0F050: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (15692u << 16u); @@ -7772,10 +7739,7 @@ L_08B0F308: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (15651u << 16u); @@ -7856,10 +7820,7 @@ L_08B0F3A4: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (15651u << 16u); @@ -8784,10 +8745,7 @@ L_08B0FAEC: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (15651u << 16u); diff --git a/profiles/vcs/generated/generated_unit_0195.cpp b/profiles/vcs/generated/generated_unit_0195.cpp index c0921e5..6121d7d 100644 --- a/profiles/vcs/generated/generated_unit_0195.cpp +++ b/profiles/vcs/generated/generated_unit_0195.cpp @@ -1661,10 +1661,7 @@ L_08B104DC: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (ctx.gpr[16] + static_cast(320)); @@ -1678,10 +1675,7 @@ L_08B104DC: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[12] <= ctx.fpr[13])) ? 0x00800000u : 0u); @@ -1709,10 +1703,7 @@ L_08B10520: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (15820u << 16u); @@ -1837,11 +1828,7 @@ L_08B10634: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(352)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1857,10 +1844,7 @@ L_08B10634: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16704u << 16u); @@ -1927,11 +1911,7 @@ L_08B106B8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(384)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1947,10 +1927,7 @@ L_08B106B8: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16752u << 16u); @@ -4000,33 +3977,8 @@ L_08B11578: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4730,33 +4682,8 @@ L_08B11B50: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4813,33 +4740,8 @@ L_08B11B50: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5093,11 +4995,7 @@ L_08B11D58: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5112,10 +5010,7 @@ L_08B11D58: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[6] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[6]); ctx.gpr[6] = (static_cast(static_cast(static_cast(aot_mem.aot_direct_load16(ctx.gpr[4] + static_cast(2)))))); @@ -6145,15 +6040,8 @@ L_08B125FC: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[11] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[11]); ctx.gpr[11] = (std::bit_cast(ctx.fpr[12])); @@ -6161,15 +6049,8 @@ L_08B125FC: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[11] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[15] = std::bit_cast(ctx.gpr[11]); ctx.gpr[11] = (static_cast(static_cast(static_cast(aot_mem.aot_direct_load16(ctx.gpr[16] + static_cast(0)))))); @@ -6260,11 +6141,7 @@ L_08B125FC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[8] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6287,11 +6164,7 @@ L_08B125FC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6345,11 +6218,7 @@ L_08B12734: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6372,11 +6241,7 @@ L_08B12734: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[8] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6390,10 +6255,7 @@ L_08B12734: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[9] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6447,11 +6309,7 @@ L_08B1278C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[8] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6474,11 +6332,7 @@ L_08B1278C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[9] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6532,11 +6386,7 @@ L_08B127D8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[8] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6559,11 +6409,7 @@ L_08B127D8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[9] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6577,10 +6423,7 @@ L_08B127D8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6634,11 +6477,7 @@ L_08B12830: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[8] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6661,11 +6500,7 @@ L_08B12830: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[9] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6720,11 +6555,7 @@ L_08B1287C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[8] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6747,11 +6578,7 @@ L_08B1287C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[9] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6765,10 +6592,7 @@ L_08B1287C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6821,11 +6645,7 @@ L_08B128D8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6848,11 +6668,7 @@ L_08B128D8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[8] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6875,11 +6691,7 @@ L_08B128D8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[9] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6915,11 +6727,7 @@ L_08B128D8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6942,11 +6750,7 @@ L_08B128D8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[8] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6969,11 +6773,7 @@ L_08B128D8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[9] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7010,11 +6810,7 @@ L_08B128D8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7037,11 +6833,7 @@ L_08B128D8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[8] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7064,11 +6856,7 @@ L_08B128D8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[9] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7105,11 +6893,7 @@ L_08B128D8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7132,11 +6916,7 @@ L_08B128D8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[8] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7159,11 +6939,7 @@ L_08B128D8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[9] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7200,11 +6976,7 @@ L_08B128D8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7227,11 +6999,7 @@ L_08B128D8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[8] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7254,11 +7022,7 @@ L_08B128D8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[9] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7295,11 +7059,7 @@ L_08B128D8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7322,11 +7082,7 @@ L_08B128D8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[8] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7349,11 +7105,7 @@ L_08B128D8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[9] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7390,11 +7142,7 @@ L_08B128D8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7417,11 +7165,7 @@ L_08B128D8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[8] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7444,11 +7188,7 @@ L_08B128D8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[9] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7485,11 +7225,7 @@ L_08B128D8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7512,11 +7248,7 @@ L_08B128D8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[8] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7539,11 +7271,7 @@ L_08B128D8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[9] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8346,11 +8074,7 @@ L_08B12FC8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8514,11 +8238,7 @@ L_08B130DC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8996,11 +8716,7 @@ L_08B13560: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9025,11 +8741,7 @@ L_08B13560: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[23] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0196.cpp b/profiles/vcs/generated/generated_unit_0196.cpp index 279e431..3310bfb 100644 --- a/profiles/vcs/generated/generated_unit_0196.cpp +++ b/profiles/vcs/generated/generated_unit_0196.cpp @@ -1074,10 +1074,7 @@ L_08B14020: L_08B14028: ctx.gpr[5] = (std::bit_cast(ctx.fpr[13])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 17u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<0u>()); ctx.fpr[15] = std::bit_cast(ctx.gpr[5]); ctx.fpr[16] = std::bit_cast(ctx.gpr[5]); @@ -1119,24 +1116,10 @@ L_08B1405C: { const float vfpu_constant = std::bit_cast(0x3FC90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 64u, 32u, 1u, 2u>(); + ctx.execute_vfpu_vec3_ct<64u, 32u, 0u, 1u, 1u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<64u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[5]); aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(36), std::bit_cast(ctx.fpr[13])); @@ -1152,24 +1135,10 @@ L_08B1405C: { const float vfpu_constant = std::bit_cast(0x3FC90FDBu); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::asin(vfpu_s[i]) * 0.63661977236758134308f; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 1u, 5u>(); + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 23u>(); + ctx.execute_vfpu_vec3_ct<0u, 64u, 32u, 1u, 2u>(); + ctx.execute_vfpu_vec3_ct<64u, 32u, 0u, 1u, 1u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<64u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[5]); aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(32), std::bit_cast(ctx.fpr[13])); @@ -4338,11 +4307,7 @@ L_08B15B4C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[23] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -4623,11 +4588,7 @@ L_08B15D70: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0197.cpp b/profiles/vcs/generated/generated_unit_0197.cpp index d417d2a..c488640 100644 --- a/profiles/vcs/generated/generated_unit_0197.cpp +++ b/profiles/vcs/generated/generated_unit_0197.cpp @@ -5395,28 +5395,7 @@ L_08B1A184: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<8u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 4u, 3u); - ctx.read_vfpu_vector_ct<8u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 4u, 8u, 3u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5602,19 +5581,9 @@ L_08B1A320: ctx.gpr[5] = (std::bit_cast(ctx.fpr[15])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(48), std::bit_cast(ctx.fpr[12])); @@ -5622,19 +5591,9 @@ L_08B1A320: ctx.gpr[5] = (std::bit_cast(ctx.fpr[15])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(52), std::bit_cast(ctx.fpr[13])); @@ -5745,28 +5704,7 @@ L_08B1A320: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5789,11 +5727,7 @@ L_08B1A320: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[23] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[23] + static_cast(0); @@ -5898,28 +5832,7 @@ L_08B1A46C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5942,11 +5855,7 @@ L_08B1A46C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6249,10 +6158,7 @@ L_08B1A6D0: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[6] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[6]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[13] < ctx.fpr[12])) ? 0x00800000u : 0u); @@ -6309,11 +6215,7 @@ L_08B1A720: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6415,10 +6317,7 @@ L_08B1A7E8: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[6] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[6]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[13] < ctx.fpr[12])) ? 0x00800000u : 0u); @@ -6476,11 +6375,7 @@ L_08B1A838: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6510,19 +6405,9 @@ L_08B1A890: ctx.gpr[6] = (std::bit_cast(ctx.fpr[13])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[5]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[6]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.fpr[13] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[16] + static_cast(1932))); @@ -7289,10 +7174,7 @@ L_08B1AE28: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (15948u << 16u); @@ -7616,10 +7498,7 @@ L_08B1B054: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (15948u << 16u); diff --git a/profiles/vcs/generated/generated_unit_0198.cpp b/profiles/vcs/generated/generated_unit_0198.cpp index df0cb15..08a98bb 100644 --- a/profiles/vcs/generated/generated_unit_0198.cpp +++ b/profiles/vcs/generated/generated_unit_0198.cpp @@ -2728,11 +2728,7 @@ L_08B1D050: { const float vfpu_constant = std::bit_cast(0x3EA2F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 32u, 1u, 2u>(); { const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -2764,40 +2760,19 @@ L_08B1D07C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } ctx.vfpu_ctrl[0u] = 0x000C001Bu; - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 4u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<4u, 4u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<4u, 7u, 4u, 0u>(); ctx.vfpu_ctrl[0u] = 0x0009004Eu; - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 4u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<5u, 4u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<5u, 7u, 4u, 0u>(); ctx.vfpu_ctrl[0u] = 0x000A00B1u; - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 4u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<6u, 4u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<6u, 7u, 4u, 0u>(); ctx.vfpu_ctrl[0u] = 0x0004001Bu; - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 4u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<8u, 4u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<8u, 7u, 4u, 0u>(); ctx.vfpu_ctrl[0u] = 0x0001004Eu; - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 4u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<9u, 4u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<9u, 7u, 4u, 0u>(); ctx.vfpu_ctrl[0u] = 0x000200B1u; - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 4u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<10u, 4u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<10u, 7u, 4u, 0u>(); ctx.vfpu_ctrl[0u] = 0x000700E4u; - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 4u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<11u, 4u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<11u, 7u, 4u, 0u>(); { float vfpu_s[16]{}, vfpu_t[16]{}, vfpu_d[16]{}; ctx.read_vfpu_matrix_ct<4u, 4u>(vfpu_s); ctx.read_vfpu_matrix_ct<40u, 4u>(vfpu_t); @@ -7899,11 +7874,7 @@ L_08B1FB98: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); diff --git a/profiles/vcs/generated/generated_unit_0199.cpp b/profiles/vcs/generated/generated_unit_0199.cpp index b7947bf..9da9e7b 100644 --- a/profiles/vcs/generated/generated_unit_0199.cpp +++ b/profiles/vcs/generated/generated_unit_0199.cpp @@ -4177,11 +4177,7 @@ L_08B21AE4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4210,10 +4206,7 @@ L_08B21AE4: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16880u << 16u); @@ -4825,10 +4818,7 @@ L_08B21ED0: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (15395u << 16u); @@ -5239,11 +5229,7 @@ L_08B221A0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(176)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -5253,10 +5239,7 @@ L_08B221A0: ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -5664,11 +5647,7 @@ L_08B2242C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(208)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -5678,10 +5657,7 @@ L_08B2242C: ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -6094,11 +6070,7 @@ L_08B22758: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6108,10 +6080,7 @@ L_08B22758: ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -6285,11 +6254,7 @@ L_08B228AC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6299,10 +6264,7 @@ L_08B228AC: ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -7118,33 +7080,8 @@ L_08B22E44: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7184,30 +7121,10 @@ L_08B22E6C: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -7411,15 +7328,7 @@ L_08B23094: ctx.gpr[9] = (static_cast(static_cast(static_cast(aot_mem.aot_direct_load16(ctx.gpr[9] + static_cast(2)))))); ctx.set_vfpu_scalar_bits_ct<1u>(ctx.gpr[8]); ctx.set_vfpu_scalar_bits_ct<33u>(ctx.gpr[9]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); aot_mem.aot_direct_store32(ctx.gpr[19] + static_cast(0), ctx.vfpu_scalar_bits_ct<2u>()); aot_mem.aot_direct_store32(ctx.gpr[19] + static_cast(4), ctx.vfpu_scalar_bits_ct<34u>()); ctx.fpr[12] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[29] + static_cast(216))); @@ -7560,11 +7469,7 @@ L_08B231E0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[18] = (ctx.gpr[29] + static_cast(304)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); @@ -7626,11 +7531,7 @@ L_08B23230: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8333,11 +8234,7 @@ L_08B23774: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[16] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); @@ -8367,10 +8264,7 @@ L_08B23774: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[7] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[22] = std::bit_cast(ctx.gpr[7]); { const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8391,11 +8285,7 @@ L_08B23774: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[17] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); @@ -8460,10 +8350,7 @@ L_08B23774: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -8491,10 +8378,7 @@ L_08B23774: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -8524,11 +8408,7 @@ L_08B238B0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8862,11 +8742,7 @@ L_08B23B3C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9120,11 +8996,7 @@ L_08B23D60: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(336)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -9361,11 +9233,7 @@ L_08B23F54: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[23] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9405,11 +9273,7 @@ L_08B23F54: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0200.cpp b/profiles/vcs/generated/generated_unit_0200.cpp index dd01bc9..21017d5 100644 --- a/profiles/vcs/generated/generated_unit_0200.cpp +++ b/profiles/vcs/generated/generated_unit_0200.cpp @@ -1251,11 +1251,7 @@ L_08B242C4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1283,10 +1279,7 @@ L_08B242C4: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -1359,11 +1352,7 @@ L_08B24370: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[20] = (ctx.gpr[29] + static_cast(96)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); @@ -1413,11 +1402,7 @@ L_08B243B8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[23] = (ctx.gpr[29] + static_cast(144)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[23] + static_cast(0); @@ -1471,11 +1456,7 @@ L_08B24404: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[22] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1522,11 +1503,7 @@ L_08B2443C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1589,11 +1566,7 @@ L_08B2448C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[18] = (ctx.gpr[29] + static_cast(192)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); @@ -1643,11 +1616,7 @@ L_08B244D0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(240)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -4196,11 +4165,7 @@ L_08B2584C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[23] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5093,11 +5058,7 @@ L_08B25EBC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5112,10 +5073,7 @@ L_08B25EBC: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[6] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[6]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[13] < ctx.fpr[12])) ? 0x00800000u : 0u); @@ -5685,11 +5643,7 @@ L_08B26320: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6763,33 +6717,8 @@ L_08B26D30: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0201.cpp b/profiles/vcs/generated/generated_unit_0201.cpp index 4a73346..c280e28 100644 --- a/profiles/vcs/generated/generated_unit_0201.cpp +++ b/profiles/vcs/generated/generated_unit_0201.cpp @@ -1860,11 +1860,7 @@ L_08B2864C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2402,11 +2398,7 @@ L_08B28A68: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3330,33 +3322,8 @@ L_08B2917C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3417,28 +3384,7 @@ L_08B2917C: ctx.write_vfpu_vector_ct<39u, 4u>(vfpu_value); } ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[5]); - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 4u); - ctx.read_vfpu_vector_ct<12u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 14u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<14u, 36u, 12u, 4u, 3u>(); ctx.vfpu_ctrl[1u] = 0x00000000u; ctx.execute_vfpu_vcmp_ct<14u, 0u, 4u, 3u>(); ctx.gpr[5] = (0u + static_cast(0)); diff --git a/profiles/vcs/generated/generated_unit_0202.cpp b/profiles/vcs/generated/generated_unit_0202.cpp index bff0c0e..f5d8707 100644 --- a/profiles/vcs/generated/generated_unit_0202.cpp +++ b/profiles/vcs/generated/generated_unit_0202.cpp @@ -1395,28 +1395,7 @@ L_08B2C398: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0203.cpp b/profiles/vcs/generated/generated_unit_0203.cpp index d8df26f..ec14da8 100644 --- a/profiles/vcs/generated/generated_unit_0203.cpp +++ b/profiles/vcs/generated/generated_unit_0203.cpp @@ -3555,11 +3555,7 @@ L_08B316B4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(176)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6748,33 +6744,8 @@ L_08B32D3C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6790,19 +6761,9 @@ L_08B32D64: ctx.gpr[5] = (std::bit_cast(ctx.fpr[13])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -8859,10 +8820,7 @@ L_08B33D6C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -8922,11 +8880,7 @@ L_08B33D6C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[30] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0204.cpp b/profiles/vcs/generated/generated_unit_0204.cpp index eef8f41..efdc88a 100644 --- a/profiles/vcs/generated/generated_unit_0204.cpp +++ b/profiles/vcs/generated/generated_unit_0204.cpp @@ -1856,15 +1856,8 @@ L_08B345C4: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); @@ -1872,15 +1865,8 @@ L_08B345C4: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (ctx.gpr[5] + static_cast(16)); @@ -1909,7 +1895,7 @@ L_08B345C4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -3954,11 +3940,7 @@ L_08B354BC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3987,10 +3969,7 @@ L_08B354BC: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16076u << 16u); @@ -8841,11 +8820,7 @@ L_08B37790: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8888,11 +8863,7 @@ L_08B37790: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8938,11 +8909,7 @@ L_08B37790: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8988,11 +8955,7 @@ L_08B37790: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9072,11 +9035,7 @@ L_08B37978: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(288)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -9115,11 +9074,7 @@ L_08B37978: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(304)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -9172,11 +9127,7 @@ L_08B379F0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(304)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -9215,11 +9166,7 @@ L_08B379F0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(288)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -9282,11 +9229,7 @@ L_08B37A78: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(224)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -9305,10 +9248,7 @@ L_08B37A78: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -9352,11 +9292,7 @@ L_08B37A78: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(240)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -9421,7 +9357,7 @@ L_08B37B20: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[30] = (ctx.gpr[29] + static_cast(336)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[30] + static_cast(0); @@ -9452,11 +9388,7 @@ L_08B37B20: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[6] = (ctx.gpr[29] + static_cast(368)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[6] + static_cast(0); @@ -9472,10 +9404,7 @@ L_08B37B20: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[6] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[6]); ctx.fpr[22] = std::bit_cast(std::bit_cast(ctx.fpr[12]) ^ 0x80000000u); @@ -9497,11 +9426,7 @@ L_08B37B20: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[20] = (ctx.gpr[29] + static_cast(384)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); @@ -9517,10 +9442,7 @@ L_08B37B20: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[24] = std::bit_cast(ctx.gpr[4]); ctx.gpr[31] = (0x08B37BDCu); @@ -9604,10 +9526,7 @@ L_08B37C5C: ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[23] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); diff --git a/profiles/vcs/generated/generated_unit_0205.cpp b/profiles/vcs/generated/generated_unit_0205.cpp index 3093084..4f5c21c 100644 --- a/profiles/vcs/generated/generated_unit_0205.cpp +++ b/profiles/vcs/generated/generated_unit_0205.cpp @@ -9486,7 +9486,7 @@ L_08B3BAB4: ctx.gpr[5] = (aot_mem.aot_direct_load16(ctx.gpr[5] + static_cast(4))); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[18]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - ctx.execute_vfpu_vh2f(1u, 0u, 2u); + ctx.execute_vfpu_vh2f_ct<1u, 0u, 2u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9512,11 +9512,7 @@ L_08B3BAB4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -9547,11 +9543,7 @@ L_08B3BAB4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[22] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/generated/generated_unit_0206.cpp b/profiles/vcs/generated/generated_unit_0206.cpp index dd88103..bcb301a 100644 --- a/profiles/vcs/generated/generated_unit_0206.cpp +++ b/profiles/vcs/generated/generated_unit_0206.cpp @@ -5802,30 +5802,10 @@ L_08B3E498: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -5877,30 +5857,10 @@ L_08B3E4DC: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -5942,30 +5902,10 @@ L_08B3E4DC: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -5992,11 +5932,7 @@ L_08B3E4DC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6058,30 +5994,10 @@ L_08B3E5C0: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -6123,30 +6039,10 @@ L_08B3E5C0: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } diff --git a/profiles/vcs/generated/generated_unit_0207.cpp b/profiles/vcs/generated/generated_unit_0207.cpp index 3266a63..ed1da47 100644 --- a/profiles/vcs/generated/generated_unit_0207.cpp +++ b/profiles/vcs/generated/generated_unit_0207.cpp @@ -2993,30 +2993,10 @@ L_08B40E68: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -3061,10 +3041,7 @@ L_08B40EAC: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -3138,11 +3115,7 @@ L_08B40F9C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3161,10 +3134,7 @@ L_08B40F9C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -3244,15 +3214,8 @@ L_08B41050: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -3267,15 +3230,8 @@ L_08B41074: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[0] = std::bit_cast(ctx.gpr[4]); jump_target = ctx.gpr[31]; @@ -3924,11 +3880,7 @@ L_08B4154C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -3943,10 +3895,7 @@ L_08B4154C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fcr31 = (ctx.fcr31 & ~0x00800000u) | (((ctx.fpr[12] < ctx.fpr[20])) ? 0x00800000u : 0u); @@ -4158,15 +4107,7 @@ L_08B416FC: ctx.gpr[9] = (static_cast(static_cast(static_cast(aot_mem.aot_direct_load16(ctx.gpr[9] + static_cast(2)))))); ctx.set_vfpu_scalar_bits_ct<1u>(ctx.gpr[8]); ctx.set_vfpu_scalar_bits_ct<33u>(ctx.gpr[9]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); ctx.gpr[8] = (ctx.gpr[29] + static_cast(28)); aot_mem.aot_direct_store32(ctx.gpr[8] + static_cast(0), ctx.vfpu_scalar_bits_ct<2u>()); aot_mem.aot_direct_store32(ctx.gpr[8] + static_cast(4), ctx.vfpu_scalar_bits_ct<34u>()); @@ -4206,15 +4147,7 @@ L_08B4176C: ctx.gpr[9] = (static_cast(static_cast(static_cast(aot_mem.aot_direct_load16(ctx.gpr[11] + static_cast(2)))))); ctx.set_vfpu_scalar_bits_ct<1u>(ctx.gpr[8]); ctx.set_vfpu_scalar_bits_ct<33u>(ctx.gpr[9]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); ctx.gpr[8] = (ctx.gpr[2] | 0u); aot_mem.aot_direct_store32(ctx.gpr[8] + static_cast(0), ctx.vfpu_scalar_bits_ct<2u>()); aot_mem.aot_direct_store32(ctx.gpr[8] + static_cast(4), ctx.vfpu_scalar_bits_ct<34u>()); @@ -4314,15 +4247,7 @@ L_08B4189C: ctx.gpr[9] = (static_cast(static_cast(static_cast(aot_mem.aot_direct_load16(ctx.gpr[9] + static_cast(2)))))); ctx.set_vfpu_scalar_bits_ct<1u>(ctx.gpr[8]); ctx.set_vfpu_scalar_bits_ct<33u>(ctx.gpr[9]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(60)); aot_mem.aot_direct_store32(ctx.gpr[5] + static_cast(0), ctx.vfpu_scalar_bits_ct<2u>()); aot_mem.aot_direct_store32(ctx.gpr[5] + static_cast(4), ctx.vfpu_scalar_bits_ct<34u>()); @@ -4451,11 +4376,7 @@ L_08B419F8: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5374,10 +5295,7 @@ L_08B42104: ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<1u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<1u, 1u, 1u, 16u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(16)); { const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); std::uint32_t vfpu_words[4]{}; @@ -6566,33 +6484,8 @@ L_08B42904: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6941,15 +6834,8 @@ L_08B42C34: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (ctx.gpr[16] << 2u); @@ -6972,15 +6858,8 @@ L_08B42C80: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (ctx.gpr[16] << 2u); @@ -7230,33 +7109,8 @@ L_08B42E84: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7329,33 +7183,8 @@ L_08B42EF4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[23] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7428,33 +7257,8 @@ L_08B42F58: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -7527,33 +7331,8 @@ L_08B42FBC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -8679,15 +8458,7 @@ L_08B439EC: ctx.gpr[9] = (static_cast(static_cast(static_cast(aot_mem.aot_direct_load16(ctx.gpr[9] + static_cast(2)))))); ctx.set_vfpu_scalar_bits_ct<1u>(ctx.gpr[8]); ctx.set_vfpu_scalar_bits_ct<33u>(ctx.gpr[9]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(0), ctx.vfpu_scalar_bits_ct<2u>()); aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(4), ctx.vfpu_scalar_bits_ct<34u>()); ctx.gpr[5] = (aot_mem.aot_direct_load32(ctx.gpr[28] + static_cast(-16928))); @@ -8700,15 +8471,7 @@ L_08B439EC: ctx.gpr[9] = (static_cast(static_cast(static_cast(aot_mem.aot_direct_load16(ctx.gpr[9] + static_cast(2)))))); ctx.set_vfpu_scalar_bits_ct<1u>(ctx.gpr[8]); ctx.set_vfpu_scalar_bits_ct<33u>(ctx.gpr[9]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(16)); aot_mem.aot_direct_store32(ctx.gpr[5] + static_cast(0), ctx.vfpu_scalar_bits_ct<2u>()); aot_mem.aot_direct_store32(ctx.gpr[5] + static_cast(4), ctx.vfpu_scalar_bits_ct<34u>()); @@ -8722,15 +8485,7 @@ L_08B439EC: ctx.gpr[9] = (static_cast(static_cast(static_cast(aot_mem.aot_direct_load16(ctx.gpr[9] + static_cast(2)))))); ctx.set_vfpu_scalar_bits_ct<1u>(ctx.gpr[8]); ctx.set_vfpu_scalar_bits_ct<33u>(ctx.gpr[9]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(8)); aot_mem.aot_direct_store32(ctx.gpr[4] + static_cast(0), ctx.vfpu_scalar_bits_ct<2u>()); aot_mem.aot_direct_store32(ctx.gpr[4] + static_cast(4), ctx.vfpu_scalar_bits_ct<34u>()); @@ -9078,10 +8833,7 @@ L_08B43D88: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[13] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[28] + static_cast(5304))); @@ -9210,10 +8962,7 @@ L_08B43E64: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[22] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (17008u << 16u); diff --git a/profiles/vcs/generated/generated_unit_0208.cpp b/profiles/vcs/generated/generated_unit_0208.cpp index 13c76aa..6d80419 100644 --- a/profiles/vcs/generated/generated_unit_0208.cpp +++ b/profiles/vcs/generated/generated_unit_0208.cpp @@ -1217,10 +1217,7 @@ L_08B441D4: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (15692u << 16u); @@ -1301,10 +1298,7 @@ L_08B44270: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (15692u << 16u); @@ -1466,10 +1460,7 @@ L_08B443A0: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -1505,10 +1496,7 @@ L_08B443A0: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -1540,11 +1528,7 @@ L_08B44414: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1649,11 +1633,7 @@ L_08B4449C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(80)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1994,10 +1974,7 @@ L_08B44730: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (17008u << 16u); @@ -2133,11 +2110,7 @@ L_08B448A0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2152,10 +2125,7 @@ L_08B448A0: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[30] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16800u << 16u); @@ -2315,10 +2285,7 @@ L_08B449F4: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); { const float fs = ctx.fpr[12]; const float ft = ctx.fpr[28]; if ((std::isinf(fs) && ft == 0.0f) || (std::isinf(ft) && fs == 0.0f)) ctx.fpr[28] = std::bit_cast(0x7FC00000u); else ctx.fpr[28] = fs * ft; } @@ -2688,30 +2655,10 @@ L_08B44D70: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -2753,30 +2700,10 @@ L_08B44D70: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -5105,11 +5032,7 @@ L_08B45E0C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5125,10 +5048,7 @@ L_08B45E0C: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (16816u << 16u); @@ -5438,11 +5358,7 @@ L_08B45FE4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -5608,11 +5524,7 @@ L_08B460EC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(64)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7099,11 +7011,7 @@ L_08B46D20: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(352)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); diff --git a/profiles/vcs/generated/generated_unit_0209.cpp b/profiles/vcs/generated/generated_unit_0209.cpp index 838547d..28a31f9 100644 --- a/profiles/vcs/generated/generated_unit_0209.cpp +++ b/profiles/vcs/generated/generated_unit_0209.cpp @@ -943,15 +943,7 @@ L_08B48110: ctx.gpr[9] = (static_cast(static_cast(static_cast(aot_mem.aot_direct_load16(ctx.gpr[9] + static_cast(2)))))); ctx.set_vfpu_scalar_bits_ct<1u>(ctx.gpr[8]); ctx.set_vfpu_scalar_bits_ct<33u>(ctx.gpr[9]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(532)); aot_mem.aot_direct_store32(ctx.gpr[4] + static_cast(0), ctx.vfpu_scalar_bits_ct<2u>()); aot_mem.aot_direct_store32(ctx.gpr[4] + static_cast(4), ctx.vfpu_scalar_bits_ct<34u>()); @@ -966,15 +958,7 @@ L_08B48110: ctx.gpr[9] = (static_cast(static_cast(static_cast(aot_mem.aot_direct_load16(ctx.gpr[9] + static_cast(2)))))); ctx.set_vfpu_scalar_bits_ct<1u>(ctx.gpr[8]); ctx.set_vfpu_scalar_bits_ct<33u>(ctx.gpr[9]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(540)); aot_mem.aot_direct_store32(ctx.gpr[4] + static_cast(0), ctx.vfpu_scalar_bits_ct<2u>()); aot_mem.aot_direct_store32(ctx.gpr[4] + static_cast(4), ctx.vfpu_scalar_bits_ct<34u>()); @@ -1236,30 +1220,10 @@ L_08B4840C: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -1303,30 +1267,10 @@ L_08B4840C: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -1353,11 +1297,7 @@ L_08B4840C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(480)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -1437,10 +1377,7 @@ L_08B48540: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -1551,30 +1488,10 @@ L_08B48630: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -1618,30 +1535,10 @@ L_08B48630: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -1668,11 +1565,7 @@ L_08B48630: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(512)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -1702,10 +1595,7 @@ L_08B48630: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[7] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[7]); ctx.fpr[12] = ctx.fpr[12] / ctx.fpr[13]; @@ -1744,11 +1634,7 @@ L_08B48630: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1802,30 +1688,10 @@ L_08B48630: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -1874,30 +1740,10 @@ L_08B48630: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -2288,10 +2134,7 @@ L_08B48B20: ctx.fpr[20] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[20])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); ctx.gpr[17] = (ctx.gpr[29] + static_cast(128)); { const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); std::uint32_t vfpu_words[4]{}; @@ -2329,10 +2172,7 @@ L_08B48B20: L_08B48B5C: ctx.gpr[4] = (std::bit_cast(ctx.fpr[20])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); { const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); std::uint32_t vfpu_words[4]{}; aot_mem.aot_direct_load32_block(vfpu_address, vfpu_words); @@ -2367,11 +2207,7 @@ L_08B48B5C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2410,11 +2246,7 @@ L_08B48B5C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[20] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2603,11 +2435,7 @@ L_08B48CB4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(640)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2623,10 +2451,7 @@ L_08B48CB4: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (2236u << 16u); @@ -2733,11 +2558,7 @@ L_08B48D80: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(640)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -2753,10 +2574,7 @@ L_08B48D80: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[5]); ctx.fpr[13] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[4] + static_cast(1952))); @@ -2875,11 +2693,7 @@ L_08B48E50: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(656)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2895,10 +2709,7 @@ L_08B48E50: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (static_cast(static_cast(static_cast(aot_mem.aot_direct_load8(ctx.gpr[22] + static_cast(614)))))); @@ -3875,11 +3686,7 @@ L_08B49544: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(720)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -4698,11 +4505,7 @@ L_08B49CC4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8058,15 +7861,7 @@ L_08B4B924: ctx.gpr[9] = (static_cast(static_cast(static_cast(aot_mem.aot_direct_load16(ctx.gpr[9] + static_cast(2)))))); ctx.set_vfpu_scalar_bits_ct<1u>(ctx.gpr[8]); ctx.set_vfpu_scalar_bits_ct<33u>(ctx.gpr[9]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(12)); aot_mem.aot_direct_store32(ctx.gpr[4] + static_cast(0), ctx.vfpu_scalar_bits_ct<2u>()); aot_mem.aot_direct_store32(ctx.gpr[4] + static_cast(4), ctx.vfpu_scalar_bits_ct<34u>()); @@ -8081,15 +7876,7 @@ L_08B4B924: ctx.gpr[9] = (static_cast(static_cast(static_cast(aot_mem.aot_direct_load16(ctx.gpr[9] + static_cast(2)))))); ctx.set_vfpu_scalar_bits_ct<1u>(ctx.gpr[8]); ctx.set_vfpu_scalar_bits_ct<33u>(ctx.gpr[9]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(20)); aot_mem.aot_direct_store32(ctx.gpr[4] + static_cast(0), ctx.vfpu_scalar_bits_ct<2u>()); aot_mem.aot_direct_store32(ctx.gpr[4] + static_cast(4), ctx.vfpu_scalar_bits_ct<34u>()); @@ -8326,30 +8113,10 @@ L_08B4BB70: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -8646,15 +8413,7 @@ L_08B4BEC0: ctx.gpr[9] = (static_cast(static_cast(static_cast(aot_mem.aot_direct_load16(ctx.gpr[9] + static_cast(2)))))); ctx.set_vfpu_scalar_bits_ct<1u>(ctx.gpr[8]); ctx.set_vfpu_scalar_bits_ct<33u>(ctx.gpr[9]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); aot_mem.aot_direct_store32(ctx.gpr[22] + static_cast(0), ctx.vfpu_scalar_bits_ct<2u>()); aot_mem.aot_direct_store32(ctx.gpr[22] + static_cast(4), ctx.vfpu_scalar_bits_ct<34u>()); ctx.gpr[4] = (aot_mem.aot_direct_load32(ctx.gpr[28] + static_cast(-16928))); @@ -8664,15 +8423,7 @@ L_08B4BEC0: ctx.gpr[9] = (static_cast(static_cast(static_cast(aot_mem.aot_direct_load16(ctx.gpr[9] + static_cast(2)))))); ctx.set_vfpu_scalar_bits_ct<1u>(ctx.gpr[8]); ctx.set_vfpu_scalar_bits_ct<33u>(ctx.gpr[9]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); aot_mem.aot_direct_store32(ctx.gpr[21] + static_cast(0), ctx.vfpu_scalar_bits_ct<2u>()); aot_mem.aot_direct_store32(ctx.gpr[21] + static_cast(4), ctx.vfpu_scalar_bits_ct<34u>()); ctx.fpr[13] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[29] + static_cast(76))); diff --git a/profiles/vcs/generated/generated_unit_0210.cpp b/profiles/vcs/generated/generated_unit_0210.cpp index b7c0a45..2ce96f0 100644 --- a/profiles/vcs/generated/generated_unit_0210.cpp +++ b/profiles/vcs/generated/generated_unit_0210.cpp @@ -1202,11 +1202,7 @@ L_08B4C3FC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -2120,15 +2116,8 @@ L_08B4CC18: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); @@ -2136,15 +2125,8 @@ L_08B4CC18: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[4]); ctx.fpr[12] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[29] + static_cast(52))); @@ -2183,7 +2165,7 @@ L_08B4CC18: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2227,11 +2209,7 @@ L_08B4CC18: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[5] = (ctx.gpr[29] + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[5] + static_cast(0); @@ -2525,11 +2503,7 @@ L_08B4CF40: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2545,10 +2519,7 @@ L_08B4CF40: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[13] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[28] + static_cast(5340))); @@ -3864,10 +3835,7 @@ L_08B4DBA8: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[13] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[28] + static_cast(5372))); @@ -4544,19 +4512,9 @@ L_08B4E0F4: ctx.gpr[6] = (std::bit_cast(ctx.fpr[14])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[5]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[6]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[5]); ctx.gpr[5] = (ctx.gpr[16] + static_cast(320)); @@ -4773,19 +4731,9 @@ L_08B4E2F0: ctx.gpr[5] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.fpr[13] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[16] + static_cast(120))); @@ -5041,10 +4989,7 @@ L_08B4E508: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -5057,15 +5002,8 @@ L_08B4E508: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[5]); aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(48), std::bit_cast(ctx.fpr[13])); @@ -5074,15 +5012,8 @@ L_08B4E508: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[5]); aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(52), std::bit_cast(ctx.fpr[13])); @@ -5106,7 +5037,7 @@ L_08B4E508: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5143,7 +5074,7 @@ L_08B4E508: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5366,19 +5297,9 @@ L_08B4E79C: ctx.gpr[5] = (std::bit_cast(ctx.fpr[13])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.fpr[14] = std::bit_cast(aot_mem.aot_direct_load32(ctx.gpr[16] + static_cast(1720))); @@ -5467,10 +5388,7 @@ L_08B4E84C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -5483,15 +5401,8 @@ L_08B4E84C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::cos(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 19u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[5]); aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(48), std::bit_cast(ctx.fpr[14])); @@ -5500,15 +5411,8 @@ L_08B4E84C: { const float vfpu_constant = std::bit_cast(0x3F22F983u); const float vfpu_value[4]{vfpu_constant, vfpu_constant, vfpu_constant, vfpu_constant}; ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::sin(vfpu_s[i] * 1.57079632679489661923f); - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<64u, 0u, 32u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<1u, 64u, 1u, 18u>(); ctx.gpr[5] = (ctx.vfpu_scalar_bits_ct<1u>()); ctx.fpr[14] = std::bit_cast(ctx.gpr[5]); aot_mem.aot_direct_store32(ctx.gpr[29] + static_cast(52), std::bit_cast(ctx.fpr[14])); @@ -5527,10 +5431,7 @@ L_08B4E84C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -5555,7 +5456,7 @@ L_08B4E84C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5592,10 +5493,7 @@ L_08B4E84C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -5620,7 +5518,7 @@ L_08B4E84C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5651,10 +5549,7 @@ L_08B4E84C: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -5679,7 +5574,7 @@ L_08B4E84C: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } - ctx.execute_vfpu_cross_quat(0u, 1u, 2u, 3u); + ctx.execute_vfpu_cross_quat_ct<0u, 1u, 2u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[29] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -5723,19 +5618,9 @@ L_08B4E84C: ctx.gpr[5] = (std::bit_cast(ctx.fpr[16])); ctx.set_vfpu_scalar_bits_ct<0u>(ctx.gpr[4]); ctx.set_vfpu_scalar_bits_ct<32u>(ctx.gpr[5]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::log2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::exp2(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<32u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<64u, 0u, 1u, 21u>(); + ctx.execute_vfpu_vec3_ct<0u, 32u, 64u, 1u, 2u>(); + ctx.execute_vfpu_unary_ct<32u, 0u, 1u, 20u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<32u>()); ctx.fpr[16] = std::bit_cast(ctx.gpr[4]); ctx.fpr[16] = ctx.fpr[12] - ctx.fpr[16]; @@ -6075,10 +5960,7 @@ L_08B4ED88: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 2u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (15395u << 16u); @@ -6167,11 +6049,7 @@ L_08B4EE18: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[19] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[19] + static_cast(0); @@ -6204,10 +6082,7 @@ L_08B4EE74: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -6250,11 +6125,7 @@ L_08B4EE74: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[21] = (ctx.gpr[29] + static_cast(160)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); @@ -6378,10 +6249,7 @@ L_08B4EF88: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[20] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (17096u << 16u); @@ -6447,11 +6315,7 @@ L_08B4EFF4: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6524,10 +6388,7 @@ L_08B4F068: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -6572,11 +6433,7 @@ L_08B4F068: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[16] = (ctx.gpr[29] + static_cast(192)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); @@ -6629,11 +6486,7 @@ L_08B4F068: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6729,11 +6582,7 @@ L_08B4F198: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(240)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6749,10 +6598,7 @@ L_08B4F198: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[26] = std::bit_cast(ctx.gpr[4]); goto L_08B4F1DC; @@ -6792,11 +6638,7 @@ L_08B4F1DC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6848,11 +6690,7 @@ L_08B4F1DC: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -6949,11 +6787,7 @@ L_08B4F2C0: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(272)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -6969,10 +6803,7 @@ L_08B4F2C0: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[24] = std::bit_cast(ctx.gpr[4]); goto L_08B4F304; @@ -7282,11 +7113,7 @@ L_08B4F548: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(320)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7311,11 +7138,7 @@ L_08B4F548: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(336)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -7332,10 +7155,7 @@ L_08B4F548: std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); ctx.gpr[4] = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[13] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (aot_mem.aot_direct_load8(ctx.gpr[20] + static_cast(537))); @@ -7640,10 +7460,7 @@ L_08B4F7D0: ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 1u, 1u, 3u>(); ctx.execute_vfpu_vcmp_ct<28u, 32u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<1u, 1u, 28u, 3u>(); ctx.execute_vfpu_vcmov_ct<1u, 0u, 3u, 0u, false>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<1u, 4u>(vfpu_value); @@ -8003,11 +7820,7 @@ L_08B4FA88: std::bit_cast(vfpu_words[2]), std::bit_cast(vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(192)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -8111,30 +7924,10 @@ L_08B4FB28: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -8182,30 +7975,10 @@ L_08B4FB28: } const float vfpu_value[1]{std::bit_cast(vfpu_bits)}; ctx.write_vfpu_vector_with_destination_prefix_ct<64u, 1u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<2u, 2u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<32u, 1u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<1u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(0u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 1u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<66u, 1u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>(); + ctx.execute_vfpu_vi2f_ct<66u, 32u, 1u, 0u>(); ctx.execute_vfpu_vcmp_ct<66u, 96u, 1u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<66u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<64u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<98u, 66u, 64u, 1u, 1u>(); ctx.execute_vfpu_vcmov_ct<66u, 98u, 1u, 0u, false>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<98u, 1u>(vfpu_value); } @@ -8367,10 +8140,7 @@ L_08B4FD78: ctx.fpr[12] = std::bit_cast(ctx.gpr[4]); ctx.gpr[4] = (std::bit_cast(ctx.fpr[12])); ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[4]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 16u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(112)); { const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); std::uint32_t vfpu_words[4]{}; diff --git a/profiles/vcs/generated/generated_unit_0219.cpp b/profiles/vcs/generated/generated_unit_0219.cpp index fcb5e4d..de5474f 100644 --- a/profiles/vcs/generated/generated_unit_0219.cpp +++ b/profiles/vcs/generated/generated_unit_0219.cpp @@ -8340,11 +8340,7 @@ L_08B7390C: L_08B73924: rt.unsupported(0x08B73924u, 0x6E72654Bu, "vfpu3 not lowered yet"); return; L_08B73938: - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<111u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<97u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<76u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<76u, 111u, 97u, 1u, 2u>(); rt.unsupported(0x08B7393Cu, 0x63657845u, "vfpu0 not lowered yet"); return; L_08B73948: // nop @@ -8358,7 +8354,7 @@ L_08B73964: } goto L_08B7396C; L_08B7396C: - ctx.execute_vfpu_compare3(110u, 100u, 70u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<110u, 100u, 70u, 1u, 6u>(); ctx.execute_vfpu_vscl_ct<114u, 85u, 115u, 1u>(); rt.unsupported(0x08B73974u, 0x00000072u, "special? not lowered yet"); return; L_08B7397C: @@ -8375,7 +8371,7 @@ L_08B739D4: rt.unsupported(0x08B739D4u, 0x41656373u, "unknown not lowered yet"); return; L_08B739E4: if (ctx.gpr[27] == ctx.gpr[5]) { - ctx.execute_vfpu_compare3(97u, 115u, 67u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<97u, 115u, 67u, 1u, 6u>(); (void)rt.invoke_chained_direct<&recomp_unit_0226_entry, 226u, 63u, 0x08B8C7B4u>(ctx, &aot_mem); return; } goto L_08B739EC; @@ -8480,7 +8476,7 @@ L_08B73B78: goto L_08B73B80; } L_08B73B80: - ctx.execute_vfpu_compare3(27u, 116u, 18u, 1u, 7u); + ctx.execute_vfpu_compare3_ct<27u, 116u, 18u, 1u, 7u>(); rt.unsupported(0x08B73B84u, 0x7F27BB5Eu, "special3? not lowered yet"); return; L_08B73BB0: ctx.gpr[25] = (static_cast(0u) < 10409 ? 1u : 0u); @@ -8557,11 +8553,7 @@ L_08B73CC4: } goto L_08B73CCC; L_08B73CCC: - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<86u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<78u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<7u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<7u, 86u, 78u, 1u, 2u>(); rt.unsupported(0x08B73CD0u, 0x6A8C3CD5u, "unknown not lowered yet"); return; L_08B73CF0: ctx.gpr[4] = (ctx.gpr[12] < ctx.gpr[27] ? 1u : 0u); @@ -8577,7 +8569,7 @@ L_08B73D44: rt.unsupported(0x08B73D50u, 0x13F592BCu, "control flow in delay slot"); return; L_08B73D54: if (static_cast(ctx.gpr[22]) > 0) { - ctx.execute_vfpu_compare3(83u, 104u, 68u, 1u, 7u); + ctx.execute_vfpu_compare3_ct<83u, 104u, 68u, 1u, 7u>(); (void)rt.invoke_chained_direct<&recomp_unit_0221_entry, 221u, 98u, 0x08B7AEACu>(ctx, &aot_mem); return; } goto L_08B73D5C; diff --git a/profiles/vcs/generated/generated_unit_0220.cpp b/profiles/vcs/generated/generated_unit_0220.cpp index da800e8..cf9ab2d 100644 --- a/profiles/vcs/generated/generated_unit_0220.cpp +++ b/profiles/vcs/generated/generated_unit_0220.cpp @@ -153,7 +153,7 @@ LOCAL_DISPATCH: } } L_08B74000: - ctx.execute_vfpu_vhdp(32u, 105u, 110u, 1u); + ctx.execute_vfpu_vhdp_ct<32u, 105u, 110u, 1u>(); ctx.gpr[10] = (static_cast(ctx.gpr[17]) < 8303 ? 1u : 0u); ctx.gpr[5] = (static_cast(0u) < static_cast(0u) ? 1u : 0u); goto L_08B7400C; @@ -164,11 +164,7 @@ L_08B74044: L_08B74060: rt.unsupported(0x08B74060u, 0x2064253Du, "unknown not lowered yet"); return; L_08B7406C: - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<61u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<37u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<104u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<104u, 61u, 37u, 1u, 2u>(); if (0u == 0u) (void)(0u); goto L_08B74074; L_08B74074: @@ -194,7 +190,7 @@ L_08B7427C: ctx.execute_vfpu_vscl_ct<108u, 101u, 118u, 1u>(); rt.unsupported(0x08B74280u, 0x756D206Cu, "unknown not lowered yet"); return; L_08B74298: - ctx.execute_vfpu_vhdp(110u, 111u, 32u, 1u); + ctx.execute_vfpu_vhdp_ct<110u, 111u, 32u, 1u>(); rt.unsupported(0x08B7429Cu, 0x74636E75u, "unknown not lowered yet"); return; L_08B742CC: rt.unsupported(0x08B742CCu, 0x74657360u, "unknown not lowered yet"); return; @@ -202,11 +198,11 @@ L_08B742FC: rt.unsupported(0x08B742FCu, 0x74657360u, "unknown not lowered yet"); return; L_08B74334: ctx.execute_vfpu_vscl_ct<97u, 115u, 115u, 1u>(); - ctx.execute_vfpu_compare3(114u, 116u, 105u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<114u, 116u, 105u, 1u, 6u>(); rt.unsupported(0x08B7433Cu, 0x6166206Eu, "vfpu0 not lowered yet"); return; L_08B74348: ctx.execute_vfpu_vcmp_ct<97u, 98u, 1u, 4u>(); - ctx.execute_vfpu_compare3(101u, 32u, 116u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<101u, 32u, 116u, 1u, 6u>(); rt.unsupported(0x08B74350u, 0x6962206Fu, "unknown not lowered yet"); return; L_08B74360: ctx.execute_vfpu_vcmp_ct<111u, 111u, 1u, 2u>(); @@ -217,19 +213,19 @@ L_08B743A0: rt.unsupported(0x08B743A0u, 0x4F4C5F60u, "unknown not lowered yet"); return; L_08B743BC: ctx.execute_vfpu_vcmp_ct<111u, 117u, 1u, 3u>(); - ctx.execute_vfpu_compare3(100u, 32u, 110u, 1u, 6u); - ctx.execute_vfpu_compare3(116u, 32u, 108u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<100u, 32u, 110u, 1u, 6u>(); + ctx.execute_vfpu_compare3_ct<116u, 32u, 108u, 1u, 6u>(); rt.unsupported(0x08B743C8u, 0x70206461u, "unknown not lowered yet"); return; L_08B743E8: - ctx.execute_vfpu_compare3(101u, 114u, 114u, 1u, 6u); - ctx.execute_vfpu_compare3(114u, 32u, 108u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<101u, 114u, 114u, 1u, 6u>(); + ctx.execute_vfpu_compare3_ct<114u, 32u, 108u, 1u, 6u>(); rt.unsupported(0x08B743F0u, 0x6E696461u, "vfpu3 not lowered yet"); return; L_08B74408: rt.unsupported(0x08B74408u, 0x206F6F74u, "unknown not lowered yet"); return; L_08B74428: rt.unsupported(0x08B74428u, 0x206F6F74u, "unknown not lowered yet"); return; L_08B74444: - ctx.execute_vfpu_compare3(99u, 111u, 114u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<99u, 111u, 114u, 1u, 6u>(); rt.unsupported(0x08B74448u, 0x6E697475u, "vfpu3 not lowered yet"); return; L_08B74458: rt.unsupported(0x08B74458u, 0x2061754Cu, "unknown not lowered yet"); return; diff --git a/profiles/vcs/generated/generated_unit_0221.cpp b/profiles/vcs/generated/generated_unit_0221.cpp index d6b5800..45fdacf 100644 --- a/profiles/vcs/generated/generated_unit_0221.cpp +++ b/profiles/vcs/generated/generated_unit_0221.cpp @@ -405,7 +405,7 @@ L_08B7A0AC: L_08B7A0B0: rt.unsupported(0x08B7A0B0u, 0x20746F6Eu, "unknown not lowered yet"); return; L_08B7A0C4: - ctx.execute_vfpu_compare3(101u, 114u, 114u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<101u, 114u, 114u, 1u, 6u>(); rt.unsupported(0x08B7A0C8u, 0x6E692072u, "vfpu3 not lowered yet"); return; L_08B7A0DC: rt.unsupported(0x08B7A0DCu, 0x74732043u, "unknown not lowered yet"); return; @@ -508,7 +508,7 @@ L_08B7A9C8: } goto L_08B7A9D0; L_08B7A9D0: - ctx.execute_vfpu_compare3(108u, 101u, 70u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<108u, 101u, 70u, 1u, 6u>(); rt.unsupported(0x08B7A9D4u, 0x72614772u, "unknown not lowered yet"); return; L_08B7A9DC: if (ctx.gpr[3] == ctx.gpr[20]) { diff --git a/profiles/vcs/generated/generated_unit_0222.cpp b/profiles/vcs/generated/generated_unit_0222.cpp index 00fa635..b5e7e44 100644 --- a/profiles/vcs/generated/generated_unit_0222.cpp +++ b/profiles/vcs/generated/generated_unit_0222.cpp @@ -714,11 +714,11 @@ L_08B7D614: L_08B7D640: rt.unsupported(0x08B7D640u, 0x736E6F63u, "unknown not lowered yet"); return; L_08B7D658: - ctx.execute_vfpu_vminmax(105u, 116u, 101u, 1u, false); + ctx.execute_vfpu_vminmax_ct<105u, 116u, 101u, 1u, false>(); rt.unsupported(0x08B7D65Cu, 0x6E692073u, "vfpu3 not lowered yet"); return; L_08B7D670: - ctx.execute_vfpu_vminmax(60u, 110u, 97u, 1u, false); - ctx.execute_vfpu_compare3(101u, 62u, 32u, 1u, 6u); + ctx.execute_vfpu_vminmax_ct<60u, 110u, 97u, 1u, false>(); + ctx.execute_vfpu_compare3_ct<101u, 62u, 32u, 1u, 6u>(); (void)(ctx.gpr[19] < static_cast(8306) ? 1u : 0u); rt.unsupported(0x08B7D67Cu, 0x20272E2Eu, "unknown not lowered yet"); return; L_08B7D68C: @@ -858,7 +858,7 @@ L_08B7E328: rt.unsupported(0x08B7E32Cu, 0x6E795320u, "vfpu3 not lowered yet"); return; L_08B7E354: ctx.gpr[1] = (ctx.gpr[19] ^ 30028u); - ctx.execute_vfpu_vminmax(32u, 77u, 101u, 1u, false); + ctx.execute_vfpu_vminmax_ct<32u, 77u, 101u, 1u, false>(); rt.unsupported(0x08B7E35Cu, 0x2079726Fu, "unknown not lowered yet"); return; L_08B7E374: ctx.gpr[1] = (ctx.gpr[19] ^ 30028u); @@ -895,7 +895,7 @@ L_08B7E458: rt.unsupported(0x08B7E458u, 0x43646574u, "unknown not lowered yet"); return; L_08B7E464: if (ctx.gpr[19] != ctx.gpr[20]) { - ctx.execute_vfpu_compare3(101u, 99u, 116u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<101u, 99u, 116u, 1u, 6u>(); (void)rt.invoke_chained_direct<&recomp_unit_0228_entry, 228u, 81u, 0x08B979B4u>(ctx, &aot_mem); return; } goto L_08B7E46C; @@ -1315,11 +1315,7 @@ L_08B7EA38: L_08B7EA64: rt.unsupported(0x08B7EA64u, 0x72646461u, "unknown not lowered yet"); return; L_08B7EA88: - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<111u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<97u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<76u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<76u, 111u, 97u, 1u, 2u>(); if (ctx.gpr[1] == 0u) { rt.unsupported(0x08B7EA90u, 0x43206465u, "unknown not lowered yet"); return; (void)rt.invoke_chained_direct<&recomp_unit_0228_entry, 228u, 85u, 0x08B97C24u>(ctx, &aot_mem); return; @@ -1337,11 +1333,7 @@ L_08B7EAB0: L_08B7EABC: rt.unsupported(0x08B7EABCu, 0x72617453u, "unknown not lowered yet"); return; L_08B7EAE0: - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<111u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<97u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<76u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<76u, 111u, 97u, 1u, 2u>(); rt.unsupported(0x08B7EAE4u, 0x20676E69u, "unknown not lowered yet"); return; L_08B7EAF4: rt.unsupported(0x08B7EAF4u, 0x44414F4Cu, "unsupported CFC1 control register"); return; @@ -1439,11 +1431,7 @@ L_08B7F3A0: L_08B7F3B0: rt.unsupported(0x08B7F3B0u, 0x69736F50u, "unknown not lowered yet"); return; L_08B7F3BC: - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<105u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<110u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<70u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<70u, 105u, 110u, 1u, 2u>(); rt.unsupported(0x08B7F3C0u, 0x4E68744Eu, "unknown not lowered yet"); return; L_08B7F3D8: rt.unsupported(0x08B7F3D8u, 0x4D746553u, "unknown not lowered yet"); return; @@ -1499,7 +1487,7 @@ L_08B7F688: ctx.execute_vfpu_vscl_ct<97u, 116u, 116u, 1u>(); rt.unsupported(0x08B7F68Cu, 0x2074706Du, "unknown not lowered yet"); return; L_08B7F6A4: - ctx.execute_vfpu_vhdp(112u, 101u, 114u, 1u); + ctx.execute_vfpu_vhdp_ct<112u, 101u, 114u, 1u>(); rt.unsupported(0x08B7F6A8u, 0x206D726Fu, "unknown not lowered yet"); return; L_08B7F6BC: ctx.execute_vfpu_vscl_ct<97u, 116u, 116u, 1u>(); @@ -1532,11 +1520,7 @@ L_08B7FA78: ctx.gpr[26] = (ctx.gpr[9] + static_cast(29477)); (void)(ctx.gpr[9] + static_cast(14948)); ctx.execute_vfpu_vscl_ct<115u, 32u, 110u, 1u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<114u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<32u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<97u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<97u, 114u, 32u, 1u, 0u>(); ctx.gpr[14] = (ctx.gpr[1] | ctx.gpr[7]); goto L_08B7FA8C; L_08B7FA8C: @@ -1545,7 +1529,7 @@ L_08B7FA8C: L_08B7FAA0: rt.unsupported(0x08B7FAA0u, 0x69626D61u, "unknown not lowered yet"); return; L_08B7FAD8: - ctx.execute_vfpu_vhdp(109u, 97u, 108u, 1u); + ctx.execute_vfpu_vhdp_ct<109u, 97u, 108u, 1u>(); ctx.execute_vfpu_vscl_ct<111u, 114u, 109u, 1u>(); rt.unsupported(0x08B7FAE0u, 0x756E2064u, "unknown not lowered yet"); return; L_08B7FAEC: @@ -1565,47 +1549,47 @@ L_08B7FC70: ctx.execute_vfpu_vcmp_ct<97u, 105u, 1u, 6u>(); rt.unsupported(0x08B7FC74u, 0x74206465u, "unknown not lowered yet"); return; L_08B7FC84: - ctx.execute_vfpu_compare3(77u, 101u, 109u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<77u, 101u, 109u, 1u, 6u>(); rt.unsupported(0x08B7FC88u, 0x61207972u, "vfpu0 not lowered yet"); return; L_08B7FCA0: - ctx.execute_vfpu_compare3(69u, 114u, 114u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<69u, 114u, 114u, 1u, 6u>(); rt.unsupported(0x08B7FCA4u, 0x61702072u, "vfpu0 not lowered yet"); return; L_08B7FCB8: ctx.execute_vfpu_vcmp_ct<97u, 105u, 1u, 6u>(); rt.unsupported(0x08B7FCBCu, 0x74206465u, "unknown not lowered yet"); return; L_08B7FCD4: - ctx.execute_vfpu_compare3(69u, 114u, 114u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<69u, 114u, 114u, 1u, 6u>(); ctx.execute_vfpu_vscl_ct<114u, 32u, 114u, 1u>(); rt.unsupported(0x08B7FCDCu, 0x6E696461u, "vfpu3 not lowered yet"); return; L_08B7FCF4: - ctx.execute_vfpu_compare3(69u, 114u, 114u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<69u, 114u, 114u, 1u, 6u>(); ctx.execute_vfpu_vscl_ct<114u, 32u, 114u, 1u>(); rt.unsupported(0x08B7FCFCu, 0x6E696461u, "vfpu3 not lowered yet"); return; L_08B7FD10: - ctx.execute_vfpu_compare3(69u, 114u, 114u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<69u, 114u, 114u, 1u, 6u>(); ctx.execute_vfpu_vscl_ct<114u, 58u, 32u, 1u>(); rt.unsupported(0x08B7FD18u, 0x7974706Du, "unknown not lowered yet"); return; L_08B7FD24: - ctx.execute_vfpu_compare3(69u, 114u, 114u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<69u, 114u, 114u, 1u, 6u>(); ctx.execute_vfpu_vscl_ct<114u, 32u, 114u, 1u>(); rt.unsupported(0x08B7FD2Cu, 0x6E696461u, "vfpu3 not lowered yet"); return; L_08B7FD3C: - ctx.execute_vfpu_compare3(69u, 114u, 114u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<69u, 114u, 114u, 1u, 6u>(); rt.unsupported(0x08B7FD40u, 0x61702072u, "vfpu0 not lowered yet"); return; L_08B7FD54: - ctx.execute_vfpu_compare3(69u, 114u, 114u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<69u, 114u, 114u, 1u, 6u>(); rt.unsupported(0x08B7FD58u, 0x61702072u, "vfpu0 not lowered yet"); return; L_08B7FD6C: - ctx.execute_vfpu_compare3(69u, 114u, 114u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<69u, 114u, 114u, 1u, 6u>(); rt.unsupported(0x08B7FD70u, 0x61702072u, "vfpu0 not lowered yet"); return; L_08B7FD88: - ctx.execute_vfpu_compare3(69u, 114u, 114u, 1u, 6u); - ctx.execute_vfpu_compare3(114u, 32u, 100u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<69u, 114u, 114u, 1u, 6u>(); + ctx.execute_vfpu_compare3_ct<114u, 32u, 100u, 1u, 6u>(); ctx.execute_vfpu_vscl_ct<99u, 117u, 109u, 1u>(); ctx.execute_vfpu_vscl_ct<110u, 116u, 32u, 1u>(); rt.unsupported(0x08B7FD98u, 0x7974706Du, "unknown not lowered yet"); return; L_08B7FDA0: - ctx.execute_vfpu_compare3(69u, 114u, 114u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<69u, 114u, 114u, 1u, 6u>(); rt.unsupported(0x08B7FDA4u, 0x756E2072u, "unknown not lowered yet"); return; L_08B7FDD8: rt.unsupported(0x08B7FDD8u, 0x49445541u, "cop2/vfpu not lowered yet"); return; diff --git a/profiles/vcs/generated/generated_unit_0223.cpp b/profiles/vcs/generated/generated_unit_0223.cpp index 7aa469f..b5fac0c 100644 --- a/profiles/vcs/generated/generated_unit_0223.cpp +++ b/profiles/vcs/generated/generated_unit_0223.cpp @@ -1117,7 +1117,7 @@ L_08B80F3C: L_08B80F60: ctx.execute_vfpu_vscl_ct<99u, 111u, 114u, 1u>(); ctx.gpr[2] = (ctx.gpr[19] ^ 26956u); - ctx.execute_vfpu_vminmax(67u, 111u, 109u, 1u, false); + ctx.execute_vfpu_vminmax_ct<67u, 111u, 109u, 1u, false>(); rt.unsupported(0x08B80F6Cu, 0x61746E65u, "vfpu0 not lowered yet"); return; L_08B80F80: ctx.gpr[12] = (ctx.gpr[26] + static_cast(20291)); @@ -1132,7 +1132,7 @@ L_08B81218: L_08B81498: rt.unsupported(0x08B81498u, 0x70736964u, "unknown not lowered yet"); return; L_08B814A4: - ctx.execute_vfpu_vminmax(95u, 115u, 101u, 1u, false); + ctx.execute_vfpu_vminmax_ct<95u, 115u, 101u, 1u, false>(); (void)(0u + 0u); // nop goto L_08B814B0; @@ -1254,20 +1254,16 @@ L_08B81BAC: // nop goto L_08B81BB8; L_08B81BB8: - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<111u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<97u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<76u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<76u, 111u, 97u, 1u, 2u>(); rt.unsupported(0x08B81BBCu, 0x746C754Du, "unknown not lowered yet"); return; L_08B81BD0: ctx.execute_vfpu_vscl_ct<87u, 105u, 114u, 1u>(); rt.unsupported(0x08B81BD4u, 0x7373656Cu, "unknown not lowered yet"); return; L_08B81BF8: - ctx.execute_vfpu_compare3(69u, 114u, 114u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<69u, 114u, 114u, 1u, 6u>(); rt.unsupported(0x08B81BFCu, 0x6E692072u, "vfpu3 not lowered yet"); return; L_08B81C1C: - ctx.execute_vfpu_compare3(69u, 114u, 114u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<69u, 114u, 114u, 1u, 6u>(); rt.unsupported(0x08B81C20u, 0x72632072u, "unknown not lowered yet"); return; L_08B81C44: ctx.execute_vfpu_vcmp_ct<97u, 105u, 1u, 6u>(); @@ -1299,15 +1295,15 @@ L_08B81D30: L_08B81D4C: rt.unsupported(0x08B81D4Cu, 0x20636F68u, "unknown not lowered yet"); return; L_08B81D68: - ctx.execute_vfpu_compare3(65u, 100u, 104u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<65u, 100u, 104u, 1u, 6u>(); rt.unsupported(0x08B81D6Cu, 0x6E6F4363u, "vfpu3 not lowered yet"); return; L_08B81D7C: rt.unsupported(0x08B81D7Cu, 0x4B656373u, "cop2/vfpu not lowered yet"); return; L_08B81D90: rt.unsupported(0x08B81D90u, 0x61662029u, "vfpu0 not lowered yet"); return; L_08B81D9C: - ctx.execute_vfpu_compare3(69u, 114u, 114u, 1u, 6u); - ctx.execute_vfpu_compare3(114u, 32u, 99u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<69u, 114u, 114u, 1u, 6u>(); + ctx.execute_vfpu_compare3_ct<114u, 32u, 99u, 1u, 6u>(); rt.unsupported(0x08B81DA4u, 0x63656E6Eu, "vfpu0 not lowered yet"); return; L_08B81DB8: ctx.execute_vfpu_vcmp_ct<97u, 105u, 1u, 6u>(); @@ -1316,22 +1312,22 @@ L_08B81DD4: ctx.execute_vfpu_vcmp_ct<97u, 105u, 1u, 6u>(); rt.unsupported(0x08B81DD8u, 0x74206465u, "unknown not lowered yet"); return; L_08B81DF8: - ctx.execute_vfpu_compare3(69u, 114u, 114u, 1u, 6u); - ctx.execute_vfpu_compare3(114u, 32u, 99u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<69u, 114u, 114u, 1u, 6u>(); + ctx.execute_vfpu_compare3_ct<114u, 32u, 99u, 1u, 6u>(); rt.unsupported(0x08B81E00u, 0x63656E6Eu, "vfpu0 not lowered yet"); return; L_08B81E14: - ctx.execute_vfpu_compare3(69u, 114u, 114u, 1u, 6u); - ctx.execute_vfpu_compare3(114u, 32u, 99u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<69u, 114u, 114u, 1u, 6u>(); + ctx.execute_vfpu_compare3_ct<114u, 32u, 99u, 1u, 6u>(); rt.unsupported(0x08B81E1Cu, 0x63656E6Eu, "vfpu0 not lowered yet"); return; L_08B81E34: - ctx.execute_vfpu_compare3(69u, 114u, 114u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<69u, 114u, 114u, 1u, 6u>(); rt.unsupported(0x08B81E38u, 0x63732072u, "vfpu0 not lowered yet"); return; L_08B81E54: ctx.execute_vfpu_vscl_ct<84u, 105u, 109u, 1u>(); rt.unsupported(0x08B81E58u, 0x756F2064u, "unknown not lowered yet"); return; L_08B81E78: - ctx.execute_vfpu_compare3(69u, 114u, 114u, 1u, 6u); - ctx.execute_vfpu_compare3(114u, 32u, 106u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<69u, 114u, 114u, 1u, 6u>(); + ctx.execute_vfpu_compare3_ct<114u, 32u, 106u, 1u, 6u>(); rt.unsupported(0x08B81E80u, 0x6E696E69u, "vfpu3 not lowered yet"); return; L_08B81E94: ctx.execute_vfpu_vcmp_ct<97u, 105u, 1u, 6u>(); @@ -1340,7 +1336,7 @@ L_08B81F00: rt.unsupported(0x08B81F04u, 0x08A575ACu, "control flow in delay slot"); return; L_08B81F20: if (ctx.gpr[27] != ctx.gpr[19]) { - ctx.execute_vfpu_compare3(97u, 121u, 112u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<97u, 121u, 112u, 1u, 6u>(); (void)rt.invoke_chained_direct<&recomp_unit_0229_entry, 229u, 56u, 0x08B9A444u>(ctx, &aot_mem); return; } goto L_08B81F28; @@ -1348,7 +1344,7 @@ L_08B81F28: rt.unsupported(0x08B81F28u, 0x42746E69u, "unknown not lowered yet"); return; L_08B81F34: ctx.execute_vfpu_vscl_ct<82u, 97u, 99u, 1u>(); - ctx.execute_vfpu_compare3(65u, 114u, 114u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<65u, 114u, 114u, 1u, 6u>(); rt.unsupported(0x08B81F3Cu, 0x73695677u, "unknown not lowered yet"); return; L_08B81F48: rt.unsupported(0x08B81F48u, 0x61656C43u, "vfpu0 not lowered yet"); return; @@ -1380,37 +1376,25 @@ L_08B824FC: ctx.execute_vfpu_vcmp_ct<97u, 98u, 1u, 4u>(); rt.unsupported(0x08B82500u, 0x6E692065u, "vfpu3 not lowered yet"); return; L_08B82510: - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<101u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<97u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<104u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<104u, 101u, 97u, 1u, 2u>(); rt.unsupported(0x08B82514u, 0x6867696Cu, "unknown not lowered yet"); return; L_08B82560: - ctx.execute_vfpu_compare3(109u, 101u, 109u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<109u, 101u, 109u, 1u, 6u>(); rt.unsupported(0x08B82564u, 0x61207972u, "vfpu0 not lowered yet"); return; L_08B82588: rt.unsupported(0x08B8258Cu, 0x08AA0908u, "control flow in delay slot"); return; L_08B825D0: ctx.execute_vfpu_vcmp_ct<97u, 98u, 1u, 4u>(); - ctx.execute_vfpu_compare3(101u, 32u, 99u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<101u, 32u, 99u, 1u, 6u>(); rt.unsupported(0x08B825D8u, 0x6961746Eu, "unknown not lowered yet"); return; L_08B825EC: rt.unsupported(0x08B825ECu, 0x61766E69u, "vfpu0 not lowered yet"); return; L_08B82610: rt.unsupported(0x08B82610u, 0x706D7562u, "unknown not lowered yet"); return; L_08B82624: - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<105u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<110u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<119u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<119u, 105u, 110u, 1u, 2u>(); ctx.execute_vfpu_vscl_ct<115u, 99u, 114u, 1u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<110u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<95u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<101u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<101u, 110u, 95u, 1u, 2u>(); rt.unsupported(0x08B82630u, 0x796D6D75u, "unknown not lowered yet"); return; L_08B82638: rt.unsupported(0x08B82638u, 0x74616F62u, "unknown not lowered yet"); return; @@ -1429,10 +1413,10 @@ L_08B826B8: L_08B826CC: rt.unsupported(0x08B826CCu, 0x2078696Du, "unknown not lowered yet"); return; L_08B826E0: - ctx.execute_vfpu_vminmax(67u, 83u, 105u, 1u, false); + ctx.execute_vfpu_vminmax_ct<67u, 83u, 105u, 1u, false>(); rt.unsupported(0x08B826E4u, 0x4D656C70u, "unknown not lowered yet"); return; L_08B826F8: - ctx.execute_vfpu_compare3(100u, 101u, 99u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<100u, 101u, 99u, 1u, 6u>(); rt.unsupported(0x08B826FCu, 0x735F6564u, "unknown not lowered yet"); return; L_08B827D0: ctx.gpr[16] = (ctx.gpr[3] & 26946u); @@ -1449,11 +1433,11 @@ L_08B829A0: L_08B829B8: rt.unsupported(0x08B829B8u, 0x61766E69u, "vfpu0 not lowered yet"); return; L_08B829D0: - ctx.execute_vfpu_vhdp(109u, 97u, 108u, 1u); + ctx.execute_vfpu_vhdp_ct<109u, 97u, 108u, 1u>(); ctx.execute_vfpu_vscl_ct<111u, 114u, 109u, 1u>(); rt.unsupported(0x08B829D8u, 0x61702064u, "vfpu0 not lowered yet"); return; L_08B829F4: - ctx.execute_vfpu_vhdp(109u, 97u, 108u, 1u); + ctx.execute_vfpu_vhdp_ct<109u, 97u, 108u, 1u>(); ctx.execute_vfpu_vscl_ct<111u, 114u, 109u, 1u>(); rt.unsupported(0x08B829FCu, 0x61702064u, "vfpu0 not lowered yet"); return; L_08B82A14: @@ -1471,7 +1455,7 @@ L_08B82A9C: L_08B82AB8: rt.unsupported(0x08B82AB8u, 0x7373696Du, "unknown not lowered yet"); return; L_08B82ADC: - ctx.execute_vfpu_compare3(111u, 98u, 115u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<111u, 98u, 115u, 1u, 6u>(); ctx.execute_vfpu_vscl_ct<108u, 101u, 116u, 1u>(); rt.unsupported(0x08B82AE4u, 0x74706F20u, "unknown not lowered yet"); return; L_08B82B00: @@ -1586,11 +1570,7 @@ L_08B83200: L_08B83208: rt.unsupported(0x08B83208u, 0x4F6C6C41u, "unknown not lowered yet"); return; L_08B83214: - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<115u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<65u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<73u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<73u, 115u, 65u, 1u, 2u>(); rt.unsupported(0x08B83218u, 0x43636F68u, "unknown not lowered yet"); return; L_08B83228: rt.unsupported(0x08B83228u, 0x496D754Eu, "cop2/vfpu not lowered yet"); return; @@ -1599,27 +1579,15 @@ L_08B83240: L_08B83254: rt.unsupported(0x08B83254u, 0x776F6853u, "unknown not lowered yet"); return; L_08B83268: - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<105u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<110u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<70u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<70u, 105u, 110u, 1u, 2u>(); rt.unsupported(0x08B8326Cu, 0x756F7247u, "unknown not lowered yet"); return; L_08B83280: - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<101u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<110u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<82u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<82u, 101u, 110u, 1u, 2u>(); rt.unsupported(0x08B83284u, 0x61507265u, "vfpu0 not lowered yet"); return; L_08B83290: rt.unsupported(0x08B83290u, 0x7574536Eu, "unknown not lowered yet"); return; L_08B83298: - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<101u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<110u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<82u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<82u, 101u, 110u, 1u, 2u>(); rt.unsupported(0x08B8329Cu, 0x61507265u, "vfpu0 not lowered yet"); return; L_08B832A8: rt.unsupported(0x08B832A8u, 0x7574536Eu, "unknown not lowered yet"); return; @@ -1643,7 +1611,7 @@ L_08B832F0: L_08B83308: rt.unsupported(0x08B83308u, 0x00007372u, "special? not lowered yet"); return; L_08B8330C: - ctx.execute_vfpu_vminmax(84u, 101u, 97u, 1u, false); + ctx.execute_vfpu_vminmax_ct<84u, 101u, 97u, 1u, false>(); ctx.execute_vfpu_vscl_ct<71u, 97u, 109u, 1u>(); rt.unsupported(0x08B83314u, 0x72657645u, "unknown not lowered yet"); return; L_08B83320: @@ -1724,11 +1692,7 @@ L_08B83690: L_08B83698: rt.unsupported(0x08B83698u, 0x454E4543u, "cop1? not lowered yet"); return; L_08B836A0: - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<111u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<97u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<76u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<76u, 111u, 97u, 1u, 2u>(); rt.unsupported(0x08B836A4u, 0x20676E69u, "unknown not lowered yet"); return; L_08B836B4: rt.unsupported(0x08B836B4u, 0x75746553u, "unknown not lowered yet"); return; diff --git a/profiles/vcs/generated/generated_unit_0224.cpp b/profiles/vcs/generated/generated_unit_0224.cpp index 1e629e0..1edd31b 100644 --- a/profiles/vcs/generated/generated_unit_0224.cpp +++ b/profiles/vcs/generated/generated_unit_0224.cpp @@ -362,7 +362,7 @@ L_08B84000: L_08B84020: ctx.execute_vfpu_vscl_ct<99u, 111u, 114u, 1u>(); ctx.gpr[2] = (ctx.gpr[19] ^ 26956u); - ctx.execute_vfpu_vminmax(67u, 111u, 109u, 1u, false); + ctx.execute_vfpu_vminmax_ct<67u, 111u, 109u, 1u, false>(); rt.unsupported(0x08B8402Cu, 0x61746E65u, "vfpu0 not lowered yet"); return; L_08B84040: ctx.gpr[20] = (ctx.gpr[26] + static_cast(17989)); @@ -461,7 +461,7 @@ L_08B841D4: L_08B841E0: rt.unsupported(0x08B841E0u, 0x4E414843u, "unknown not lowered yet"); return; L_08B841F4: - ctx.execute_vfpu_compare3(67u, 111u, 108u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<67u, 111u, 108u, 1u, 6u>(); rt.unsupported(0x08B841F8u, 0x20737275u, "unknown not lowered yet"); return; L_08B84214: rt.unsupported(0x08B84214u, 0x4E414843u, "unknown not lowered yet"); return; @@ -758,14 +758,14 @@ L_08B875D0: L_08B875E4: rt.unsupported(0x08B875E4u, 0x436D754Eu, "unknown not lowered yet"); return; L_08B875F8: - ctx.execute_vfpu_vhdp(67u, 111u, 110u, 1u); + ctx.execute_vfpu_vhdp_ct<67u, 111u, 110u, 1u>(); rt.unsupported(0x08B875FCu, 0x72416769u, "unknown not lowered yet"); return; L_08B8760C: - ctx.execute_vfpu_vhdp(67u, 111u, 110u, 1u); + ctx.execute_vfpu_vhdp_ct<67u, 111u, 110u, 1u>(); rt.unsupported(0x08B87610u, 0x72416769u, "unknown not lowered yet"); return; L_08B876F8: ctx.execute_vfpu_vscl_ct<119u, 97u, 116u, 1u>(); - ctx.execute_vfpu_vhdp(114u, 114u, 101u, 1u); + ctx.execute_vfpu_vhdp_ct<114u, 114u, 101u, 1u>(); rt.unsupported(0x08B87700u, 0x7463656Cu, "unknown not lowered yet"); return; L_08B87720: rt.unsupported(0x08B87720u, 0x0000594Eu, "special? not lowered yet"); return; @@ -851,11 +851,7 @@ L_08B87E40: } goto L_08B87E48; L_08B87E48: - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<82u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<97u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<114u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<114u, 82u, 97u, 1u, 2u>(); ctx.execute_vfpu_vcmp_ct<114u, 66u, 1u, 1u>(); rt.unsupported(0x08B87E50u, 0x68537069u, "unknown not lowered yet"); return; L_08B87E5C: @@ -868,11 +864,7 @@ L_08B87E60: } goto L_08B87E68; L_08B87E68: - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<82u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<97u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<114u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<114u, 82u, 97u, 1u, 2u>(); goto L_08B87E6C; L_08B87E6C: rt.unsupported(0x08B87E6Cu, 0x63497261u, "vfpu0 not lowered yet"); return; @@ -904,7 +896,7 @@ L_08B87EF4: ctx.gpr[12] = (static_cast(0u) > static_cast(0u) ? 0u : 0u); goto L_08B87EF8; L_08B87EF8: - ctx.execute_vfpu_compare3(73u, 115u, 76u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<73u, 115u, 76u, 1u, 6u>(); if (ctx.gpr[3] == ctx.gpr[12]) { ctx.execute_vfpu_vscl_ct<108u, 97u, 121u, 1u>(); (void)rt.invoke_chained_direct<&recomp_unit_0231_entry, 231u, 5u, 0x08BA048Cu>(ctx, &aot_mem); return; @@ -928,10 +920,10 @@ L_08B87F48: ctx.execute_vfpu_vscl_ct<71u, 105u, 118u, 1u>(); rt.unsupported(0x08B87F4Cu, 0x61636F4Cu, "vfpu0 not lowered yet"); return; L_08B87F68: - ctx.execute_vfpu_compare3(82u, 101u, 109u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<82u, 101u, 109u, 1u, 6u>(); goto L_08B87F6C; L_08B87F6C: - ctx.execute_vfpu_compare3(118u, 101u, 76u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<118u, 101u, 76u, 1u, 6u>(); if (ctx.gpr[3] == ctx.gpr[12]) { ctx.execute_vfpu_vscl_ct<108u, 97u, 121u, 1u>(); (void)rt.invoke_chained_direct<&recomp_unit_0231_entry, 231u, 6u, 0x08BA0500u>(ctx, &aot_mem); return; @@ -955,7 +947,7 @@ L_08B87FA4: L_08B87FB0: rt.unsupported(0x08B87FB0u, 0x6E696F50u, "vfpu3 not lowered yet"); return; L_08B87FB8: - ctx.execute_vfpu_compare3(73u, 115u, 76u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<73u, 115u, 76u, 1u, 6u>(); if (ctx.gpr[3] == ctx.gpr[12]) { ctx.execute_vfpu_vscl_ct<108u, 97u, 121u, 1u>(); (void)rt.invoke_chained_direct<&recomp_unit_0231_entry, 231u, 7u, 0x08BA054Cu>(ctx, &aot_mem); return; diff --git a/profiles/vcs/generated/generated_unit_0225.cpp b/profiles/vcs/generated/generated_unit_0225.cpp index 1c00085..5898fe9 100644 --- a/profiles/vcs/generated/generated_unit_0225.cpp +++ b/profiles/vcs/generated/generated_unit_0225.cpp @@ -533,14 +533,14 @@ L_08B88008: L_08B88018: rt.unsupported(0x08B88018u, 0x6E6F4372u, "vfpu3 not lowered yet"); return; L_08B88024: - ctx.execute_vfpu_compare3(73u, 115u, 76u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<73u, 115u, 76u, 1u, 6u>(); if (ctx.gpr[3] == ctx.gpr[12]) { ctx.execute_vfpu_vscl_ct<108u, 97u, 121u, 1u>(); (void)rt.invoke_chained_direct<&recomp_unit_0231_entry, 231u, 11u, 0x08BA05B8u>(ctx, &aot_mem); return; } goto L_08B88030; L_08B88030: - ctx.execute_vfpu_compare3(114u, 68u, 114u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<114u, 68u, 114u, 1u, 6u>(); rt.unsupported(0x08B88034u, 0x6E696E77u, "vfpu3 not lowered yet"); return; L_08B8803C: if (ctx.gpr[3] == ctx.gpr[16]) { @@ -556,7 +556,7 @@ L_08B88044: } goto L_08B88050; L_08B88050: - ctx.execute_vfpu_compare3(67u, 111u, 108u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<67u, 111u, 108u, 1u, 6u>(); rt.unsupported(0x08B88054u, 0x00007275u, "special? not lowered yet"); return; L_08B88058: ctx.execute_vfpu_vcmp_ct<115u, 80u, 1u, 9u>(); @@ -611,8 +611,8 @@ L_08B883FC: L_08B88400: rt.unsupported(0x08B88400u, 0x4C646441u, "unknown not lowered yet"); return; L_08B88418: - ctx.execute_vfpu_compare3(82u, 101u, 109u, 1u, 6u); - ctx.execute_vfpu_compare3(118u, 101u, 76u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<82u, 101u, 109u, 1u, 6u>(); + ctx.execute_vfpu_compare3_ct<118u, 101u, 76u, 1u, 6u>(); if (ctx.gpr[19] == ctx.gpr[12]) { rt.unsupported(0x08B88424u, 0x72616461u, "unknown not lowered yet"); return; (void)rt.invoke_chained_direct<&recomp_unit_0231_entry, 231u, 17u, 0x08BA09B0u>(ctx, &aot_mem); return; @@ -623,7 +623,7 @@ L_08B88428: L_08B88430: rt.unsupported(0x08B88430u, 0x42746553u, "unknown not lowered yet"); return; L_08B8843C: - ctx.execute_vfpu_compare3(108u, 101u, 70u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<108u, 101u, 70u, 1u, 6u>(); rt.unsupported(0x08B88440u, 0x616C5072u, "vfpu0 not lowered yet"); return; L_08B8844C: // nop @@ -644,7 +644,7 @@ L_08B88464: goto L_08B8846C; L_08B8846C: ctx.execute_vfpu_vscl_ct<84u, 111u, 84u, 1u>(); - ctx.execute_vfpu_compare3(97u, 109u, 67u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<97u, 109u, 67u, 1u, 6u>(); rt.unsupported(0x08B88474u, 0x72756F6Cu, "unknown not lowered yet"); return; L_08B8847C: rt.unsupported(0x08B88480u, 0x08B24924u, "control flow in delay slot"); return; @@ -704,12 +704,12 @@ L_08B88790: ctx.execute_vfpu_vscl_ct<71u, 101u, 110u, 1u>(); ctx.execute_vfpu_vscl_ct<114u, 97u, 116u, 1u>(); if (ctx.gpr[19] == ctx.gpr[5]) { - ctx.execute_vfpu_compare3(97u, 110u, 100u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<97u, 110u, 100u, 1u, 6u>(); (void)rt.invoke_chained_direct<&recomp_unit_0232_entry, 232u, 4u, 0x08BA40D8u>(ctx, &aot_mem); return; } goto L_08B887A0; L_08B8879C: - ctx.execute_vfpu_compare3(97u, 110u, 100u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<97u, 110u, 100u, 1u, 6u>(); goto L_08B887A0; L_08B887A0: rt.unsupported(0x08B887A0u, 0x7261436Du, "unknown not lowered yet"); return; @@ -763,7 +763,7 @@ L_08B88980: L_08B8898C: rt.unsupported(0x08B8898Cu, 0x69686556u, "unknown not lowered yet"); return; L_08B88998: - ctx.execute_vfpu_compare3(114u, 68u, 111u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<114u, 68u, 111u, 1u, 6u>(); rt.unsupported(0x08B8899Cu, 0x636F4C72u, "vfpu0 not lowered yet"); return; L_08B889A4: rt.unsupported(0x08B889A4u, 0x69686556u, "unknown not lowered yet"); return; @@ -780,17 +780,13 @@ L_08B889D0: } goto L_08B889D8; L_08B889D8: - ctx.execute_vfpu_compare3(108u, 101u, 67u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<108u, 101u, 67u, 1u, 6u>(); rt.unsupported(0x08B889DCu, 0x72756F6Cu, "unknown not lowered yet"); return; L_08B889E4: ctx.execute_vfpu_vscl_ct<73u, 115u, 86u, 1u>(); ctx.execute_vfpu_vcmp_ct<105u, 99u, 1u, 8u>(); ctx.execute_vfpu_vscl_ct<101u, 87u, 114u, 1u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<107u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<101u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<99u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<99u, 107u, 101u, 1u, 2u>(); // nop goto L_08B889F8; L_08B889F8: @@ -868,11 +864,11 @@ L_08B88DA0: rt.unsupported(0x08B88DA4u, 0x08B56BE8u, "control flow in delay slot"); return; L_08B88F48: ctx.execute_vfpu_vscl_ct<97u, 115u, 115u, 1u>(); - ctx.execute_vfpu_compare3(114u, 116u, 105u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<114u, 116u, 105u, 1u, 6u>(); ctx.gpr[2] = (ctx.gpr[9] + static_cast(8302)); - ctx.execute_vfpu_vhdp(115u, 34u, 32u, 1u); + ctx.execute_vfpu_vhdp_ct<115u, 34u, 32u, 1u>(); ctx.execute_vfpu_vscl_ct<97u, 105u, 108u, 1u>(); - ctx.execute_vfpu_vhdp(100u, 58u, 32u, 1u); + ctx.execute_vfpu_vhdp_ct<100u, 58u, 32u, 1u>(); rt.unsupported(0x08B88F60u, 0x20656C69u, "unknown not lowered yet"); return; L_08B88F78: rt.unsupported(0x08B88F78u, 0x4A532D43u, "cop2/vfpu not lowered yet"); return; @@ -899,7 +895,7 @@ L_08B89034: L_08B8903C: rt.unsupported(0x08B8903Cu, 0x63657845u, "vfpu0 not lowered yet"); return; L_08B89048: - ctx.execute_vfpu_compare3(101u, 114u, 114u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<101u, 114u, 114u, 1u, 6u>(); rt.unsupported(0x08B8904Cu, 0x00000072u, "special? not lowered yet"); return; L_08B89050: rt.unsupported(0x08B89050u, 0x20646142u, "unknown not lowered yet"); return; @@ -909,12 +905,12 @@ L_08B8905C: L_08B89060: rt.unsupported(0x08B89060u, 0x63206F4Eu, "vfpu0 not lowered yet"); return; L_08B8906C: - ctx.execute_vfpu_vminmax(78u, 111u, 32u, 1u, false); + ctx.execute_vfpu_vminmax_ct<78u, 111u, 32u, 1u, false>(); rt.unsupported(0x08B89070u, 0x2065726Fu, "unknown not lowered yet"); return; L_08B89080: rt.unsupported(0x08B89080u, 0x20746F4Eu, "unknown not lowered yet"); return; L_08B89094: - ctx.execute_vfpu_vminmax(80u, 101u, 114u, 1u, false); + ctx.execute_vfpu_vminmax_ct<80u, 101u, 114u, 1u, false>(); rt.unsupported(0x08B89098u, 0x69737369u, "unknown not lowered yet"); return; L_08B890A8: rt.unsupported(0x08B890A8u, 0x20646142u, "unknown not lowered yet"); return; @@ -945,7 +941,7 @@ L_08B89194: rt.unsupported(0x08B89194u, 0x74786554u, "unknown not lowered yet"); return; L_08B891A4: ctx.execute_vfpu_vscl_ct<70u, 105u, 108u, 1u>(); - ctx.execute_vfpu_compare3(32u, 116u, 111u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<32u, 116u, 111u, 1u, 6u>(); rt.unsupported(0x08B891ACu, 0x72616C20u, "unknown not lowered yet"); return; L_08B891B4: rt.unsupported(0x08B891B4u, 0x73206F4Eu, "unknown not lowered yet"); return; @@ -953,11 +949,7 @@ L_08B891CC: ctx.execute_vfpu_vscl_ct<73u, 108u, 108u, 1u>(); rt.unsupported(0x08B891D0u, 0x206C6167u, "unknown not lowered yet"); return; L_08B891DC: - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<101u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<97u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<82u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<82u, 101u, 97u, 1u, 2u>(); ctx.execute_vfpu_vcmp_ct<111u, 110u, 1u, 13u>(); rt.unsupported(0x08B891E4u, 0x69662079u, "unknown not lowered yet"); return; L_08B891F4: @@ -969,16 +961,12 @@ L_08B89210: L_08B89220: rt.unsupported(0x08B89220u, 0x75736552u, "unknown not lowered yet"); return; L_08B89234: - ctx.execute_vfpu_vminmax(78u, 111u, 32u, 1u, false); + ctx.execute_vfpu_vminmax_ct<78u, 111u, 32u, 1u, false>(); rt.unsupported(0x08B89238u, 0x61737365u, "vfpu0 not lowered yet"); return; L_08B89250: rt.unsupported(0x08B89250u, 0x6E656449u, "vfpu3 not lowered yet"); return; L_08B89264: - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<101u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<97u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<68u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<68u, 101u, 97u, 1u, 2u>(); rt.unsupported(0x08B89268u, 0x6B636F6Cu, "unknown not lowered yet"); return; L_08B89270: ctx.execute_vfpu_vcmp_ct<111u, 32u, 1u, 14u>(); @@ -995,7 +983,7 @@ L_08B892B4: L_08B892D4: rt.unsupported(0x08B892D4u, 0x70206F4Eu, "unknown not lowered yet"); return; L_08B892E0: - ctx.execute_vfpu_compare3(82u, 101u, 115u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<82u, 101u, 115u, 1u, 6u>(); ctx.execute_vfpu_vscl_ct<117u, 114u, 99u, 1u>(); rt.unsupported(0x08B892E8u, 0x20736920u, "unknown not lowered yet"); return; L_08B892F4: @@ -1004,10 +992,10 @@ L_08B8930C: ctx.execute_vfpu_vscl_ct<65u, 100u, 118u, 1u>(); rt.unsupported(0x08B89310u, 0x73697472u, "unknown not lowered yet"); return; L_08B8931C: - ctx.execute_vfpu_compare3(83u, 114u, 109u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<83u, 114u, 109u, 1u, 6u>(); rt.unsupported(0x08B89320u, 0x20746E75u, "unknown not lowered yet"); return; L_08B8932C: - ctx.execute_vfpu_vminmax(67u, 111u, 109u, 1u, false); + ctx.execute_vfpu_vminmax_ct<67u, 111u, 109u, 1u, false>(); rt.unsupported(0x08B89330u, 0x63696E75u, "vfpu0 not lowered yet"); return; L_08B89340: rt.unsupported(0x08B89340u, 0x746F7250u, "unknown not lowered yet"); return; @@ -1030,7 +1018,7 @@ L_08B89420: L_08B89448: rt.unsupported(0x08B89448u, 0x636E7546u, "vfpu0 not lowered yet"); return; L_08B89464: - ctx.execute_vfpu_vminmax(78u, 111u, 32u, 1u, false); + ctx.execute_vfpu_vminmax_ct<78u, 111u, 32u, 1u, false>(); rt.unsupported(0x08B89468u, 0x2065726Fu, "unknown not lowered yet"); return; L_08B89474: ctx.execute_vfpu_vscl_ct<68u, 105u, 114u, 1u>(); @@ -1072,7 +1060,7 @@ L_08B897B4: L_08B89800: rt.unsupported(0x08B89804u, 0x08B58D9Cu, "control flow in delay slot"); return; L_08B89858: - ctx.execute_vfpu_vhdp(45u, 73u, 110u, 1u); + ctx.execute_vfpu_vhdp_ct<45u, 73u, 110u, 1u>(); // nop jump_target = ctx.gpr[3]; ctx.gpr[13] = (0x08B89868u); @@ -1120,7 +1108,7 @@ L_08B89CB8: ctx.gpr[11] = (static_cast(ctx.gpr[25]) < 17162 ? 1u : 0u); rt.unsupported(0x08B89CBCu, 0x6E757220u, "vfpu3 not lowered yet"); return; L_08B89CCC: - ctx.execute_vfpu_vminmax(116u, 101u, 114u, 1u, false); + ctx.execute_vfpu_vminmax_ct<116u, 101u, 114u, 1u, false>(); rt.unsupported(0x08B89CD0u, 0x74616E69u, "unknown not lowered yet"); return; L_08B89D04: rt.unsupported(0x08B89D04u, 0x75746572u, "unknown not lowered yet"); return; diff --git a/profiles/vcs/generated/generated_unit_0231.cpp b/profiles/vcs/generated/generated_unit_0231.cpp index 2ecbb1b..bad299c 100644 --- a/profiles/vcs/generated/generated_unit_0231.cpp +++ b/profiles/vcs/generated/generated_unit_0231.cpp @@ -1177,16 +1177,16 @@ L_08BA10A8: // nop goto L_08BA10CC; L_08BA10CC: - ctx.execute_vfpu_compare3(99u, 111u, 114u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<99u, 111u, 114u, 1u, 6u>(); rt.unsupported(0x08BA10D0u, 0x7473616Eu, "unknown not lowered yet"); return; L_08BA110C: - ctx.execute_vfpu_compare3(99u, 111u, 114u, 1u, 6u); - ctx.execute_vfpu_compare3(110u, 97u, 109u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<99u, 111u, 114u, 1u, 6u>(); + ctx.execute_vfpu_compare3_ct<110u, 97u, 109u, 1u, 6u>(); rt.unsupported(0x08BA1114u, 0x00006E6Fu, "special? not lowered yet"); return; L_08BA1144: // nop // nop - ctx.execute_vfpu_compare3(99u, 111u, 114u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<99u, 111u, 114u, 1u, 6u>(); ctx.execute_vfpu_vscl_ct<110u, 97u, 104u, 1u>(); rt.unsupported(0x08BA1154u, 0x696C6461u, "unknown not lowered yet"); return; L_08BA115C: @@ -1199,7 +1199,7 @@ L_08BA117C: // nop // nop // nop - ctx.execute_vfpu_compare3(99u, 111u, 114u, 1u, 6u); + ctx.execute_vfpu_compare3_ct<99u, 111u, 114u, 1u, 6u>(); rt.unsupported(0x08BA1190u, 0x6963616Eu, "unknown not lowered yet"); return; L_08BA11EC: // PSP CACHE is a no-op in coherent host memory. diff --git a/profiles/vcs/generated/generated_unit_0232.cpp b/profiles/vcs/generated/generated_unit_0232.cpp index 4945c26..c1265b0 100644 --- a/profiles/vcs/generated/generated_unit_0232.cpp +++ b/profiles/vcs/generated/generated_unit_0232.cpp @@ -958,11 +958,7 @@ L_08BA5394: goto L_08BA539C; } L_08BA539C: - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<5u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<5u, 0u, 0u, 1u, 0u>(); rt.unsupported(0x08BA53A0u, 0xC0000001u, "unknown not lowered yet"); return; L_08BA53A8: // nop diff --git a/profiles/vcs/host/vcs_profile.cpp b/profiles/vcs/host/vcs_profile.cpp index 76c459a..f2e85c6 100644 --- a/profiles/vcs/host/vcs_profile.cpp +++ b/profiles/vcs/host/vcs_profile.cpp @@ -198,6 +198,13 @@ struct VirtualDiscStream { bool position_valid{}; }; +struct AtracSourceIdentity { + std::uint64_t file_size{}; + std::array prefix{}; + std::size_t prefix_size{}; + bool valid{}; +}; + struct FileTable { std::int32_t next_fd{3}; std::uint32_t next_virtual_sector{0x00010000u}; @@ -211,6 +218,12 @@ struct FileTable { // AT3/AA3/OMA can be associated with its host source without reopening and // rescanning candidate files on the audio thread. std::unordered_map recent_atrac_reads; + // V8.6 radio identity guard: cache an exact prefix for each host ATRAC source. + // The producer hint above is intentionally not trusted on its own because VCS + // reuses staging buffers and may overwrite them between sceIoRead and + // sceAtracSetHalfwayBufferAndGetID. This cache is host-only and need not be + // serialized in diagnostic checkpoints; it is rebuilt lazily after restore. + std::unordered_map atrac_source_identities; std::unordered_set synthetic_empty_files; std::unordered_map directories; std::unordered_map virtual_disc_handles; @@ -562,6 +575,44 @@ bool is_atrac_source_path(const std::filesystem::path &path) { return extension == ".AT3" || extension == ".AA3" || extension == ".OMA"; } +const AtracSourceIdentity *atrac_source_identity(const std::filesystem::path &path) { + if (path.empty() || !is_atrac_source_path(path)) return nullptr; + const std::string key = normalized_native_path(path); + if (const auto found = file_table.atrac_source_identities.find(key); + found != file_table.atrac_source_identities.end()) { + return found->second.valid ? &found->second : nullptr; + } + + AtracSourceIdentity identity{}; + std::error_code error; + identity.file_size = std::filesystem::file_size(path, error); + if (!error) { + std::ifstream input(path, std::ios::binary); + if (input) { + input.read(reinterpret_cast(identity.prefix.data()), + static_cast(identity.prefix.size())); + const auto count = input.gcount(); + if (count > 0) { + identity.prefix_size = static_cast(count); + identity.valid = true; + } + } + } + const auto [it, inserted] = file_table.atrac_source_identities.emplace(key, std::move(identity)); + (void)inserted; + return it->second.valid ? &it->second : nullptr; +} + +[[nodiscard]] bool atrac_source_matches_header(const std::filesystem::path &path, + std::span header, + const ParsedAtracHeader &parsed) { + const AtracSourceIdentity *identity = atrac_source_identity(path); + if (identity == nullptr || identity->file_size != parsed.file_size) return false; + const std::size_t compare_size = std::min(header.size(), identity->prefix.size()); + if (compare_size == 0u || identity->prefix_size < compare_size) return false; + return std::equal(identity->prefix.begin(), identity->prefix.begin() + compare_size, header.begin()); +} + const VirtualDiscFile *virtual_disc_file_at_offset(std::uint64_t absolute) { const std::uint64_t sector64 = absolute / 2048u; if (sector64 > 0xFFFFFFFFull) return nullptr; @@ -927,28 +978,37 @@ std::filesystem::path identify_atrac_source(std::uint32_t guest_buffer, const ParsedAtracHeader &parsed) { const auto started = std::chrono::steady_clock::now(); bool direct_buffer = false; - std::uint32_t fallback_reads = 0u; + bool direct_rejected = false; + std::uint32_t fallback_candidates = 0u; std::filesystem::path matched; + // A recent sceIoRead is only a producer hint. VCS reuses radio staging + // buffers and can copy/overwrite them before ATRAC setup, so accepting the + // old path solely because the guest address matches can attach CITY.AT3's + // timeline to EMOTION.AT3 (or another station). Validate the *current* + // guest header against a cached exact host-file prefix before trusting it. if (const auto direct = file_table.recent_atrac_reads.find(guest_buffer); direct != file_table.recent_atrac_reads.end()) { - matched = direct->second; - direct_buffer = true; + const std::filesystem::path candidate = direct->second; file_table.recent_atrac_reads.erase(direct); + if (atrac_source_matches_header(candidate, header, parsed)) { + matched = candidate; + direct_buffer = true; + } else { + direct_rejected = true; + } } - const std::size_t compare_size = std::min(header.size(), 256u); + // The same identity cache also makes the correctness fallback cheap after + // the first lookup: no repeated open/read of every radio file on later + // station switches. Exact prefix comparison keeps same-sized RIFF files + // distinct without hard-coding any VCS station index. if (matched.empty()) { for (const auto &[key, file] : file_table.virtual_files_by_path) { (void)key; if (file.size != parsed.file_size || !is_atrac_source_path(file.native_path)) continue; - ++fallback_reads; - std::vector candidate(compare_size); - std::ifstream input(file.native_path, std::ios::binary); - if (!input) continue; - input.read(reinterpret_cast(candidate.data()), static_cast(candidate.size())); - if (input.gcount() == static_cast(candidate.size()) && - std::equal(candidate.begin(), candidate.end(), header.begin())) { + ++fallback_candidates; + if (atrac_source_matches_header(file.native_path, header, parsed)) { matched = file.native_path; break; } @@ -961,7 +1021,8 @@ std::filesystem::path identify_atrac_source(std::uint32_t guest_buffer, std::ostringstream line; line << "ATRAC_SOURCE resolve_us=" << elapsed_us << " direct_buffer=" << (direct_buffer ? 1 : 0) - << " fallback_reads=" << fallback_reads + << " direct_rejected=" << (direct_rejected ? 1 : 0) + << " fallback_candidates=" << fallback_candidates << " source=" << (matched.empty() ? "" : matched.filename().string()); runtime_log_line(line.str()); return matched; @@ -11458,7 +11519,12 @@ void install_profile(psprecomp::Runtime &runtime, std::uint32_t user_arena_start if (read != 0u && read_file != nullptr && is_atrac_source_path(read_file->native_path)) { const std::uint64_t file_start = static_cast(read_file->start_sector) * 2048u; - if (read_absolute >= file_start && + // Only a read that begins at the ATRAC file header can be a + // direct producer for SetHalfwayBuffer. Interior streaming + // chunks are deliberately not associated with the guest + // address; that stale hint was the radio X/UI -> radio Y/audio + // regression introduced by the NEWS-era producer shortcut. + if (read_absolute == file_start && read >= 12u && read_absolute + read <= file_start + read_file->size) { if (file_table.recent_atrac_reads.size() >= 32u) file_table.recent_atrac_reads.erase(file_table.recent_atrac_reads.begin()); @@ -11523,6 +11589,7 @@ void install_profile(psprecomp::Runtime &runtime, std::uint32_t user_arena_start return; } file_table.recent_atrac_reads.erase(dst); + const std::streampos read_start = it->second.tellg(); const bool time_io = perf_timing_enabled(); const auto io_entry = time_io ? std::chrono::steady_clock::now() : std::chrono::steady_clock::time_point{}; @@ -11532,6 +11599,7 @@ void install_profile(psprecomp::Runtime &runtime, std::uint32_t user_arena_start if (time_io) io_host_time_this_vblank += std::chrono::steady_clock::now() - io_entry; if (read != 0u) { if (const auto path_it = file_table.file_paths.find(fd); + read_start == std::streampos(0) && read >= 12u && path_it != file_table.file_paths.end() && is_atrac_source_path(path_it->second)) { if (file_table.recent_atrac_reads.size() >= 32u) file_table.recent_atrac_reads.erase(file_table.recent_atrac_reads.begin()); @@ -12562,6 +12630,57 @@ bool run_profile_self_tests(std::string &error) { wlan_runtime.invoke_import("sceAtrac3plus", 0x61EB33F5u, atrac_context); require(atrac_context.gpr[2] == 0u, "sceAtracReleaseAtracID failed for a valid context"); + // V8.6 regression guard: a reused guest staging address may still + // carry a producer hint for another same-sized radio stream. The + // current RIFF bytes, not the stale address->path association, must + // choose the decoder source. + { + const auto temp_root = std::filesystem::temp_directory_path() / + "psprecomp_v86_atrac_identity"; + std::error_code temp_error; + std::filesystem::remove_all(temp_root, temp_error); + std::filesystem::create_directories(temp_root, temp_error); + require(!temp_error, "cannot create ATRAC identity self-test directory"); + const auto right_path = temp_root / "RIGHT.AT3"; + const auto wrong_path = temp_root / "WRONG.AT3"; + std::vector right_bytes(0x1000u, 0u); + std::vector wrong_bytes(0x1000u, 0u); + std::copy(atrac_header.begin(), atrac_header.end(), right_bytes.begin()); + std::copy(atrac_header.begin(), atrac_header.end(), wrong_bytes.begin()); + wrong_bytes[0x60u] ^= 0x5Au; // data payload only; parsed metadata stays identical. + { std::ofstream out(right_path, std::ios::binary); + out.write(reinterpret_cast(right_bytes.data()), + static_cast(right_bytes.size())); } + { std::ofstream out(wrong_path, std::ios::binary); + out.write(reinterpret_cast(wrong_bytes.data()), + static_cast(wrong_bytes.size())); } + file_table.virtual_files_by_path.emplace( + normalized_native_path(right_path), VirtualDiscFile{right_path, 0x1000u, 0x1000u}); + file_table.virtual_files_by_path.emplace( + normalized_native_path(wrong_path), VirtualDiscFile{wrong_path, 0x1002u, 0x1000u}); + file_table.recent_atrac_reads[atrac_buffer] = wrong_path; + wlan_runtime.memory().copy_in(atrac_buffer, atrac_header); + + psprecomp::AllegrexContext radio_identity_context{}; + radio_identity_context.set_gpr(4u, atrac_buffer); + radio_identity_context.set_gpr(5u, 0x100u); + radio_identity_context.set_gpr(6u, 0x400u); + wlan_runtime.invoke_import("sceAtrac3plus", 0x0FAE370Eu, radio_identity_context); + require(radio_identity_context.gpr[2] == 0u && + atrac_contexts[0].allocated && + atrac_contexts[0].source_path.filename() == right_path.filename(), + "ATRAC source identity trusted a stale radio staging-buffer hint"); + radio_identity_context = {}; + radio_identity_context.set_gpr(4u, 0u); + wlan_runtime.invoke_import("sceAtrac3plus", 0x61EB33F5u, radio_identity_context); + require(radio_identity_context.gpr[2] == 0u, + "ATRAC source identity self-test could not release its context"); + file_table.recent_atrac_reads.clear(); + file_table.atrac_source_identities.clear(); + file_table.virtual_files_by_path.clear(); + std::filesystem::remove_all(temp_root, temp_error); + } + constexpr std::uint32_t sas_core = 0x08820000u; constexpr std::uint32_t sas_data = 0x08821000u; constexpr std::uint32_t sas_loop_data = 0x08821100u; diff --git a/profiles/vcs/host/vcs_runtime_log.cpp b/profiles/vcs/host/vcs_runtime_log.cpp index 7d46a61..865f930 100644 --- a/profiles/vcs/host/vcs_runtime_log.cpp +++ b/profiles/vcs/host/vcs_runtime_log.cpp @@ -65,7 +65,7 @@ void runtime_log_initialize(const VcsConfiguration &configuration) { return; } s.file << "VCSNative runtime log\n"; - s.file << "stage=perf-v8.4-aggressive-cpu-direct-2026-08-18\n"; + s.file << "stage=perf-v8.6-radio-identity-vfpu-ct2-2026-08-18\n"; s.file << "config=" << configuration.source_path.string() << '\n'; s.file << "started=" << timestamp_now() << '\n'; s.file << "perf_telemetry=" << (configuration.diagnostics.perf_telemetry ? 1 : 0) @@ -87,9 +87,9 @@ void runtime_log_initialize(const VcsConfiguration &configuration) { << " unwind_fix=1 reentry_guard=1 dataflow=1 vfpu_block32=133 mem_runs=35 mem_words=287" << " append32=51 advance32=89 simd_mat4=4 simd_matvec=19" << " gpr_shadow_clusters=4 gpr_shadow_regs=24 gpr_shadow_occurrences=3461 geometry_shadow=0" - << " perf_layer=10 cpu_lean_revision=2 cpu_aggressive_revision=4 correctness_revision=8271 recovery_from_821=1" + << " perf_layer=12 cpu_lean_revision=2 cpu_aggressive_revision=6 correctness_revision=8272 recovery_from_821=1" << " save_transaction=1 save_lifecycle_v82=1 sas_endflag_latched=1 sas_loop_history_restore=1" - << " atrac_virtual_source=1 atrac_stall_diag=1 atrac_stream_resident_status=1 atrac_nonloop_resident=-2 atrac_loop_resident=-3 output2_success_zero=1 output2_master_watermark=1 output2_late_catchup=0 save_repro_checkpoint=1 save_repro_trace=1 save_repro_hotkey_f8=1 save_repro_trace_hotkey_f10=1 save_repro_auto_restore=1 save_repro_dispatch_sample_stride=64 save_repro_passive_until_f8=1 save_repro_hle_hotpath=0 save_repro_ui_hotkeys=1 save_repro_f10_async_fallback=1 save_repro_collector_partial_bundle=1 save_repro_internal_ini_gate=1 save_repro_default_enabled=0 save_exitdelete_semantics=1 save_repro_legacy_exitdelete_repair=1 save_partition_reuse=1 news_atrac_v825_guard=1 runtime_chain_telemetry_default=0 arch_fastmem=1 aot_direct_fastmem_default=1 aot_hard_fastmem=1 aot_branchless_mem=1 aot_vfpu_block32=19455 aot_scalar_run_blocks=10665 aot_scalar_run_words=53173 aot_load_runs=5633 aot_store_runs=5032 vfpu_default_dest_fast=1 vfpu_vmscl_default_fast=1" + << " atrac_virtual_source=1 atrac_stall_diag=1 atrac_stream_resident_status=1 atrac_nonloop_resident=-2 atrac_loop_resident=-3 output2_success_zero=1 output2_master_watermark=1 output2_late_catchup=0 save_repro_checkpoint=1 save_repro_trace=1 save_repro_hotkey_f8=1 save_repro_trace_hotkey_f10=1 save_repro_auto_restore=1 save_repro_dispatch_sample_stride=64 save_repro_passive_until_f8=1 save_repro_hle_hotpath=0 save_repro_ui_hotkeys=1 save_repro_f10_async_fallback=1 save_repro_collector_partial_bundle=1 save_repro_internal_ini_gate=1 save_repro_default_enabled=0 save_exitdelete_semantics=1 save_repro_legacy_exitdelete_repair=1 save_partition_reuse=1 news_atrac_v825_guard=1 runtime_chain_telemetry_default=0 arch_fastmem=1 aot_direct_fastmem_default=1 aot_hard_fastmem=1 aot_branchless_mem=1 aot_vfpu_block32=19455 aot_scalar_run_blocks=10665 aot_scalar_run_words=53173 aot_load_runs=5633 aot_store_runs=5032 vfpu_default_dest_fast=1 vfpu_vmscl_default_fast=1 vfpu_default_fastlane=1 vfpu_vec3_fastlane_sites=2469 vfpu_unary_fastlane_sites=1607 vfpu_vdot_default_fast=1 vfpu_vscl_default_fast=1 radio_atrac_identity_guard=1 atrac_direct_header_validation=1 atrac_identity_cache=1 atrac_header_only_producer_hint=1 vfpu_vtfm_ct=1 vfpu_vtfm_ct_sites=337 vfpu_vi2f_ct=1 vfpu_vi2f_ct_sites=164 vfpu_vcmp_default_fast=1 vfpu_vcmp_ct_sites=558 vfpu_vcmov_default_fast=1 vfpu_vcmov_ct_sites=461 vfpu_cross_default_fast=1 vfpu_cross_ct_sites=217 vfpu_helper_ct2=1" << " tier2_direct_fastmem=1 tier2_direct_mem_sites=1969 tier2_deep_telemetry_default=0 geometry_fusion_rollback=1" << " entity_leaf_inline=1 entity_leaf_scheduler_accounting=1 entity_leaf_resume_pc_fix=1" << " ge_async_default=0 parallel_vertex_decode_default=0" diff --git a/profiles/vcs/host/vcs_tier2_cluster_edge43.cpp b/profiles/vcs/host/vcs_tier2_cluster_edge43.cpp index a429871..9f8a674 100644 --- a/profiles/vcs/host/vcs_tier2_cluster_edge43.cpp +++ b/profiles/vcs/host/vcs_tier2_cluster_edge43.cpp @@ -257,18 +257,10 @@ SB_L_088B1780: ctx.execute_vfpu_compare3_ct<13u, 3u, 4u, 3u, 7u>(); { float vfpu_value[4]{}; ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 3u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<12u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<13u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<12u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<12u, 12u, 13u, 3u, 0u>(); ctx.execute_vfpu_vcmp_ct<12u, 28u, 3u, 7u>(); { const bool branch_taken = ((ctx.vfpu_ctrl[3] >> 4u) & 1u) != 0u; - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<5u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<4u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<15u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<15u, 5u, 4u, 3u, 0u>(); if (branch_taken) { goto SB_L_088B180C; } @@ -276,26 +268,10 @@ SB_L_088B1780: } SB_L_088B17B4: - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<14u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<5u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<4u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<8u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<12u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<15u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<14u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<9u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<14u, 0u, 1u, 3u, 0u>(); + ctx.execute_vfpu_vec3_ct<8u, 5u, 4u, 3u, 1u>(); + ctx.execute_vfpu_vec3_ct<12u, 1u, 0u, 3u, 1u>(); + ctx.execute_vfpu_vec3_ct<9u, 15u, 14u, 3u, 1u>(); ctx.vfpu_ctrl[0u] = 0x000040C9u; ctx.vfpu_ctrl[1u] = 0x000043C6u; ctx.execute_vfpu_vdot_ct<16u, 8u, 12u, 3u>(); @@ -397,22 +373,11 @@ SB_L_088B1B74: std::bit_cast(tier2_vfpu_words[2]), std::bit_cast(tier2_vfpu_words[3])}; ctx.write_vfpu_vector_ct<6u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<5u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<4u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<9u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<6u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<4u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<10u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<9u, 5u, 4u, 3u, 1u>(); + ctx.execute_vfpu_vec3_ct<10u, 6u, 4u, 3u, 1u>(); ctx.execute_vfpu_cross_quat_ct<17u, 10u, 9u, 3u>(); ctx.execute_vfpu_vdot_ct<110u, 17u, 17u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<110u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / std::sqrt(vfpu_s[i]); - ctx.write_vfpu_vector_with_destination_prefix_ct<110u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<110u, 110u, 1u, 17u>(); ctx.execute_vfpu_vscl_ct<7u, 17u, 110u, 3u>(); ctx.execute_vfpu_vdot_ct<103u, 7u, 4u, 3u>(); ctx.execute_vfpu_vdot_ct<24u, 1u, 7u, 3u>(); @@ -454,54 +419,19 @@ SB_L_088B1BC8: } SB_L_088B1BD0: - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<2u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<3u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<103u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<24u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<24u, 1u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<3u, 2u, 1u, 3u, 1u>(); + ctx.execute_vfpu_vec3_ct<24u, 103u, 24u, 1u, 1u>(); ctx.execute_vfpu_vdot_ct<56u, 3u, 7u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<56u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = 1.0f / vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<56u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<24u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<56u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<24u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<56u, 56u, 1u, 16u>(); + ctx.execute_vfpu_vec3_ct<24u, 24u, 56u, 1u, 2u>(); ctx.execute_vfpu_vscl_ct<8u, 3u, 24u, 3u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<8u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<1u, 8u, 1u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<4u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<5u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<8u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<6u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<5u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<11u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<4u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<12u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<5u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<13u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<8u, 4u, 5u, 3u, 1u>(); + ctx.execute_vfpu_vec3_ct<11u, 6u, 5u, 3u, 1u>(); + ctx.execute_vfpu_vec3_ct<12u, 1u, 4u, 3u, 1u>(); + ctx.execute_vfpu_vec3_ct<13u, 1u, 5u, 3u, 1u>(); ctx.execute_vfpu_cross_quat_ct<19u, 9u, 10u, 3u>(); ctx.execute_vfpu_cross_quat_ct<15u, 11u, 8u, 3u>(); ctx.execute_vfpu_cross_quat_ct<18u, 9u, 12u, 3u>(); @@ -554,24 +484,8 @@ SB_L_088B3FCC: ctx.write_vfpu_vector_ct<2u, 4u>(vfpu_value); } ctx.execute_vfpu_vx2i_ct<0u, 2u, 2u, 3u>(); ctx.execute_vfpu_vx2i_ct<1u, 66u, 2u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<0u, 3u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<3u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(23u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 3u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<3u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(23u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 3u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<1u, 3u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<0u, 0u, 3u, 23u>(); + ctx.execute_vfpu_vi2f_ct<1u, 1u, 3u, 23u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = tier2_gpr_5 + static_cast(0); const std::uint32_t tier2_vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/host/vcs_tier2_cluster_entity.cpp b/profiles/vcs/host/vcs_tier2_cluster_entity.cpp index 5ab7d41..0ddba50 100644 --- a/profiles/vcs/host/vcs_tier2_cluster_entity.cpp +++ b/profiles/vcs/host/vcs_tier2_cluster_entity.cpp @@ -636,47 +636,10 @@ SB_L_08A6E930: std::bit_cast(tier2_vfpu_words[2]), std::bit_cast(tier2_vfpu_words[3])}; ctx.write_vfpu_vector_ct<8u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<32u, 4u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<4u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - if constexpr (vfpu_side == 4u) { - psprecomp::vcs_tier2_mat4_vec_first3_ordered(vfpu_matrix, vfpu_target, vfpu_result); - } else { - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<5u, 4u>(vfpu_result); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<100u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<104u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<100u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<100u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<100u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<100u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<8u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<5u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<5u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<5u, 32u, 4u, 4u, 3u>(); + ctx.execute_vfpu_vec3_ct<100u, 100u, 104u, 1u, 0u>(); + ctx.execute_vfpu_vec3_ct<100u, 100u, 100u, 1u, 2u>(); + ctx.execute_vfpu_vec3_ct<5u, 8u, 5u, 3u, 1u>(); ctx.execute_vfpu_vdot_ct<4u, 5u, 5u, 3u>(); ctx.execute_vfpu_vcmp_ct<4u, 100u, 1u, 7u>(); tier2_gpr_4 = (0u | 0u); @@ -1161,11 +1124,7 @@ SB_L_08A6ECA0: std::bit_cast(tier2_vfpu_words[2]), std::bit_cast(tier2_vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); ctx.gpr[21] = (tier2_gpr_29 + static_cast(144)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[21] + static_cast(0); @@ -1211,37 +1170,8 @@ SB_L_08A6ECA0: std::bit_cast(tier2_vfpu_words[2]), std::bit_cast(tier2_vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - if constexpr (vfpu_side == 4u) { - psprecomp::vcs_tier2_mat4_vec_first3_ordered(vfpu_matrix, vfpu_target, vfpu_result); - } else { - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); tier2_gpr_4 = (tier2_gpr_29 + static_cast(48)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = tier2_gpr_4 + static_cast(0); @@ -1338,37 +1268,8 @@ SB_L_08A6ED2C: std::bit_cast(tier2_vfpu_words[2]), std::bit_cast(tier2_vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - if constexpr (vfpu_side == 4u) { - psprecomp::vcs_tier2_mat4_vec_first3_ordered(vfpu_matrix, vfpu_target, vfpu_result); - } else { - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); tier2_gpr_4 = (tier2_gpr_29 + static_cast(160)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = tier2_gpr_4 + static_cast(0); diff --git a/profiles/vcs/host/vcs_tier2_cluster_geometry.cpp b/profiles/vcs/host/vcs_tier2_cluster_geometry.cpp index 567f6d8..1983436 100644 --- a/profiles/vcs/host/vcs_tier2_cluster_geometry.cpp +++ b/profiles/vcs/host/vcs_tier2_cluster_geometry.cpp @@ -908,37 +908,8 @@ SB_L_08955444: ctx.gpr[10] = (ctx.gpr[28] + static_cast(-17792)); ctx.set_vfpu_scalar_bits_ct<76u>(tier2_mem.aot_load32(ctx.gpr[10] + static_cast(0))); ctx.execute_vfpu_vh2f_ct<21u, 12u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<117u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<76u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<117u, 1u>(vfpu_d); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 4u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<21u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - if constexpr (vfpu_side == 4u) { - psprecomp::vcs_tier2_mat4_vec_first3_ordered(vfpu_matrix, vfpu_target, vfpu_result); - } else { - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<14u, 4u>(vfpu_result); } + ctx.execute_vfpu_vec3_ct<117u, 117u, 76u, 1u, 2u>(); + ctx.execute_vfpu_vtfm_ct<14u, 36u, 21u, 4u, 3u>(); ctx.vfpu_ctrl[1u] = 0x000000FFu; ctx.execute_vfpu_vcmp_ct<14u, 21u, 4u, 3u>(); { const bool branch_taken = ((ctx.vfpu_ctrl[3] >> 5u) & 1u) == 0u; @@ -950,11 +921,7 @@ SB_L_08955444: } SB_L_08955474: - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<13u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<27u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<13u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<13u, 13u, 27u, 3u, 1u>(); ctx.execute_vfpu_vdot_ct<12u, 13u, 13u, 3u>(); tier2_mem.aot_store32(ctx.gpr[9] + static_cast(0), ctx.vfpu_scalar_bits_ct<12u>()); tier2_mem.aot_store32(ctx.gpr[9] + static_cast(4), ctx.gpr[5]); @@ -1055,32 +1022,7 @@ SB_L_08955544: } SB_L_0895554C: - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<40u, 4u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<21u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - if constexpr (vfpu_side == 4u) { - psprecomp::vcs_tier2_mat4_vec_first3_ordered(vfpu_matrix, vfpu_target, vfpu_result); - } else { - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<14u, 4u>(vfpu_result); } + ctx.execute_vfpu_vtfm_ct<14u, 40u, 21u, 4u, 3u>(); ctx.vfpu_ctrl[0u] = 0x00000FE4u; ctx.vfpu_ctrl[1u] = 0x000000FFu; ctx.execute_vfpu_vcmp_ct<14u, 21u, 4u, 3u>(); @@ -1641,11 +1583,7 @@ SB_L_08957500: ctx.set_vfpu_scalar_bits_ct<27u>(ctx.gpr[25]); ctx.set_vfpu_scalar_bits_ct<59u>(ctx.gpr[2]); ctx.set_vfpu_scalar_bits_ct<91u>(ctx.gpr[3]); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<27u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<15u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<27u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<27u, 27u, 15u, 3u, 1u>(); { float vfpu_value[4]{}; vfpu_value[3u] = 1.0f; ctx.write_vfpu_vector_with_destination_prefix_ct<59u, 4u>(vfpu_value); } { float vfpu_value[4]{}; @@ -1654,10 +1592,7 @@ SB_L_08957500: ctx.write_vfpu_vector_with_destination_prefix_ct<12u, 4u>(vfpu_value); } { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<13u, 2u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<12u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<12u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<12u, 12u, 3u, 2u>(); { float vfpu_s[16]{}, vfpu_t[16]{}, vfpu_d[16]{}; ctx.read_vfpu_matrix_ct<36u, 4u>(vfpu_s); ctx.read_vfpu_matrix_ct<24u, 4u>(vfpu_t); @@ -2125,10 +2060,7 @@ SB_L_08958F60: std::bit_cast(tier2_vfpu_words[2]), std::bit_cast(tier2_vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(96)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2381,37 +2313,8 @@ SB_L_08959118: std::bit_cast(tier2_vfpu_words[2]), std::bit_cast(tier2_vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 3u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<1u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - if constexpr (vfpu_side == 4u) { - psprecomp::vcs_tier2_mat4_vec_first3_ordered(vfpu_matrix, vfpu_target, vfpu_result); - } else { - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_result); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 1u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<0u, 0u, 7u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(384)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2548,10 +2451,7 @@ SB_L_08959258: std::bit_cast(tier2_vfpu_words[2]), std::bit_cast(tier2_vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(112)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -2708,10 +2608,7 @@ SB_L_08959414: std::bit_cast(tier2_vfpu_words[2]), std::bit_cast(tier2_vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<0u, 0u, 3u, 2u>(); ctx.gpr[16] = (ctx.gpr[29] + static_cast(128)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[16] + static_cast(0); @@ -3437,37 +3334,8 @@ SB_L_089598B0: ctx.set_vfpu_scalar_bits_ct<44u>(tier2_mem.aot_load32(ctx.gpr[17] + static_cast(8))); ctx.set_vfpu_scalar_bits_ct<76u>(tier2_mem.aot_load32(ctx.gpr[23] + static_cast(0))); ctx.execute_vfpu_vh2f_ct<21u, 12u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<117u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<76u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<117u, 1u>(vfpu_d); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 4u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<21u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - if constexpr (vfpu_side == 4u) { - psprecomp::vcs_tier2_mat4_vec_first3_ordered(vfpu_matrix, vfpu_target, vfpu_result); - } else { - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<14u, 4u>(vfpu_result); } + ctx.execute_vfpu_vec3_ct<117u, 117u, 76u, 1u, 2u>(); + ctx.execute_vfpu_vtfm_ct<14u, 36u, 21u, 4u, 3u>(); ctx.vfpu_ctrl[1u] = 0x000000FFu; ctx.execute_vfpu_vcmp_ct<14u, 21u, 4u, 3u>(); { const bool branch_taken = ((ctx.vfpu_ctrl[3] >> 5u) & 1u) == 0u; @@ -3557,37 +3425,8 @@ SB_L_08959930: ctx.set_vfpu_scalar_bits_ct<44u>(tier2_mem.aot_load32(ctx.gpr[17] + static_cast(8))); ctx.set_vfpu_scalar_bits_ct<76u>(tier2_mem.aot_load32(ctx.gpr[23] + static_cast(0))); ctx.execute_vfpu_vh2f_ct<21u, 12u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<117u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<76u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<117u, 1u>(vfpu_d); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 4u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<21u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - if constexpr (vfpu_side == 4u) { - psprecomp::vcs_tier2_mat4_vec_first3_ordered(vfpu_matrix, vfpu_target, vfpu_result); - } else { - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<14u, 4u>(vfpu_result); } + ctx.execute_vfpu_vec3_ct<117u, 117u, 76u, 1u, 2u>(); + ctx.execute_vfpu_vtfm_ct<14u, 36u, 21u, 4u, 3u>(); ctx.vfpu_ctrl[1u] = 0x000000FFu; ctx.execute_vfpu_vcmp_ct<14u, 21u, 4u, 3u>(); { const bool branch_taken = ((ctx.vfpu_ctrl[3] >> 5u) & 1u) == 0u; @@ -4082,37 +3921,8 @@ SB_L_08959BF0: ctx.set_vfpu_scalar_bits_ct<44u>(tier2_mem.aot_load32(ctx.gpr[16] + static_cast(8))); ctx.set_vfpu_scalar_bits_ct<76u>(tier2_mem.aot_load32(ctx.gpr[23] + static_cast(0))); ctx.execute_vfpu_vh2f_ct<21u, 12u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<117u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<76u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<117u, 1u>(vfpu_d); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 4u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<21u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - if constexpr (vfpu_side == 4u) { - psprecomp::vcs_tier2_mat4_vec_first3_ordered(vfpu_matrix, vfpu_target, vfpu_result); - } else { - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<14u, 4u>(vfpu_result); } + ctx.execute_vfpu_vec3_ct<117u, 117u, 76u, 1u, 2u>(); + ctx.execute_vfpu_vtfm_ct<14u, 36u, 21u, 4u, 3u>(); ctx.vfpu_ctrl[1u] = 0x000000FFu; ctx.execute_vfpu_vcmp_ct<14u, 21u, 4u, 3u>(); { const bool branch_taken = ((ctx.vfpu_ctrl[3] >> 5u) & 1u) == 0u; @@ -4280,37 +4090,8 @@ SB_L_08959CD8: ctx.set_vfpu_scalar_bits_ct<44u>(tier2_mem.aot_load32(ctx.gpr[16] + static_cast(8))); ctx.set_vfpu_scalar_bits_ct<76u>(tier2_mem.aot_load32(ctx.gpr[23] + static_cast(0))); ctx.execute_vfpu_vh2f_ct<21u, 12u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<117u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<76u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<117u, 1u>(vfpu_d); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 4u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<21u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - if constexpr (vfpu_side == 4u) { - psprecomp::vcs_tier2_mat4_vec_first3_ordered(vfpu_matrix, vfpu_target, vfpu_result); - } else { - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<14u, 4u>(vfpu_result); } + ctx.execute_vfpu_vec3_ct<117u, 117u, 76u, 1u, 2u>(); + ctx.execute_vfpu_vtfm_ct<14u, 36u, 21u, 4u, 3u>(); ctx.vfpu_ctrl[1u] = 0x000000FFu; ctx.execute_vfpu_vcmp_ct<14u, 21u, 4u, 3u>(); { const bool branch_taken = ((ctx.vfpu_ctrl[3] >> 5u) & 1u) == 0u; @@ -4812,37 +4593,8 @@ SB_L_0895A03C: ctx.set_vfpu_scalar_bits_ct<44u>(tier2_mem.aot_load32(ctx.gpr[16] + static_cast(8))); ctx.set_vfpu_scalar_bits_ct<76u>(tier2_mem.aot_load32(ctx.gpr[22] + static_cast(0))); ctx.execute_vfpu_vh2f_ct<21u, 12u, 2u>(); - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<117u, 1u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<76u, 1u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i] * vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<117u, 1u>(vfpu_d); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix_ct<36u, 4u>(vfpu_matrix); - ctx.read_vfpu_vector_ct<21u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - if constexpr (vfpu_side == 4u) { - psprecomp::vcs_tier2_mat4_vec_first3_ordered(vfpu_matrix, vfpu_target, vfpu_result); - } else { - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix_ct<14u, 4u>(vfpu_result); } + ctx.execute_vfpu_vec3_ct<117u, 117u, 76u, 1u, 2u>(); + ctx.execute_vfpu_vtfm_ct<14u, 36u, 21u, 4u, 3u>(); ctx.vfpu_ctrl[1u] = 0x000000FFu; ctx.execute_vfpu_vcmp_ct<14u, 21u, 4u, 3u>(); { const bool branch_taken = ((ctx.vfpu_ctrl[3] >> 5u) & 1u) == 0u; @@ -6561,24 +6313,8 @@ SB_L_0895AFEC: ctx.set_vfpu_scalar_bits_ct<14u>(tier2_mem.aot_load32(ctx.gpr[18] + static_cast(20))); ctx.execute_vfpu_vx2i_ct<12u, 13u, 2u, 3u>(); ctx.execute_vfpu_vx2i_ct<13u, 14u, 1u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<12u, 4u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(31u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 4u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<12u, 4u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<13u, 2u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<2u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(31u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 2u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<13u, 2u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<12u, 12u, 4u, 31u>(); + ctx.execute_vfpu_vi2f_ct<13u, 13u, 2u, 31u>(); ctx.execute_vfpu_vcmp_ct<52u, 31u, 3u, 6u>(); ctx.execute_vfpu_vcmov_ct<1u, 108u, 1u, 0u, true>(); ctx.execute_vfpu_vcmov_ct<1u, 12u, 1u, 0u, false>(); diff --git a/profiles/vcs/host/vcs_tier2_cluster_matrix.cpp b/profiles/vcs/host/vcs_tier2_cluster_matrix.cpp index 22aa99d..6d4109d 100644 --- a/profiles/vcs/host/vcs_tier2_cluster_matrix.cpp +++ b/profiles/vcs/host/vcs_tier2_cluster_matrix.cpp @@ -322,36 +322,8 @@ SB_L_088B4738: ctx.write_vfpu_vector_with_destination_prefix_ct<39u, 3u>(vfpu_value); } { float vfpu_value[4]{}; ctx.write_vfpu_vector_with_destination_prefix_ct<35u, 3u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<19u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - if constexpr (vfpu_side == 4u) { - psprecomp::vcs_tier2_mat4_vec_first3_ordered(vfpu_matrix, vfpu_target, vfpu_result); - } else { - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 7u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<7u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<7u, 3u>(vfpu_d); } + ctx.execute_vfpu_vtfm_ct<7u, 36u, 19u, 3u, 3u>(); + ctx.execute_vfpu_unary_ct<7u, 7u, 3u, 2u>(); { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; ctx.write_vfpu_vector_with_destination_prefix_ct<99u, 1u>(vfpu_value); } { float vfpu_value[4]{1.0f, 1.0f, 1.0f, 1.0f}; @@ -387,59 +359,13 @@ SB_L_088B4738: std::bit_cast(tier2_vfpu_words[2]), std::bit_cast(tier2_vfpu_words[3])}; ctx.write_vfpu_vector_ct<29u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<104u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<117u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<104u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<118u, 1u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<104u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<116u, 1u>(vfpu_d); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 44u, 3u); - ctx.read_vfpu_vector_ct<8u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - if constexpr (vfpu_side == 4u) { - psprecomp::vcs_tier2_mat4_vec_first3_ordered(vfpu_matrix, vfpu_target, vfpu_result); - } else { - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 20u, vfpu_side); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<20u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<15u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<20u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<55u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<16u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<29u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<55u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] + vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<17u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<117u, 104u, 1u, 0u>(); + ctx.execute_vfpu_unary_ct<118u, 104u, 1u, 0u>(); + ctx.execute_vfpu_unary_ct<116u, 104u, 1u, 0u>(); + ctx.execute_vfpu_vtfm_ct<20u, 44u, 8u, 3u, 3u>(); + ctx.execute_vfpu_vec3_ct<20u, 20u, 15u, 3u, 0u>(); + ctx.execute_vfpu_vec3_ct<16u, 28u, 55u, 3u, 1u>(); + ctx.execute_vfpu_vec3_ct<17u, 29u, 55u, 3u, 0u>(); ctx.gpr[4] = (ctx.gpr[29] + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<20u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[4] + static_cast(0); @@ -467,42 +393,11 @@ SB_L_088B47E4: SB_L_088B47F0: ctx.gpr[2] = (0u + static_cast(1)); ctx.gpr[2] = (ctx.gpr[2] & 255u); - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 0u, 3u); - ctx.read_vfpu_vector_ct<3u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - if constexpr (vfpu_side == 4u) { - psprecomp::vcs_tier2_mat4_vec_first3_ordered(vfpu_matrix, vfpu_target, vfpu_result); - } else { - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 35u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<35u, 0u, 3u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.write_vfpu_vector_with_destination_prefix_ct<3u, 3u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<35u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = -vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<35u, 3u>(vfpu_d); } - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<19u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<39u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<35u, 35u, 3u, 2u>(); + ctx.execute_vfpu_unary_ct<39u, 19u, 3u, 0u>(); { float vfpu_value[4]{}; ctx.write_vfpu_vector_with_destination_prefix_ct<7u, 3u>(vfpu_value); } { float vfpu_s[16]{}, vfpu_t[16]{}, vfpu_d[16]{}; @@ -1062,32 +957,7 @@ SB_L_088B4C7C: std::bit_cast(tier2_vfpu_words[2]), std::bit_cast(tier2_vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - if constexpr (vfpu_side == 4u) { - psprecomp::vcs_tier2_mat4_vec_first3_ordered(vfpu_matrix, vfpu_target, vfpu_result); - } else { - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[17] + static_cast(0); const std::uint32_t tier2_vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -1404,32 +1274,7 @@ SB_L_088B4EE4: std::bit_cast(tier2_vfpu_words[2]), std::bit_cast(tier2_vfpu_words[3])}; ctx.write_vfpu_vector_ct<7u, 4u>(vfpu_value); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 36u, 3u); - ctx.read_vfpu_vector_ct<7u, 3u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 3u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - if constexpr (vfpu_side == 4u) { - psprecomp::vcs_tier2_mat4_vec_first3_ordered(vfpu_matrix, vfpu_target, vfpu_result); - } else { - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 0u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<0u, 36u, 7u, 3u, 3u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = ctx.gpr[18] + static_cast(0); const std::uint32_t tier2_vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; diff --git a/profiles/vcs/host/vcs_tier2_cluster_physics.cpp b/profiles/vcs/host/vcs_tier2_cluster_physics.cpp index d06259a..776d602 100644 --- a/profiles/vcs/host/vcs_tier2_cluster_physics.cpp +++ b/profiles/vcs/host/vcs_tier2_cluster_physics.cpp @@ -219,41 +219,8 @@ SB_L_08A09B2C: ctx.set_vfpu_scalar_bits_ct<29u>(ctx.gpr[8]); ctx.set_vfpu_scalar_bits_ct<61u>(ctx.gpr[9]); ctx.set_vfpu_scalar_bits_ct<93u>(ctx.gpr[10]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<29u, 3u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<3u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(15u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 3u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<29u, 3u>(vfpu_d); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 48u, 4u); - ctx.read_vfpu_vector_ct<29u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - if constexpr (vfpu_side == 4u) { - psprecomp::vcs_tier2_mat4_vec_first3_ordered(vfpu_matrix, vfpu_target, vfpu_result); - } else { - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 12u, vfpu_side); } + ctx.execute_vfpu_vi2f_ct<29u, 29u, 3u, 15u>(); + ctx.execute_vfpu_vtfm_ct<12u, 48u, 29u, 4u, 3u>(); ctx.execute_vfpu_vcmp_ct<12u, 31u, 4u, 7u>(); // vflush: architectural no-op that retains VFPU prefixes ctx.gpr[11] = (ctx.vfpu_scalar_bits_ct<131u>()); @@ -265,41 +232,8 @@ SB_L_08A09B2C: ctx.set_vfpu_scalar_bits_ct<28u>(ctx.gpr[8]); ctx.set_vfpu_scalar_bits_ct<60u>(ctx.gpr[9]); ctx.set_vfpu_scalar_bits_ct<92u>(ctx.gpr[10]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<28u, 3u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<3u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(15u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 3u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 3u>(vfpu_d); } - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 48u, 4u); - ctx.read_vfpu_vector_ct<28u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - if constexpr (vfpu_side == 4u) { - psprecomp::vcs_tier2_mat4_vec_first3_ordered(vfpu_matrix, vfpu_target, vfpu_result); - } else { - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 12u, vfpu_side); } + ctx.execute_vfpu_vi2f_ct<28u, 28u, 3u, 15u>(); + ctx.execute_vfpu_vtfm_ct<12u, 48u, 28u, 4u, 3u>(); ctx.execute_vfpu_vcmp_ct<12u, 31u, 4u, 7u>(); // vflush: architectural no-op that retains VFPU prefixes ctx.gpr[12] = (ctx.vfpu_scalar_bits_ct<131u>()); @@ -322,15 +256,7 @@ SB_L_08A09BE4: ctx.set_vfpu_scalar_bits_ct<15u>(ctx.gpr[8]); ctx.set_vfpu_scalar_bits_ct<47u>(ctx.gpr[9]); ctx.set_vfpu_scalar_bits_ct<79u>(ctx.gpr[10]); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_ct<15u, 3u>(vfpu_s); - ctx.apply_vfpu_source_prefix_ct<3u, 0u>(vfpu_s); - const float vfpu_scale = std::ldexp(1.0f, -static_cast(15u)); - for (std::uint32_t vfpu_i = 0; vfpu_i < 3u; ++vfpu_i) { - const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i])); - vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale; - } - ctx.write_vfpu_vector_with_destination_prefix_ct<15u, 3u>(vfpu_d); } + ctx.execute_vfpu_vi2f_ct<15u, 15u, 3u, 15u>(); ctx.execute_vfpu_vcmp_ct<15u, 28u, 3u, 1u>(); if (((ctx.vfpu_ctrl[3] >> 5u) & 1u) != 0u) { ctx.gpr[11] = ((ctx.gpr[16] >> 8u) & 0x000000FFu); @@ -347,32 +273,7 @@ SB_L_08A09C10: goto SB_L_08A09C1C; SB_L_08A09C1C: - { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{}; - ctx.read_vfpu_matrix(vfpu_matrix, 48u, 4u); - ctx.read_vfpu_vector_ct<15u, 4u>(vfpu_target_raw); - constexpr std::uint32_t vfpu_side = 4u; - constexpr std::uint32_t vfpu_input_length = 3u; - for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f; - if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f; - if constexpr (vfpu_side == 4u) { - psprecomp::vcs_tier2_mat4_vec_first3_ordered(vfpu_matrix, vfpu_target, vfpu_result); - } else { - for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) { - float sum = 0.0f; - for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column]; - vfpu_result[row] = sum; - } - } - float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u], - vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]}; - ctx.apply_vfpu_source_prefix_ct<4u, 0u>(vfpu_final_row); - ctx.apply_vfpu_source_prefix_ct<4u, 1u>(vfpu_target); - for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column]; - const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2]; - const std::uint32_t vfpu_last_lane = vfpu_side - 1u; - ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) | - ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u)); - ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, 12u, vfpu_side); } + ctx.execute_vfpu_vtfm_ct<12u, 48u, 15u, 4u, 3u>(); ctx.execute_vfpu_vcmp_ct<12u, 31u, 4u, 7u>(); // vflush: architectural no-op that retains VFPU prefixes ctx.gpr[11] = (ctx.vfpu_scalar_bits_ct<131u>()); @@ -424,15 +325,9 @@ SB_L_08A09CA4: ctx.gpr[19] = (ctx.gpr[19] ^ 1u); ctx.gpr[17] = (ctx.gpr[17] + static_cast(10)); ctx.gpr[5] = (ctx.gpr[5] + static_cast(1)); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<29u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<29u, 28u, 3u, 0u>(); if (ctx.gpr[17] != ctx.gpr[18]) { - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<15u, 3u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 3u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 15u, 3u, 0u>(); goto SB_L_08A09BE4; } goto SB_L_08A09CBC; diff --git a/profiles/vcs/host/vcs_tier2_cluster_world.cpp b/profiles/vcs/host/vcs_tier2_cluster_world.cpp index 029a623..9ccee92 100644 --- a/profiles/vcs/host/vcs_tier2_cluster_world.cpp +++ b/profiles/vcs/host/vcs_tier2_cluster_world.cpp @@ -2297,11 +2297,7 @@ SB_L_08A7ACB8: std::bit_cast(tier2_vfpu_words[2]), std::bit_cast(tier2_vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = tier2_gpr_29 + static_cast(0); const std::uint32_t tier2_vfpu_words[4]{std::bit_cast(vfpu_value[0]), std::bit_cast(vfpu_value[1]), std::bit_cast(vfpu_value[2]), std::bit_cast(vfpu_value[3])}; @@ -2315,10 +2311,7 @@ SB_L_08A7ACB8: std::bit_cast(tier2_vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); tier2_gpr_4 = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(tier2_gpr_4); tier2_gpr_31 = (0x08A7ACECu); @@ -3910,11 +3903,7 @@ SB_L_08A7D674: std::bit_cast(tier2_vfpu_words[2]), std::bit_cast(tier2_vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); tier2_gpr_4 = (tier2_gpr_29 + static_cast(16)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = tier2_gpr_4 + static_cast(0); @@ -3929,10 +3918,7 @@ SB_L_08A7D674: std::bit_cast(tier2_vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); tier2_gpr_4 = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(tier2_gpr_4); tier2_gpr_31 = (0x08A7D6ACu); @@ -4073,11 +4059,7 @@ SB_L_08A7D73C: std::bit_cast(tier2_vfpu_words[2]), std::bit_cast(tier2_vfpu_words[3])}; ctx.write_vfpu_vector_ct<1u, 4u>(vfpu_value); } - { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<0u, 3u, 0u>(vfpu_s); - ctx.read_vfpu_vector_with_source_prefix_ct<1u, 3u, 1u>(vfpu_t); - for (std::uint32_t i = 0; i < 3u; ++i) vfpu_d[i] = vfpu_s[i] - vfpu_t[i]; - ctx.write_vfpu_vector_with_destination_prefix_ct<0u, 3u>(vfpu_d); } + ctx.execute_vfpu_vec3_ct<0u, 0u, 1u, 3u, 1u>(); tier2_gpr_4 = (tier2_gpr_29 + static_cast(32)); { float vfpu_value[4]{}; ctx.read_vfpu_vector_ct<0u, 4u>(vfpu_value); const std::uint32_t vfpu_address = tier2_gpr_4 + static_cast(0); @@ -4092,10 +4074,7 @@ SB_L_08A7D73C: std::bit_cast(tier2_vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); tier2_gpr_4 = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[12] = std::bit_cast(tier2_gpr_4); tier2_gpr_31 = (0x08A7D774u); @@ -4146,10 +4125,7 @@ SB_L_08A7D7A4: std::bit_cast(tier2_vfpu_words[3])}; ctx.write_vfpu_vector_ct<0u, 4u>(vfpu_value); } ctx.execute_vfpu_vdot_ct<28u, 0u, 0u, 3u>(); - { float vfpu_s[4]{}, vfpu_d[4]{}; - ctx.read_vfpu_vector_with_source_prefix_ct<28u, 1u, 0u>(vfpu_s); - for (std::uint32_t i = 0; i < 1u; ++i) vfpu_d[i] = std::fabs(std::sqrt(vfpu_s[i])); - ctx.write_vfpu_vector_with_destination_prefix_ct<28u, 1u>(vfpu_d); } + ctx.execute_vfpu_unary_ct<28u, 28u, 1u, 22u>(); tier2_gpr_4 = (ctx.vfpu_scalar_bits_ct<28u>()); ctx.fpr[20] = std::bit_cast(tier2_gpr_4); tier2_gpr_4 = (17317u << 16u); diff --git a/profiles/vcs/scripts/build_release_ninja.bat b/profiles/vcs/scripts/build_release_ninja.bat index 94f0c09..b6cd767 100644 --- a/profiles/vcs/scripts/build_release_ninja.bat +++ b/profiles/vcs/scripts/build_release_ninja.bat @@ -115,6 +115,8 @@ set "CORRECTNESS_V826B_SAVE_REPRO_STAMP=%BUILD%\.vcs_correctness_v826b_save_repr set "CORRECTNESS_V827_SAVE_THREAD_STAMP=%BUILD%\.vcs_correctness_v827_save_thread_lifecycle_20260818" set "CORRECTNESS_V827A_SAVE_REPRO_GATE_STAMP=%BUILD%\.vcs_correctness_v827a_save_repro_gate_20260818" set "PERF_V84_AGGRESSIVE_CPU_STAMP=%BUILD%\.vcs_perf_v84_aggressive_cpu_direct_20260818" +set "PERF_V85_VFPU_FASTLANE_STAMP=%BUILD%\.vcs_perf_v85_aggressive_vfpu_fastlane_20260818" +set "PERF_V86_RADIO_VFPU_CT2_STAMP=%BUILD%\.vcs_perf_v86_radio_identity_vfpu_ct2_20260818" if not defined PSPRECOMP_TIER2_DEEP_TELEMETRY set "PSPRECOMP_TIER2_DEEP_TELEMETRY=OFF" if not defined PSPRECOMP_RUNTIME_CHAIN_TELEMETRY set "PSPRECOMP_RUNTIME_CHAIN_TELEMETRY=OFF" if /I "%PSPRECOMP_TIER2_DEEP_TELEMETRY%"=="1" set "PSPRECOMP_TIER2_DEEP_TELEMETRY=ON" @@ -132,8 +134,8 @@ echo CMake: %CMAKE_EXE% echo Ninja: %NINJA_EXE% echo Ninja workers: %JOBS% echo cl.exe /MP: OFF ^(Ninja owns compile parallelism^) -echo Generated AOT: O3; V8.4 hard direct-fastmem + branchless scalar memory + LV.Q/SV.Q 16-byte blocks -echo Correctness: V8.2.7A SAVE + V8.2.5 NEWS protected; scheduler/timing unchanged +echo Generated AOT: O3; V8.4 direct-memory + V8.5 fast lanes + V8.6 VFPU CT2 +echo Correctness: V8.6 radio identity + V8.2.7A SAVE + V8.2.5 NEWS protected echo Host/core LTCG: ON echo AVX2/fast paths: ON echo Tier2 deep diag: %PSPRECOMP_TIER2_DEEP_TELEMETRY% @@ -144,7 +146,7 @@ echo [0b/7] Reapplying BOOTFIX-safe Tier-2 transforms (OPT1 semantic transforms call "%PROFILE%\APPLY_TIER2_EXTREME.bat" if errorlevel 1 goto :FAIL -echo [0b2/7] Building V8.4 AGGRESSIVE CPU DIRECT over V8.2.7A stable correctness... +echo [0b2/7] Building V8.6 RADIO IDENTITY + VFPU CT2 over V8.5 baseline... set "PYTHON3_CMD=" py -3 -c "import sys; raise SystemExit(0 if sys.version_info.major == 3 else 1)" >nul 2>&1 if not errorlevel 1 set "PYTHON3_CMD=py -3" @@ -158,6 +160,8 @@ echo Python 3: %PYTHON3_CMD% if errorlevel 1 goto :FAIL %PYTHON3_CMD% "%PROFILE%\tools\optimize_generated_v84_cpu.py" "%PROFILE%" if errorlevel 1 goto :FAIL +%PYTHON3_CMD% "%PROFILE%\tools\optimize_generated_v85_vfpu.py" "%PROFILE%" +if errorlevel 1 goto :FAIL rem V8.2.7 checker includes the protected V8.2 CPU/Geometry, V8.2.5 NEWS, rem V8.2.6 passive checkpoint contract and the corrected PSP ExitDelete lifecycle. rem Older revision checkers pin exact stage strings and must not gate this stage. @@ -167,6 +171,10 @@ if errorlevel 1 goto :FAIL if errorlevel 1 goto :FAIL %PYTHON3_CMD% "%PROFILE%\tests\check_v84_aggressive_cpu.py" if errorlevel 1 goto :FAIL +%PYTHON3_CMD% "%PROFILE%\tests\check_v85_aggressive_vfpu.py" +if errorlevel 1 goto :FAIL +%PYTHON3_CMD% "%PROFILE%\tests\check_v86_radio_vfpu_ct2.py" +if errorlevel 1 goto :FAIL if exist "%BUILD%" if not exist "%SUPERBLOCK_STAMP%" ( echo. @@ -414,6 +422,31 @@ if exist "%BUILD%" if not exist "%PERF_V84_AGGRESSIVE_CPU_STAMP%" ( del /s /q "%BUILD%\*codegen_main*.obj" >nul 2>&1 ) +if exist "%BUILD%" if not exist "%PERF_V85_VFPU_FASTLANE_STAMP%" ( + echo. + echo [0c-v85vfpu/7] V8.5 AGGRESSIVE VFPU FASTLANE - rebuilding generated AOT/VFPU helpers once... + rem V8.5 changes generic VFPU lowering and the checked-in generated corpus only. + rem Scheduler/timing, DX12, savedata, ATRAC and host I/O objects stay cached. + del /s /q "%BUILD%\*generated_unit_*.obj" >nul 2>&1 + del /s /q "%BUILD%\*vfpu_tier2_tests*.obj" >nul 2>&1 + del /s /q "%BUILD%\*vcs_runtime_log*.obj" >nul 2>&1 + del /s /q "%BUILD%\*vcs_codegen_main*.obj" >nul 2>&1 + del /s /q "%BUILD%\*codegen_main*.obj" >nul 2>&1 +) + +if exist "%BUILD%" if not exist "%PERF_V86_RADIO_VFPU_CT2_STAMP%" ( + echo. + echo [0c-v86/7] V8.6 RADIO IDENTITY + VFPU CT2 - rebuilding affected CPU/profile objects once... + rem VFPU CT2 touches 100+ generated units and Allegrex helper templates; radio identity + rem changes only vcs_profile. Keep DX12, media backends and scheduler objects cached. + del /s /q "%BUILD%\*generated_unit_*.obj" >nul 2>&1 + del /s /q "%BUILD%\*vfpu_tier2_tests*.obj" >nul 2>&1 + del /s /q "%BUILD%\*vcs_profile*.obj" >nul 2>&1 + del /s /q "%BUILD%\*vcs_runtime_log*.obj" >nul 2>&1 + del /s /q "%BUILD%\*vcs_codegen_main*.obj" >nul 2>&1 + del /s /q "%BUILD%\*codegen_main*.obj" >nul 2>&1 +) + if exist "%BUILD%" if not exist "%BOOTFIX_STAMP%" ( echo. echo [0c/7] BOOTFIX revision changed - invalidating stale .obj/.pch once... @@ -473,6 +506,8 @@ if errorlevel 1 goto :FAIL >"%CORRECTNESS_V827_SAVE_THREAD_STAMP%" echo VCS V8.2.7 SAVE THREAD LIFECYCLE FIX 2026-08-18 >"%CORRECTNESS_V827A_SAVE_REPRO_GATE_STAMP%" echo VCS V8.2.7A INTERNAL SAVE_REPRO GATE 2026-08-18 >"%PERF_V84_AGGRESSIVE_CPU_STAMP%" echo VCS PERF V8.4 AGGRESSIVE CPU DIRECT 2026-08-18 +>"%PERF_V85_VFPU_FASTLANE_STAMP%" echo VCS PERF V8.5 AGGRESSIVE VFPU FASTLANE 2026-08-18 +>"%PERF_V86_RADIO_VFPU_CT2_STAMP%" echo VCS PERF V8.6 RADIO IDENTITY VFPU CT2 2026-08-18 echo. echo [2b/7] Building tests and DX12 probes... diff --git a/profiles/vcs/tests/check_v827_save_thread_lifecycle.py b/profiles/vcs/tests/check_v827_save_thread_lifecycle.py index e8000c7..8d7f94e 100644 --- a/profiles/vcs/tests/check_v827_save_thread_lifecycle.py +++ b/profiles/vcs/tests/check_v827_save_thread_lifecycle.py @@ -20,9 +20,11 @@ def need(cond, msg): need(('stage=correctness-v8.2.7-save-thread-lifecycle-fix-2026-08-18' in log) or ('stage=correctness-v8.2.7a-internal-save-repro-gate-2026-08-18' in log) or - ('stage=perf-v8.4-aggressive-cpu-direct-2026-08-18' in log), + ('stage=perf-v8.4-aggressive-cpu-direct-2026-08-18' in log) or + ('stage=perf-v8.5-aggressive-vfpu-fastlane-2026-08-18' in log) or + ('stage=perf-v8.6-radio-identity-vfpu-ct2-2026-08-18' in log), 'V8.2.7 correctness lineage runtime stage') -need(('correctness_revision=827 ' in log) or ('correctness_revision=8271 ' in log), 'V8.2.7/V8.2.7A correctness revision') +need(('correctness_revision=827 ' in log) or ('correctness_revision=8271 ' in log) or ('correctness_revision=8272 ' in log), 'V8.2.7/V8.2.7A correctness revision') need('save_exitdelete_semantics=1' in log and 'save_repro_legacy_exitdelete_repair=1' in log and 'save_partition_reuse=1' in log, 'ExitDelete/partition fix and migration metadata') need('atrac_stream_resident_status=1' in log and 'atrac_nonloop_resident=-2' in log and diff --git a/profiles/vcs/tests/check_v827a_internal_save_repro_gate.py b/profiles/vcs/tests/check_v827a_internal_save_repro_gate.py index 7ddb756..a58da96 100644 --- a/profiles/vcs/tests/check_v827a_internal_save_repro_gate.py +++ b/profiles/vcs/tests/check_v827a_internal_save_repro_gate.py @@ -20,9 +20,11 @@ def need(cond, msg): print('PASS:', msg) need(('stage=correctness-v8.2.7a-internal-save-repro-gate-2026-08-18' in log) or - ('stage=perf-v8.4-aggressive-cpu-direct-2026-08-18' in log), + ('stage=perf-v8.4-aggressive-cpu-direct-2026-08-18' in log) or + ('stage=perf-v8.5-aggressive-vfpu-fastlane-2026-08-18' in log) or + ('stage=perf-v8.6-radio-identity-vfpu-ct2-2026-08-18' in log), 'V8.2.7A correctness lineage runtime stage') -need('correctness_revision=8271' in log, 'V8.2.7A correctness revision') +need(('correctness_revision=8271' in log) or ('correctness_revision=8272' in log), 'V8.2.7A correctness revision') need('save_repro_internal_ini_gate=1' in log and 'save_repro_default_enabled=0' in log, 'internal-only gate metadata') need('struct TestingConfiguration' in config_h and 'bool save_repro{false};' in config_h, diff --git a/profiles/vcs/tests/check_v84_aggressive_cpu.py b/profiles/vcs/tests/check_v84_aggressive_cpu.py index 063b639..a4e19df 100644 --- a/profiles/vcs/tests/check_v84_aggressive_cpu.py +++ b/profiles/vcs/tests/check_v84_aggressive_cpu.py @@ -18,10 +18,10 @@ def need(c,m): print('FAIL:',m); raise SystemExit(1) print('PASS:',m) -need('stage=perf-v8.4-aggressive-cpu-direct-2026-08-18' in log, 'V8.4 stage') -need('cpu_aggressive_revision=4' in log and 'aot_hard_fastmem=1' in log and 'aot_branchless_mem=1' in log, 'aggressive CPU metadata') +need(('stage=perf-v8.4-aggressive-cpu-direct-2026-08-18' in log) or ('stage=perf-v8.5-aggressive-vfpu-fastlane-2026-08-18' in log) or ('stage=perf-v8.6-radio-identity-vfpu-ct2-2026-08-18' in log), 'V8.4 lineage stage') +need((('cpu_aggressive_revision=4' in log) or ('cpu_aggressive_revision=5' in log) or ('cpu_aggressive_revision=6' in log)) and 'aot_hard_fastmem=1' in log and 'aot_branchless_mem=1' in log, 'aggressive CPU metadata') need('aot_vfpu_block32=19455' in log and 'aot_scalar_run_blocks=10665' in log and 'aot_scalar_run_words=53173' in log and 'aot_load_runs=5633' in log and 'aot_store_runs=5032' in log and 'vfpu_default_dest_fast=1' in log and 'vfpu_vmscl_default_fast=1' in log, 'vector/VFPU and scalar-run metadata') -need('correctness_revision=8271' in log and 'save_exitdelete_semantics=1' in log, 'V8.2.7 Save fix preserved') +need((('correctness_revision=8271' in log) or ('correctness_revision=8272' in log)) and 'save_exitdelete_semantics=1' in log, 'V8.2.7 Save fix preserved') need('atrac_nonloop_resident=-2' in log and 'atrac_loop_resident=-3' in log and 'output2_late_catchup=0' in log, 'V8.2.5 NEWS fix preserved') need('hot_blocks=1060' in log and 'static_fused_calls=36' in log and 'geometry_fusion_rollback=1' in log, 'protected V8.2 Tier2 shape preserved') need('requires Win64 direct fastmem' in main and 'if (!runtime.memory().direct_fastmem_enabled())' in main, 'launch fails cleanly without required fastmem') diff --git a/profiles/vcs/tests/check_v85_aggressive_vfpu.py b/profiles/vcs/tests/check_v85_aggressive_vfpu.py new file mode 100644 index 0000000..e8520d1 --- /dev/null +++ b/profiles/vcs/tests/check_v85_aggressive_vfpu.py @@ -0,0 +1,72 @@ +#!/usr/bin/env python3 +from pathlib import Path +import sys + +root = Path(__file__).resolve().parents[3] +profile = root / 'profiles' / 'vcs' + +def read(rel): + return (root / rel).read_text(encoding='utf-8', errors='ignore') + +def need(cond, msg): + if not cond: + print(f'FAIL: {msg}') + raise SystemExit(1) + print(f'PASS: {msg}') + +log = read('profiles/vcs/host/vcs_runtime_log.cpp') +header = read('include/psprecomp/allegrex_context.hpp') +codegen = read('tools/codegen_main.cpp') +build = read('profiles/vcs/scripts/build_release_ninja.bat') +opt = read('profiles/vcs/tools/optimize_generated_v85_vfpu.py') +ini = read('profiles/vcs/config/VCSNative.ini') + +need(('stage=perf-v8.5-aggressive-vfpu-fastlane-2026-08-18' in log) or ('stage=perf-v8.6-radio-identity-vfpu-ct2-2026-08-18' in log), 'V8.5 lineage stage stamp') +need((('perf_layer=11' in log and 'cpu_aggressive_revision=5' in log) or ('perf_layer=12' in log and 'cpu_aggressive_revision=6' in log)), 'V8.5/V8.6 performance revision metadata') +for token in ( + 'aot_hard_fastmem=1', 'aot_branchless_mem=1', 'aot_vfpu_block32=19455', + 'aot_scalar_run_blocks=10665', 'aot_scalar_run_words=53173', + 'vfpu_default_fastlane=1', 'vfpu_vec3_fastlane_sites=2469', + 'vfpu_unary_fastlane_sites=1607', 'vfpu_vdot_default_fast=1', + 'vfpu_vscl_default_fast=1', 'atrac_nonloop_resident=-2', + 'atrac_loop_resident=-3', 'output2_late_catchup=0', 'save_exitdelete_semantics=1', + 'save_partition_reuse=1', 'save_repro_default_enabled=0'): + need(token in log, f'protected/runtime flag {token}') + +need('execute_vfpu_vec3_ct' in header, 'generic vec3 fast-lane helper exists') +need('execute_vfpu_unary_ct' in header, 'generic unary fast-lane helper exists') +need('vfpu_ctrl[0] == 0xE4u' in header and 'vfpu_ctrl[1] == 0xE4u' in header and 'vfpu_ctrl[2] == 0u' in header, + 'fast-lane guards require architectural default S/T/D prefixes') +need('execute_vfpu_vdot_ct' in header and 'execute_vfpu_vscl_ct' in header, 'VDOT/VSCL helpers retained') +need('ctx.execute_vfpu_vec3_ct<' in codegen, 'future codegen emits vec3 helper') +need('ctx.execute_vfpu_unary_ct<' in codegen, 'future codegen emits unary helper') +need('optimize_generated_v85_vfpu.py' in build, 'Windows build applies V8.5 corpus optimizer') +need('check_v85_aggressive_vfpu.py' in build, 'Windows build runs V8.5 auditor') +need('PERF_V85_VFPU_FASTLANE_STAMP' in build, 'V8.5 one-time rebuild stamp exists') +need('SaveRepro=false' in ini, 'internal SAVE_REPRO remains disabled by default') +need('vec3' in opt and 'unary' in opt, 'V8.5 optimizer contains vec3/unary transforms') + +files = list((profile / 'generated').glob('generated_unit_*.cpp')) +need(len(files) >= 220, 'complete generated AOT corpus present') +counts = {k:0 for k in ('vec3','unary','vdot','vscl','old_source','old_target')} +for f in files: + t=f.read_text(encoding='utf-8', errors='ignore') + counts['vec3'] += t.count('execute_vfpu_vec3_ct<') + counts['unary'] += t.count('execute_vfpu_unary_ct<') + counts['vdot'] += t.count('execute_vfpu_vdot_ct<') + counts['vscl'] += t.count('execute_vfpu_vscl_ct<') + counts['old_source'] += t.count('std::array source_values') + counts['old_target'] += t.count('std::array target_values') +need(counts['vec3'] == 2469, f"vec3 sites exact ({counts['vec3']})") +need(counts['unary'] == 1607, f"unary sites exact ({counts['unary']})") +need(counts['vdot'] == 1456, f"VDOT sites retained ({counts['vdot']})") +need(counts['vscl'] == 1668, f"VSCL sites retained ({counts['vscl']})") +need(counts['old_source'] == 0 and counts['old_target'] == 0, 'old generic vec3 temporary-array form eliminated') + +# Protect the lean Tier-2 shape which was restored for V8.4. +need('clusters=7' in log and 'hot_blocks=1060' in log and 'static_fused_calls=36' in log and 'static_fused_tail=13' in log, + 'protected V8.2.7A/V8.4 Tier-2 shape retained') +need('ge_async_default=0' in log and 'parallel_vertex_decode_default=0' in log, + 'unstable async/parallel paths remain quarantined') + +print('V8.5 aggressive VFPU fast-lane auditor: PASS') diff --git a/profiles/vcs/tests/check_v86_radio_vfpu_ct2.py b/profiles/vcs/tests/check_v86_radio_vfpu_ct2.py new file mode 100644 index 0000000..2ae2faa --- /dev/null +++ b/profiles/vcs/tests/check_v86_radio_vfpu_ct2.py @@ -0,0 +1,46 @@ +#!/usr/bin/env python3 +from pathlib import Path +import sys +root=Path(__file__).resolve().parents[3] +profile=root/'profiles'/'vcs' +def read(rel): return (root/rel).read_text(encoding='utf-8',errors='ignore') +def need(c,m): + if not c: print('FAIL:',m); raise SystemExit(1) + print('PASS:',m) +log=read('profiles/vcs/host/vcs_runtime_log.cpp') +source=read('profiles/vcs/host/vcs_profile.cpp') +header=read('include/psprecomp/allegrex_context.hpp') +gen=read('tools/codegen_main.cpp') +vgen=read('profiles/vcs/tools/vcs_codegen_main.cpp') +build=read('profiles/vcs/scripts/build_release_ninja.bat') +ini=read('profiles/vcs/config/VCSNative.ini') +need('stage=perf-v8.6-radio-identity-vfpu-ct2-2026-08-18' in log,'V8.6 stage') +need('perf_layer=12' in log and 'cpu_aggressive_revision=6' in log and 'correctness_revision=8272' in log,'V8.6 revisions') +for t in ('radio_atrac_identity_guard=1','atrac_direct_header_validation=1','atrac_identity_cache=1','atrac_header_only_producer_hint=1','vfpu_vtfm_ct=1','vfpu_vtfm_ct_sites=337','vfpu_vi2f_ct=1','vfpu_vi2f_ct_sites=164','vfpu_vcmp_default_fast=1','vfpu_vcmov_default_fast=1','vfpu_cross_default_fast=1'): + need(t in log,'metadata '+t) +for t in ('atrac_nonloop_resident=-2','atrac_loop_resident=-3','news_atrac_v825_guard=1','output2_late_catchup=0','save_exitdelete_semantics=1','save_partition_reuse=1','save_repro_default_enabled=0','aot_hard_fastmem=1','aot_branchless_mem=1','vfpu_default_fastlane=1'): + need(t in log,'protected '+t) +need('AtracSourceIdentity' in source and 'atrac_source_matches_header' in source,'ATRAC identity cache/validator') +need('direct_rejected' in source and 'fallback_candidates' in source,'stale direct source can be rejected') +need('read_absolute == file_start && read >= 12u' in source,'virtual UMD producer hints are header-only') +need('read_start == std::streampos(0) && read >= 12u' in source,'normal file producer hints are header-only') +need('ATRAC source identity trusted a stale radio staging-buffer hint' in source,'radio mismatch self-test exists') +need('kAtracRemainNonLoopOnMemory = 0xFFFFFFFEu' in source and 'kAtracRemainLoopOnMemory = 0xFFFFFFFDu' in source,'NEWS ATRAC remain-frame sentinels remain in source') +need('execute_vfpu_vtfm_ct' in header and 'execute_vfpu_vi2f_ct' in header,'new VFPU CT helpers exist') +need('execute_vfpu_vcmp_ct' in header and 'execute_vfpu_vcmov_ct' in header and 'execute_vfpu_cross_quat_ct' in header,'existing CT helpers extended') +need('ctx.execute_vfpu_vtfm_ct<' in gen and 'ctx.execute_vfpu_vi2f_ct<' in gen,'generic PSP codegen emits VTFM/VI2F CT') +need('ctx.execute_vfpu_vtfm_ct<' in vgen and 'ctx.execute_vfpu_vi2f_ct<' in vgen,'VCS codegen emits VTFM/VI2F CT') +need('execute_vfpu_cross_quat_ct' in gen and 'execute_vfpu_compare3_ct' in gen and 'execute_vfpu_vh2f_ct' in gen,'generic lowering covers remaining literal helpers') +files=sorted((profile/'generated').glob('generated_unit_*.cpp')) +need(len(files)==234,'234 generated units') +text=''.join(p.read_text(encoding='utf-8',errors='ignore') for p in files) +checks={'vtfm':('execute_vfpu_vtfm_ct<',337),'vi2f':('execute_vfpu_vi2f_ct<',164),'vcmp':('execute_vfpu_vcmp_ct<',558),'vcmov':('execute_vfpu_vcmov_ct<',461),'cross':('execute_vfpu_cross_quat_ct<',217)} +for name,(pat,want) in checks.items(): need(text.count(pat)==want,f'{name} CT sites={want} (got {text.count(pat)})') +need('constexpr std::uint32_t vfpu_side =' not in text,'old VTFM inline body eliminated') +need('const float vfpu_scale = std::ldexp(1.0f, -static_cast' not in text,'old VI2F inline body eliminated') +need('ctx.execute_vfpu_cross_quat(' not in text and 'ctx.execute_vfpu_compare3(' not in text and 'ctx.execute_vfpu_vh2f(' not in text,'remaining literal helper calls lowered') +need('clusters=7' in log and 'hot_blocks=1060' in log and 'static_fused_calls=36' in log and 'static_fused_tail=13' in log,'protected lean Tier2 shape') +need('ge_async_default=0' in log and 'parallel_vertex_decode_default=0' in log,'unstable async paths quarantined') +need('SaveRepro=false' in ini,'internal SAVE_REPRO default off') +need('PERF_V86_RADIO_VFPU_CT2_STAMP' in build and 'check_v86_radio_vfpu_ct2.py' in build,'Windows build V8.6 gate') +print('V8.6 radio identity + VFPU CT2 auditor: PASS') diff --git a/profiles/vcs/tests/vfpu_tier2_tests.cpp b/profiles/vcs/tests/vfpu_tier2_tests.cpp index c4cfce4..7a30ae7 100644 --- a/profiles/vcs/tests/vfpu_tier2_tests.cpp +++ b/profiles/vcs/tests/vfpu_tier2_tests.cpp @@ -25,7 +25,11 @@ static bool same(const AllegrexContext&a,const AllegrexContext&b){ [[noreturn]] static void fail(const char*n){std::cerr<<"DIFF "< void test_matrix_read(){for(int n=0;n<8;++n){auto a=make_ctx();float x[16]{},y[16]{};a.read_vfpu_matrix(x,M,S);a.read_vfpu_matrix_ct(y);for(int i=0;i<16;++i)if(std::bit_cast(x[i])!=std::bit_cast(y[i]))fail("matrix_read");}} template void test_matrix_write(){for(int n=0;n<8;++n){auto a=make_ctx(),b=a;float x[16]{};std::uniform_real_distributionf(-50,50);for(float&v:x)v=f(rng);a.write_vfpu_matrix(x,M,S);b.write_vfpu_matrix_ct(x);if(!same(a,b))fail("matrix_write");}} -template void test_cross(){for(int n=0;n<8;++n){auto a=make_ctx(),b=a;a.execute_vfpu_cross_quat(D,S,T,L);b.execute_vfpu_cross_quat_ct();if(!same(a,b))fail("cross");}} +template void test_cross(){for(int n=0;n<32;++n){auto a=make_ctx(),b=a;if((n&1)==0){a.vfpu_ctrl[0]=b.vfpu_ctrl[0]=0xE4u;a.vfpu_ctrl[1]=b.vfpu_ctrl[1]=0xE4u;a.vfpu_ctrl[2]=b.vfpu_ctrl[2]=0u;}a.execute_vfpu_cross_quat(D,S,T,L);b.execute_vfpu_cross_quat_ct();if(!same(a,b))fail("cross");}} +template void test_vcmp(){for(int n=0;n<64;++n){auto a=make_ctx(),b=a;if((n&1)==0){a.vfpu_ctrl[0]=b.vfpu_ctrl[0]=0xE4u;a.vfpu_ctrl[1]=b.vfpu_ctrl[1]=0xE4u;a.vfpu_ctrl[2]=b.vfpu_ctrl[2]=0u;}a.execute_vfpu_vcmp(S,T,L,C);b.execute_vfpu_vcmp_ct();if(!same(a,b))fail("vcmp");}} +template void test_vcmov(){for(int n=0;n<64;++n){auto a=make_ctx(),b=a;if((n&1)==0){a.vfpu_ctrl[0]=b.vfpu_ctrl[0]=0xE4u;a.vfpu_ctrl[1]=b.vfpu_ctrl[1]=0xE4u;a.vfpu_ctrl[2]=b.vfpu_ctrl[2]=0u;}a.execute_vfpu_vcmov(D,S,L,C,F);b.execute_vfpu_vcmov_ct();if(!same(a,b))fail("vcmov");}} +template void test_vi2f(){for(int n=0;n<64;++n){auto a=make_ctx(),b=a;if((n&1)==0){a.vfpu_ctrl[0]=b.vfpu_ctrl[0]=0xE4u;a.vfpu_ctrl[1]=b.vfpu_ctrl[1]=0xE4u;a.vfpu_ctrl[2]=b.vfpu_ctrl[2]=0u;}float src[4]{},dst[4]{};a.read_vfpu_vector(src,S,L);a.apply_vfpu_source_prefix(src,L,0u);const float scale=std::ldexp(1.0f,-static_cast(I));for(unsigned lane=0;lane(std::bit_cast(src[lane]));dst[lane]=static_cast(integer)*scale;}a.write_vfpu_vector_with_destination_prefix(dst,D,L);b.execute_vfpu_vi2f_ct();if(!same(a,b))fail("vi2f");}} +template void test_vtfm(){for(int n=0;n<64;++n){auto a=make_ctx(),b=a;if((n&1)==0){a.vfpu_ctrl[0]=b.vfpu_ctrl[0]=0xE4u;a.vfpu_ctrl[1]=b.vfpu_ctrl[1]=0xE4u;a.vfpu_ctrl[2]=b.vfpu_ctrl[2]=0u;}float matrix[16]{},raw[4]{},target[4]{},result[4]{};a.read_vfpu_matrix(matrix,M,Side);a.read_vfpu_vector(raw,T,Side);for(unsigned i=0;i<4;++i)target[i]=i=In)target[Side-1u]=1.0f;for(unsigned row=0;row+1u();if(!same(a,b))fail("vtfm");}} template void test_vh2f(){for(int n=0;n<8;++n){auto a=make_ctx(),b=a;a.execute_vfpu_vh2f(D,S,L);b.execute_vfpu_vh2f_ct();if(!same(a,b))fail("vh2f");}} template void test_vhdp(){for(int n=0;n<8;++n){auto a=make_ctx(),b=a;a.execute_vfpu_vhdp(D,S,T,L);b.execute_vfpu_vhdp_ct();if(!same(a,b))fail("vhdp");}} template void test_vx2i(){for(int n=0;n<8;++n){auto a=make_ctx(),b=a;a.execute_vfpu_vx2i(D,S,L,O);b.execute_vfpu_vx2i_ct();if(!same(a,b))fail("vx2i");}} @@ -33,6 +37,33 @@ template void test_minmax(){ template void test_cmp3(){for(int n=0;n<8;++n){auto a=make_ctx(),b=a;a.execute_vfpu_compare3(D,S,T,L,O);b.execute_vfpu_compare3_ct();if(!same(a,b))fail("cmp3");}} template void test_vrot(){for(int n=0;n<8;++n){auto a=make_ctx(),b=a;a.execute_vfpu_vrot(D,S,L,I);b.execute_vfpu_vrot_ct();if(!same(a,b))fail("vrot");}} + +template void test_vec3_fastlane(){ + for(int n=0;n<64;++n){ + auto a=make_ctx(),b=a; + if((n&1)==0){a.vfpu_ctrl[0]=b.vfpu_ctrl[0]=0xE4u;a.vfpu_ctrl[1]=b.vfpu_ctrl[1]=0xE4u;a.vfpu_ctrl[2]=b.vfpu_ctrl[2]=0u;} + float sv[4]{},tv[4]{},dv[4]{}; + a.read_vfpu_vector_with_source_prefix(sv,S,L,0u); + a.read_vfpu_vector_with_source_prefix(tv,T,L,1u); + for(unsigned i=0;i(); + if(!same(a,b))fail("vec3_fastlane"); + } +} +template void test_unary_fastlane(){ + for(int n=0;n<32;++n){ + auto a=make_ctx(),b=a; + if((n&1)==0){a.vfpu_ctrl[0]=b.vfpu_ctrl[0]=0xE4u;a.vfpu_ctrl[2]=b.vfpu_ctrl[2]=0u;} + float sv[4]{},dv[4]{};a.read_vfpu_vector_with_source_prefix(sv,S,L,0u); + for(unsigned i=0;i(sv[i]); + a.write_vfpu_vector_with_destination_prefix(dv,D,L); + b.execute_vfpu_unary_ct(); + if(!same(a,b))fail("unary_fastlane"); + } +} +template void test_vdot_fastlane(){for(int n=0;n<64;++n){auto a=make_ctx(),b=a;if((n&1)==0){a.vfpu_ctrl[0]=b.vfpu_ctrl[0]=0xE4u;a.vfpu_ctrl[1]=b.vfpu_ctrl[1]=0xE4u;a.vfpu_ctrl[2]=b.vfpu_ctrl[2]=0u;}a.execute_vfpu_vdot(D,S,T,L);b.execute_vfpu_vdot_ct();if(!same(a,b))fail("vdot_fastlane");}} +template void test_vscl_fastlane(){for(int n=0;n<64;++n){auto a=make_ctx(),b=a;if((n&1)==0){a.vfpu_ctrl[0]=b.vfpu_ctrl[0]=0xE4u;a.vfpu_ctrl[1]=b.vfpu_ctrl[1]=0xE4u;a.vfpu_ctrl[2]=b.vfpu_ctrl[2]=0u;}a.execute_vfpu_vscl(D,S,T,L);b.execute_vfpu_vscl_ct();if(!same(a,b))fail("vscl_fastlane");}} template void test_dest_prefix(){for(int n=0;n<64;++n){auto a=make_ctx(),b=a;float x[4]{};std::uniform_real_distributionf(-16,16);for(float&v:x)v=f(rng);a.write_vfpu_vector_with_destination_prefix(x,D,L);b.write_vfpu_vector_with_destination_prefix_ct(x);if(!same(a,b))fail("dest_prefix");}} template void test_vf2h(){for(int n=0;n<32;++n){auto a=make_ctx(),b=a;a.execute_vfpu_vf2h(D,S,L);b.execute_vfpu_vf2h_ct();if(!same(a,b))fail("vf2h");}} template void test_horizontal(){for(int n=0;n<32;++n){auto a=make_ctx(),b=a;a.execute_vfpu_horizontal(D,S,L,A);b.execute_vfpu_horizontal_ct();if(!same(a,b))fail("horizontal");}} @@ -126,5 +157,27 @@ int main(){ test_vmmov<32,40,4>(); test_vmmov<24,0,4>(); test_vmscl<32,36,8,3>(); + test_vcmp<14,0,4,3>(); + test_vcmp<1,2,3,9>(); + test_vcmov<66,98,1,0,false>(); + test_vcmov<14,11,3,6,true>(); + test_vi2f<2,1,2,3>(); + test_vi2f<66,32,1,0>(); + test_vi2f<12,36,4,16>(); + test_vtfm<14,36,12,4,3>(); + test_vtfm<0,36,1,3,3>(); + test_vtfm<32,40,8,4,4>(); + test_vec3_fastlane<0,1,2,3,0>(); + test_vec3_fastlane<32,36,40,4,1>(); + test_vec3_fastlane<14,11,13,3,2>(); + test_vec3_fastlane<3,4,5,2,3>(); + test_unary_fastlane<0,1,3,0>(); + test_unary_fastlane<32,36,4,2>(); + test_unary_fastlane<14,11,3,17>(); + test_unary_fastlane<5,9,2,22>(); + test_vdot_fastlane<2,0,48,4>(); + test_vdot_fastlane<34,1,52,3>(); + test_vscl_fastlane<32,36,8,3>(); + test_vscl_fastlane<0,1,2,4>(); std::cout<<"vfpu_tier2_tests: PASS\n"; } diff --git a/profiles/vcs/tools/optimize_generated_v85_vfpu.py b/profiles/vcs/tools/optimize_generated_v85_vfpu.py new file mode 100644 index 0000000..91870c9 --- /dev/null +++ b/profiles/vcs/tools/optimize_generated_v85_vfpu.py @@ -0,0 +1,76 @@ +#!/usr/bin/env python3 +from __future__ import annotations +import argparse, pathlib, re + +BINARY = re.compile( + r'\{ float vfpu_s\[4\]\{\}, vfpu_t\[4\]\{\}, vfpu_d\[4\]\{\};\s*' + r'ctx\.read_vfpu_vector_with_source_prefix_ct<(?P\d+)u, (?P[1-4])u, 0u>\(vfpu_s\);\s*' + r'ctx\.read_vfpu_vector_with_source_prefix_ct<(?P\d+)u, (?P=len)u, 1u>\(vfpu_t\);\s*' + r'for \(std::uint32_t i = 0; i < (?P=len)u; \+\+i\) vfpu_d\[i\] = (?Pvfpu_s\[i\] [\+\-\*/] vfpu_t\[i\]);\s*' + r'ctx\.write_vfpu_vector_with_destination_prefix_ct<(?P\d+)u, (?P=len)u>\(vfpu_d\); \}', re.S) + +UNARY = re.compile( + r'\{ float vfpu_s\[4\]\{\}, vfpu_d\[4\]\{\};\s*' + r'ctx\.read_vfpu_vector_with_source_prefix_ct<(?P\d+)u, (?P[1-4])u, 0u>\(vfpu_s\);\s*' + r'for \(std::uint32_t i = 0; i < (?P=len)u; \+\+i\) vfpu_d\[i\] = (?P[^;]+);\s*' + r'ctx\.write_vfpu_vector_with_destination_prefix_ct<(?P\d+)u, (?P=len)u>\(vfpu_d\); \}', re.S) + +BINARY_OP = { + 'vfpu_s[i] + vfpu_t[i]': 0, + 'vfpu_s[i] - vfpu_t[i]': 1, + 'vfpu_s[i] * vfpu_t[i]': 2, + 'vfpu_s[i] / vfpu_t[i]': 3, +} +UNARY_OP = { + 'vfpu_s[i]': 0, + 'std::fabs(vfpu_s[i])': 1, + '-vfpu_s[i]': 2, + 'vfpu_s[i] <= 0.0f ? 0.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i])': 4, + 'vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i])': 5, + '1.0f / vfpu_s[i]': 16, + '1.0f / std::sqrt(vfpu_s[i])': 17, + 'std::sin(vfpu_s[i] * 1.57079632679489661923f)': 18, + 'std::cos(vfpu_s[i] * 1.57079632679489661923f)': 19, + 'std::exp2(vfpu_s[i])': 20, + 'std::log2(vfpu_s[i])': 21, + 'std::fabs(std::sqrt(vfpu_s[i]))': 22, + 'std::asin(vfpu_s[i]) * 0.63661977236758134308f': 23, + '-1.0f / vfpu_s[i]': 24, + '-std::sin(vfpu_s[i] * 1.57079632679489661923f)': 26, + '1.0f / std::exp2(vfpu_s[i])': 31, +} + +def optimize(text: str): + stats={'vec3':0,'unary':0} + def rb(m): + op=BINARY_OP.get(m.group('expr')) + if op is None: return m.group(0) + stats['vec3'] += 1 + return (f'ctx.execute_vfpu_vec3_ct<{m.group("d")}u, {m.group("s")}u, ' + f'{m.group("t")}u, {m.group("len")}u, {op}u>();') + text=BINARY.sub(rb,text) + def ru(m): + expr=m.group('expr').strip() + op=UNARY_OP.get(expr) + if op is None: return m.group(0) + stats['unary'] += 1 + return (f'ctx.execute_vfpu_unary_ct<{m.group("d")}u, {m.group("s")}u, ' + f'{m.group("len")}u, {op}u>();') + text=UNARY.sub(ru,text) + return text,stats + +def main(): + ap=argparse.ArgumentParser(); ap.add_argument('profile',type=pathlib.Path); a=ap.parse_args() + totals={'files':0,'changed':0,'new_vec3':0,'new_unary':0,'vec3':0,'unary':0} + for p in sorted((a.profile/'generated').glob('generated_unit_*.cpp')): + old=p.read_text(encoding='utf-8'); new,st=optimize(old) + totals['files']+=1; totals['new_vec3']+=st['vec3']; totals['new_unary']+=st['unary'] + totals['vec3'] += new.count('execute_vfpu_vec3_ct<') + totals['unary'] += new.count('execute_vfpu_unary_ct<') + if new!=old: + p.write_text(new,encoding='utf-8',newline='\n'); totals['changed']+=1 + print('V8.5 VFPU fast-lane optimizer: ' + ' '.join(f'{k}={v}' for k,v in totals.items())) + if totals['vec3'] < 2000 or totals['unary'] < 1000: + raise SystemExit('V8.5 VFPU optimizer coverage unexpectedly low') + return 0 +if __name__=='__main__': raise SystemExit(main()) diff --git a/profiles/vcs/tools/vcs_codegen_main.cpp b/profiles/vcs/tools/vcs_codegen_main.cpp index cd956af..82895c7 100644 --- a/profiles/vcs/tools/vcs_codegen_main.cpp +++ b/profiles/vcs/tools/vcs_codegen_main.cpp @@ -413,15 +413,8 @@ std::string emit_regular(const psprecomp::DecodedInstruction &d, std::uint32_t p const std::uint32_t destination = d.word & 0x7Fu; const std::uint32_t source = (d.word >> 8u) & 0x7Fu; const std::uint32_t immediate = (d.word >> 16u) & 31u; - out << " { float vfpu_s[4]{}, vfpu_d[4]{};\n" - << " ctx.read_vfpu_vector(vfpu_s, " << source << "u, " << length << "u);\n" - << " ctx.apply_vfpu_source_prefix(vfpu_s, " << length << "u, 0u);\n" - << " const float vfpu_scale = std::ldexp(1.0f, -static_cast(" << immediate << "u));\n" - << " for (std::uint32_t vfpu_i = 0; vfpu_i < " << length << "u; ++vfpu_i) {\n" - << " const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i]));\n" - << " vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale;\n" - << " }\n" - << " ctx.write_vfpu_vector_with_destination_prefix(vfpu_d, " << destination << "u, " << length << "u); }\n"; + out << " ctx.execute_vfpu_vi2f_ct<" << destination << "u, " << source << "u, " + << length << "u, " << immediate << "u>();\n"; break; } case psprecomp::OpcodeKind::Vx2i: { @@ -482,28 +475,8 @@ std::string emit_regular(const psprecomp::DecodedInstruction &d, std::uint32_t p const std::uint32_t destination = d.word & 0x7Fu; const std::uint32_t source = (d.word >> 8u) & 0x7Fu; const std::uint32_t target = (d.word >> 16u) & 0x7Fu; - out << " { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{};\n" - << " ctx.read_vfpu_matrix(vfpu_matrix, " << source << "u, " << side << "u);\n" - << " ctx.read_vfpu_vector(vfpu_target_raw, " << target << "u, " << side << "u);\n" - << " constexpr std::uint32_t vfpu_side = " << side << "u;\n" - << " constexpr std::uint32_t vfpu_input_length = " << input_length << "u;\n" - << " for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f;\n" - << " if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f;\n" - << " for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) {\n" - << " float sum = 0.0f;\n" - << " for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column];\n" - << " vfpu_result[row] = sum;\n" - << " }\n" - << " float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u],\n" - << " vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]};\n" - << " ctx.apply_vfpu_source_prefix(vfpu_final_row, 4u, 0u);\n" - << " ctx.apply_vfpu_source_prefix(vfpu_target, 4u, 1u);\n" - << " for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column];\n" - << " const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2];\n" - << " const std::uint32_t vfpu_last_lane = vfpu_side - 1u;\n" - << " ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) |\n" - << " ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u));\n" - << " ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, " << destination << "u, vfpu_side); }\n"; + out << " ctx.execute_vfpu_vtfm_ct<" << destination << "u, " << source << "u, " + << target << "u, " << side << "u, " << input_length << "u>();\n"; break; } case psprecomp::OpcodeKind::VfpuVectorInit: { @@ -1539,6 +1512,71 @@ std::string lower_constant_vfpu_accesses(std::string text) { text = std::regex_replace(text, vscl, "ctx.execute_vfpu_vscl_ct<$1u, $2u, $3u, $4u>();"); + const std::regex vf2h( + R"(ctx\.execute_vfpu_vf2h\(([0-9]+)u, ([0-9]+)u, ([1-4])u\);)" ); + text = std::regex_replace(text, vf2h, + "ctx.execute_vfpu_vf2h_ct<$1u, $2u, $3u>();"); + + const std::regex vh2f( + R"(ctx\.execute_vfpu_vh2f\(([0-9]+)u, ([0-9]+)u, ([1-4])u\);)" ); + text = std::regex_replace(text, vh2f, + "ctx.execute_vfpu_vh2f_ct<$1u, $2u, $3u>();"); + + const std::regex vx2i( + R"(ctx\.execute_vfpu_vx2i\(([0-9]+)u, ([0-9]+)u, ([1-4])u, ([0-3])u\);)" ); + text = std::regex_replace(text, vx2i, + "ctx.execute_vfpu_vx2i_ct<$1u, $2u, $3u, $4u>();"); + + const std::regex vhdp( + R"(ctx\.execute_vfpu_vhdp\(([0-9]+)u, ([0-9]+)u, ([0-9]+)u, ([1-4])u\);)" ); + text = std::regex_replace(text, vhdp, + "ctx.execute_vfpu_vhdp_ct<$1u, $2u, $3u, $4u>();"); + + const std::regex cross_quat( + R"(ctx\.execute_vfpu_cross_quat\(([0-9]+)u, ([0-9]+)u, ([0-9]+)u, ([1-4])u\);)" ); + text = std::regex_replace(text, cross_quat, + "ctx.execute_vfpu_cross_quat_ct<$1u, $2u, $3u, $4u>();"); + + const std::regex vminmax( + R"(ctx\.execute_vfpu_vminmax\(([0-9]+)u, ([0-9]+)u, ([0-9]+)u, ([1-4])u, (true|false)\);)" ); + text = std::regex_replace(text, vminmax, + "ctx.execute_vfpu_vminmax_ct<$1u, $2u, $3u, $4u, $5>();"); + + const std::regex compare3( + R"(ctx\.execute_vfpu_compare3\(([0-9]+)u, ([0-9]+)u, ([0-9]+)u, ([1-4])u, ([0-9]+)u\);)" ); + text = std::regex_replace(text, compare3, + "ctx.execute_vfpu_compare3_ct<$1u, $2u, $3u, $4u, $5u>();"); + + const std::regex vrot( + R"(ctx\.execute_vfpu_vrot\(([0-9]+)u, ([0-9]+)u, ([1-4])u, ([0-9]+)u\);)" ); + text = std::regex_replace(text, vrot, + "ctx.execute_vfpu_vrot_ct<$1u, $2u, $3u, $4u>();"); + + const std::regex vocp( + R"(ctx\.execute_vfpu_vocp\(([0-9]+)u, ([0-9]+)u, ([1-4])u\);)" ); + text = std::regex_replace(text, vocp, + "ctx.execute_vfpu_vocp_ct<$1u, $2u, $3u>();"); + + const std::regex horizontal( + R"(ctx\.execute_vfpu_horizontal\(([0-9]+)u, ([0-9]+)u, ([1-4])u, (true|false)\);)" ); + text = std::regex_replace(text, horizontal, + "ctx.execute_vfpu_horizontal_ct<$1u, $2u, $3u, $4>();"); + + const std::regex vmscl( + R"(ctx\.execute_vfpu_vmscl\(([0-9]+)u, ([0-9]+)u, ([0-9]+)u, ([1-4])u\);)" ); + text = std::regex_replace(text, vmscl, + "ctx.execute_vfpu_vmscl_ct<$1u, $2u, $3u, $4u>();"); + + const std::regex vmmov( + R"(ctx\.execute_vfpu_vmmov\(([0-9]+)u, ([0-9]+)u, ([1-4])u\);)" ); + text = std::regex_replace(text, vmmov, + "ctx.execute_vfpu_vmmov_ct<$1u, $2u, $3u>();"); + + const std::regex matrix_init( + R"(ctx\.execute_vfpu_matrix_init\(([0-9]+)u, ([1-4])u, ([367])u\);)" ); + text = std::regex_replace(text, matrix_init, + "ctx.execute_vfpu_matrix_init_ct<$1u, $2u, $3u>();"); + const std::regex scalar_read(R"(ctx\.vfpu_scalar_bits\(([0-9]+)u\))"); text = std::regex_replace(text, scalar_read, "ctx.vfpu_scalar_bits_ct<$1u>()"); diff --git a/tests/test_main.cpp b/tests/test_main.cpp index 6f1a84e..8ce5aa6 100644 --- a/tests/test_main.cpp +++ b/tests/test_main.cpp @@ -605,9 +605,9 @@ static void test_codegen_vh2f_lowering() { } require(text.find("ctx.execute_vfpu_vh2f(0u, 0u, 1u)") != std::string::npos, "VH2F was not lowered to the dedicated AOT helper"); - require(text.find("const float vfpu_scale = std::ldexp(1.0f, -static_cast(3u))") != std::string::npos, + require(text.find("ctx.execute_vfpu_vi2f_ct<2u, 1u, 2u, 3u>()") != std::string::npos, "VI2F scale was not lowered into generated AOT code"); - require(text.find("static_cast(std::bit_cast(vfpu_s[vfpu_i]))") != std::string::npos, + require(text.find("execute_vfpu_vi2f_ct") != std::string::npos, "VI2F did not reinterpret source lanes as signed integers"); require(text.find("vh2f not lowered yet") == std::string::npos, "VH2F codegen retained an unsupported fallback"); diff --git a/tools/codegen_main.cpp b/tools/codegen_main.cpp index a59798a..2d71bab 100644 --- a/tools/codegen_main.cpp +++ b/tools/codegen_main.cpp @@ -412,15 +412,8 @@ std::string emit_regular(const psprecomp::DecodedInstruction &d, std::uint32_t p const std::uint32_t destination = d.word & 0x7Fu; const std::uint32_t source = (d.word >> 8u) & 0x7Fu; const std::uint32_t immediate = (d.word >> 16u) & 31u; - out << " { float vfpu_s[4]{}, vfpu_d[4]{};\n" - << " ctx.read_vfpu_vector(vfpu_s, " << source << "u, " << length << "u);\n" - << " ctx.apply_vfpu_source_prefix(vfpu_s, " << length << "u, 0u);\n" - << " const float vfpu_scale = std::ldexp(1.0f, -static_cast(" << immediate << "u));\n" - << " for (std::uint32_t vfpu_i = 0; vfpu_i < " << length << "u; ++vfpu_i) {\n" - << " const auto vfpu_integer = static_cast(std::bit_cast(vfpu_s[vfpu_i]));\n" - << " vfpu_d[vfpu_i] = static_cast(vfpu_integer) * vfpu_scale;\n" - << " }\n" - << " ctx.write_vfpu_vector_with_destination_prefix(vfpu_d, " << destination << "u, " << length << "u); }\n"; + out << " ctx.execute_vfpu_vi2f_ct<" << destination << "u, " << source << "u, " + << length << "u, " << immediate << "u>();\n"; break; } case psprecomp::OpcodeKind::Vx2i: { @@ -481,28 +474,8 @@ std::string emit_regular(const psprecomp::DecodedInstruction &d, std::uint32_t p const std::uint32_t destination = d.word & 0x7Fu; const std::uint32_t source = (d.word >> 8u) & 0x7Fu; const std::uint32_t target = (d.word >> 16u) & 0x7Fu; - out << " { float vfpu_matrix[16]{}, vfpu_target_raw[4]{}, vfpu_target[4]{}, vfpu_result[4]{};\n" - << " ctx.read_vfpu_matrix(vfpu_matrix, " << source << "u, " << side << "u);\n" - << " ctx.read_vfpu_vector(vfpu_target_raw, " << target << "u, " << side << "u);\n" - << " constexpr std::uint32_t vfpu_side = " << side << "u;\n" - << " constexpr std::uint32_t vfpu_input_length = " << input_length << "u;\n" - << " for (std::uint32_t i = 0; i < 4u; ++i) vfpu_target[i] = i < vfpu_input_length ? vfpu_target_raw[i] : 0.0f;\n" - << " if (vfpu_side - 1u >= vfpu_input_length) vfpu_target[vfpu_side - 1u] = 1.0f;\n" - << " for (std::uint32_t row = 0; row + 1u < vfpu_side; ++row) {\n" - << " float sum = 0.0f;\n" - << " for (std::uint32_t column = 0; column < vfpu_side; ++column) sum += vfpu_matrix[row * 4u + column] * vfpu_target[column];\n" - << " vfpu_result[row] = sum;\n" - << " }\n" - << " float vfpu_final_row[4]{vfpu_matrix[(vfpu_side - 1u) * 4u + 0u], vfpu_matrix[(vfpu_side - 1u) * 4u + 1u],\n" - << " vfpu_matrix[(vfpu_side - 1u) * 4u + 2u], vfpu_matrix[(vfpu_side - 1u) * 4u + 3u]};\n" - << " ctx.apply_vfpu_source_prefix(vfpu_final_row, 4u, 0u);\n" - << " ctx.apply_vfpu_source_prefix(vfpu_target, 4u, 1u);\n" - << " for (std::uint32_t column = 0; column < 4u; ++column) vfpu_result[vfpu_side - 1u] += vfpu_final_row[column] * vfpu_target[column];\n" - << " const std::uint32_t vfpu_destination_prefix = ctx.vfpu_ctrl[2];\n" - << " const std::uint32_t vfpu_last_lane = vfpu_side - 1u;\n" - << " ctx.vfpu_ctrl[2] = ((vfpu_destination_prefix & (1u << 8u)) << vfpu_last_lane) |\n" - << " ((vfpu_destination_prefix & 3u) << (vfpu_last_lane * 2u));\n" - << " ctx.write_vfpu_vector_with_destination_prefix(vfpu_result, " << destination << "u, vfpu_side); }\n"; + out << " ctx.execute_vfpu_vtfm_ct<" << destination << "u, " << source << "u, " + << target << "u, " << side << "u, " << input_length << "u>();\n"; break; } case psprecomp::OpcodeKind::VfpuVectorInit: { @@ -582,15 +555,8 @@ std::string emit_regular(const psprecomp::DecodedInstruction &d, std::uint32_t p const std::uint32_t target = (d.word >> 16u) & 0x7Fu; const std::uint32_t major = d.word >> 26u; const std::uint32_t operation = major == 0x19u ? 2u : ((d.word >> 23u) & 7u); - const char *expression = operation == 0u ? "vfpu_s[i] + vfpu_t[i]" - : operation == 1u ? "vfpu_s[i] - vfpu_t[i]" - : operation == 2u ? "vfpu_s[i] * vfpu_t[i]" - : "vfpu_s[i] / vfpu_t[i]"; - out << " { float vfpu_s[4]{}, vfpu_t[4]{}, vfpu_d[4]{};\n" - << " ctx.read_vfpu_vector_with_source_prefix(vfpu_s, " << source << "u, " << length << "u, 0u);\n" - << " ctx.read_vfpu_vector_with_source_prefix(vfpu_t, " << target << "u, " << length << "u, 1u);\n" - << " for (std::uint32_t i = 0; i < " << length << "u; ++i) vfpu_d[i] = " << expression << ";\n" - << " ctx.write_vfpu_vector_with_destination_prefix(vfpu_d, " << destination << "u, " << length << "u); }\n"; + out << " ctx.execute_vfpu_vec3_ct<" << destination << "u, " << source << "u, " + << target << "u, " << length << "u, " << operation << "u>();\n"; break; } case psprecomp::OpcodeKind::Vdot: { @@ -683,29 +649,8 @@ std::string emit_regular(const psprecomp::DecodedInstruction &d, std::uint32_t p const std::uint32_t destination = d.word & 0x7Fu; const std::uint32_t source = (d.word >> 8u) & 0x7Fu; const std::uint32_t operation = (d.word >> 16u) & 31u; - std::string expression; - switch (operation) { - case 0u: expression = "vfpu_s[i]"; break; - case 1u: expression = "std::fabs(vfpu_s[i])"; break; - case 2u: expression = "-vfpu_s[i]"; break; - case 4u: expression = "vfpu_s[i] <= 0.0f ? 0.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i])"; break; - case 5u: expression = "vfpu_s[i] < -1.0f ? -1.0f : (vfpu_s[i] > 1.0f ? 1.0f : vfpu_s[i])"; break; - case 16u: expression = "1.0f / vfpu_s[i]"; break; - case 17u: expression = "1.0f / std::sqrt(vfpu_s[i])"; break; - case 18u: expression = "std::sin(vfpu_s[i] * 1.57079632679489661923f)"; break; - case 19u: expression = "std::cos(vfpu_s[i] * 1.57079632679489661923f)"; break; - case 20u: expression = "std::exp2(vfpu_s[i])"; break; - case 21u: expression = "std::log2(vfpu_s[i])"; break; - case 22u: expression = "std::fabs(std::sqrt(vfpu_s[i]))"; break; - case 23u: expression = "std::asin(vfpu_s[i]) * 0.63661977236758134308f"; break; - case 24u: expression = "-1.0f / vfpu_s[i]"; break; - case 26u: expression = "-std::sin(vfpu_s[i] * 1.57079632679489661923f)"; break; - default: expression = "1.0f / std::exp2(vfpu_s[i])"; break; - } - out << " { float vfpu_s[4]{}, vfpu_d[4]{};\n" - << " ctx.read_vfpu_vector_with_source_prefix(vfpu_s, " << source << "u, " << length << "u, 0u);\n" - << " for (std::uint32_t i = 0; i < " << length << "u; ++i) vfpu_d[i] = " << expression << ";\n" - << " ctx.write_vfpu_vector_with_destination_prefix(vfpu_d, " << destination << "u, " << length << "u); }\n"; + out << " ctx.execute_vfpu_unary_ct<" << destination << "u, " << source << "u, " + << length << "u, " << operation << "u>();\n"; break; } case psprecomp::OpcodeKind::Vcst: { @@ -1406,6 +1351,71 @@ std::string lower_constant_vfpu_accesses(std::string text) { text = std::regex_replace(text, vscl, "ctx.execute_vfpu_vscl_ct<$1u, $2u, $3u, $4u>();"); + const std::regex vf2h( + R"(ctx\.execute_vfpu_vf2h\(([0-9]+)u, ([0-9]+)u, ([1-4])u\);)" ); + text = std::regex_replace(text, vf2h, + "ctx.execute_vfpu_vf2h_ct<$1u, $2u, $3u>();"); + + const std::regex vh2f( + R"(ctx\.execute_vfpu_vh2f\(([0-9]+)u, ([0-9]+)u, ([1-4])u\);)" ); + text = std::regex_replace(text, vh2f, + "ctx.execute_vfpu_vh2f_ct<$1u, $2u, $3u>();"); + + const std::regex vx2i( + R"(ctx\.execute_vfpu_vx2i\(([0-9]+)u, ([0-9]+)u, ([1-4])u, ([0-3])u\);)" ); + text = std::regex_replace(text, vx2i, + "ctx.execute_vfpu_vx2i_ct<$1u, $2u, $3u, $4u>();"); + + const std::regex vhdp( + R"(ctx\.execute_vfpu_vhdp\(([0-9]+)u, ([0-9]+)u, ([0-9]+)u, ([1-4])u\);)" ); + text = std::regex_replace(text, vhdp, + "ctx.execute_vfpu_vhdp_ct<$1u, $2u, $3u, $4u>();"); + + const std::regex cross_quat( + R"(ctx\.execute_vfpu_cross_quat\(([0-9]+)u, ([0-9]+)u, ([0-9]+)u, ([1-4])u\);)" ); + text = std::regex_replace(text, cross_quat, + "ctx.execute_vfpu_cross_quat_ct<$1u, $2u, $3u, $4u>();"); + + const std::regex vminmax( + R"(ctx\.execute_vfpu_vminmax\(([0-9]+)u, ([0-9]+)u, ([0-9]+)u, ([1-4])u, (true|false)\);)" ); + text = std::regex_replace(text, vminmax, + "ctx.execute_vfpu_vminmax_ct<$1u, $2u, $3u, $4u, $5>();"); + + const std::regex compare3( + R"(ctx\.execute_vfpu_compare3\(([0-9]+)u, ([0-9]+)u, ([0-9]+)u, ([1-4])u, ([0-9]+)u\);)" ); + text = std::regex_replace(text, compare3, + "ctx.execute_vfpu_compare3_ct<$1u, $2u, $3u, $4u, $5u>();"); + + const std::regex vrot( + R"(ctx\.execute_vfpu_vrot\(([0-9]+)u, ([0-9]+)u, ([1-4])u, ([0-9]+)u\);)" ); + text = std::regex_replace(text, vrot, + "ctx.execute_vfpu_vrot_ct<$1u, $2u, $3u, $4u>();"); + + const std::regex vocp( + R"(ctx\.execute_vfpu_vocp\(([0-9]+)u, ([0-9]+)u, ([1-4])u\);)" ); + text = std::regex_replace(text, vocp, + "ctx.execute_vfpu_vocp_ct<$1u, $2u, $3u>();"); + + const std::regex horizontal( + R"(ctx\.execute_vfpu_horizontal\(([0-9]+)u, ([0-9]+)u, ([1-4])u, (true|false)\);)" ); + text = std::regex_replace(text, horizontal, + "ctx.execute_vfpu_horizontal_ct<$1u, $2u, $3u, $4>();"); + + const std::regex vmscl( + R"(ctx\.execute_vfpu_vmscl\(([0-9]+)u, ([0-9]+)u, ([0-9]+)u, ([1-4])u\);)" ); + text = std::regex_replace(text, vmscl, + "ctx.execute_vfpu_vmscl_ct<$1u, $2u, $3u, $4u>();"); + + const std::regex vmmov( + R"(ctx\.execute_vfpu_vmmov\(([0-9]+)u, ([0-9]+)u, ([1-4])u\);)" ); + text = std::regex_replace(text, vmmov, + "ctx.execute_vfpu_vmmov_ct<$1u, $2u, $3u>();"); + + const std::regex matrix_init( + R"(ctx\.execute_vfpu_matrix_init\(([0-9]+)u, ([1-4])u, ([367])u\);)" ); + text = std::regex_replace(text, matrix_init, + "ctx.execute_vfpu_matrix_init_ct<$1u, $2u, $3u>();"); + const std::regex scalar_read(R"(ctx\.vfpu_scalar_bits\(([0-9]+)u\))"); text = std::regex_replace(text, scalar_read, "ctx.vfpu_scalar_bits_ct<$1u>()");