Compare commits

..

4 Commits

Author SHA1 Message Date
martipops bd219e38a3 fix: drive compile progress from ninja's [done/total] counter (#285)
The compile slice advanced a fixed step per line of build output, so how
far it moved depended on how many lines the build printed. When aurora and
its dependencies are compiled locally (as on macOS) the bar reached its cap
long before the game itself was compiled. Read ninja's edge counter instead.
2026-10-09 22:52:17 +02:00
therenow9 196b48528e Fix display lists dropping draws when the staging batch fills (#288)
* test: cover display lists that overflow the staging batch

GXCallDisplayList and GXCallDisplayListLE decode a list with a single
fifo::process() call. When a draw does not fit the staging batch,
process() stops at that draw, and the rest of the list is never drawn.

The test stubs can now admit N draws, refuse the next one as a full
batch would, and count the split that follows. The new test refuses
the middle or the last of three draws, in big- and little-endian
lists, and expects one split and the last draw's vertices.

Also stub aurora::window::set_force_aspect_16_9, called since #256,
so gx_fifo_tests links again.

Fails without the next commit.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

* fix: submit staging and continue when a display list overflows it

drain() already loops process() -> split_staging_batch() -> retry.
Move that loop into fifo::process_all() and use it for drain() and
both GXCallDisplayList variants, so a display list that fills the
staging batch submits it and draws the rest instead of dropping it.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

---------

Co-authored-by: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
2026-10-09 22:48:24 +02:00
Michael G 54c608ef9c Add experimental x86-64-v2 CPU support for older PCs and Intel Macs (#258)
* runtime: add experimental x86-64-v2 profile

* fix review findings in network and translator

* build: include mbedtls in macos compile audit

* review fixes

* cmake: restore macOS deployment target floor before project()

* fix: Preserve explicit package versions and gate v3 tests on host support

* fix: Prevent TLS session reuse after failed writes

* fix: Gate cross-compiled profile tests on emulator availability

---------

Co-authored-by: patchzyy <64382339+patchzyy@users.noreply.github.com>
2026-10-09 22:46:11 +02:00
patchzyy f3df63b529 Update README.md 2026-10-09 10:51:22 +02:00
29 changed files with 632 additions and 233 deletions
+23
View File
@@ -47,6 +47,29 @@ jobs:
- name: Test
run: dotnet test translator/Translator.sln -c Release --no-build --verbosity normal
platform_tests:
name: Host platform (${{ matrix.runner }})
strategy:
fail-fast: false
matrix:
runner: [ubuntu-24.04, macos-14, macos-15-intel]
runs-on: ${{ matrix.runner }}
steps:
- uses: actions/checkout@v7
with:
persist-credentials: false
- name: Configure without graphics dependencies or game assets
run: >-
cmake -S runtime -B build-platform -G Ninja
-DCMAKE_BUILD_TYPE=Release -DCMAKE_C_COMPILER=clang -DCMAKE_CXX_COMPILER=clang++
-DMKW_BUILD_PRODUCTS=OFF -DMKW_PLATFORM_TESTS_ONLY=ON
- name: Build and run host-platform tests
run: |
cmake --build build-platform --parallel 3
ctest --test-dir build-platform --output-on-failure
macos_substrate:
name: macOS arm64 (configure + substrate tests)
runs-on: macos-14
+13 -5
View File
@@ -11,6 +11,11 @@ on:
tags:
- '*'
workflow_dispatch:
inputs:
version:
description: Package version
required: true
type: string
permissions:
contents: read
@@ -163,14 +168,17 @@ jobs:
- name: Build Setup.pkg
env:
PACKAGE_VERSION: ${{ inputs.version }}
TAG_VERSION: ${{ github.ref_name }}
shell: bash
run: |
package_version=""
if [[ "${TAG_VERSION:-}" =~ ^v?[0-9] ]]; then
package_version="${TAG_VERSION#v}"
elif [[ -f "Launcher/Directory.Build.props" ]]; then
package_version=$(grep -m1 '<Version>' Launcher/Directory.Build.props | sed -E 's/.*<Version>([^<]+)<\/Version>.*/\1/')
package_version="$PACKAGE_VERSION"
if [[ -z "$package_version" ]]; then
if [[ "${TAG_VERSION:-}" =~ ^v?[0-9] ]]; then
package_version="${TAG_VERSION#v}"
elif [[ -f "Launcher/Directory.Build.props" ]]; then
package_version=$(grep -m1 '<Version>' Launcher/Directory.Build.props | sed -E 's/.*<Version>([^<]+)<\/Version>.*/\1/')
fi
fi
mkdir -p Launcher/dist
Launcher/macos/build-setup-pkg.command \
+14
View File
@@ -45,6 +45,20 @@ jobs:
shell: pwsh
run: ./Launcher/Prepare-PortableTools.ps1
- name: Build and run Windows host-platform tests
shell: pwsh
run: |
$tools = Join-Path $PWD 'Launcher/artifacts/portable-tools'
$env:PATH = "$tools/llvm-mingw/bin;$tools/CMake/bin;$tools/Ninja;$env:PATH"
cmake -S runtime -B build-platform -G Ninja `
-DCMAKE_BUILD_TYPE=Release -DCMAKE_C_COMPILER=clang -DCMAKE_CXX_COMPILER=clang++ `
-DMKW_BUILD_PRODUCTS=OFF -DMKW_PLATFORM_TESTS_ONLY=ON
if ($LASTEXITCODE -ne 0) { exit $LASTEXITCODE }
cmake --build build-platform --parallel 3
if ($LASTEXITCODE -ne 0) { exit $LASTEXITCODE }
ctest --test-dir build-platform --output-on-failure
if ($LASTEXITCODE -ne 0) { exit $LASTEXITCODE }
- name: Prepare pinned native dependencies
shell: pwsh
run: ./Launcher/Prepare-Dependencies.ps1
@@ -184,18 +184,31 @@ internal sealed class BuildProgressWindow
}
// Anything else - a plain MKWCBUILD note, or raw tool output - stays a diagnostic and only
// feeds the heartbeat below.
// feeds the compile progress below.
_reporter.Diagnostic(line);
// Compilation announces itself once and then emits thousands of compiler lines. Treat that
// output as a heartbeat so the slice keeps creeping forward, but only publish a progress
// line when the rounded percentage actually changes.
if (_fraction >= CompileFraction)
// Compilation announces itself once and then ninja prefixes every edge with [done/total].
// Follow that counter rather than counting lines: how many lines a build prints depends on
// whether aurora and its dependencies are prebuilt (AppImage) or compiled here (macOS), so a
// fixed per-line step reached the top long before the game itself compiled.
if (_fraction >= CompileFraction && TryParseNinjaProgress(line, out var done, out var total))
{
_fraction = Math.Min(0.97, _fraction + 0.0015);
_fraction = Math.Max(_fraction, CompileFraction + (0.97 - CompileFraction) * done / total);
Emit();
}
}
private static bool TryParseNinjaProgress(string line, out int done, out int total)
{
done = total = 0;
if (!line.StartsWith('[')) return false;
var close = line.IndexOf(']');
var slash = line.IndexOf('/');
return close > 0 && slash > 0 && slash < close
&& int.TryParse(line.AsSpan(1, slash - 1), out done)
&& int.TryParse(line.AsSpan(slash + 1, close - slash - 1), out total)
&& total > 0 && done <= total;
}
private void Emit()
{
var percent = Interpolate(_fraction);
+4 -3
View File
@@ -20,13 +20,14 @@ resulting pkg is unsigned unless
Developer ID Installer certificate.
--workspace DIR Repository root (default: script's grandparent)
--version VERSION Bundle/package version (default: 0.1.0)
--version VERSION Bundle/package version (default: project version, or 0.1.0)
--installer-identity NAME Developer ID Installer identity for productbuild
EOF
}
script_dir=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)
workspace=$(cd "$script_dir/../.." && pwd); nodtool_arm64=""; nodtool_x86_64=""; translator_arm64=""; translator_x86_64=""; cmake_root=""; ninja_arm64=""; ninja_x86_64=""; output=""; version=0.1.0; identity=""
version_supplied=false
while (($#)); do
case "$1" in
--workspace) workspace=${2:-}; shift 2 ;;
@@ -38,14 +39,14 @@ while (($#)); do
--ninja-arm64) ninja_arm64=${2:-}; shift 2 ;;
--ninja-x86_64) ninja_x86_64=${2:-}; shift 2 ;;
--output) output=${2:-}; shift 2 ;;
--version) version=${2:-}; shift 2 ;;
--version) version=${2:-}; version_supplied=true; shift 2 ;;
--installer-identity) identity=${2:-}; shift 2 ;;
-h|--help) usage; exit 0 ;;
*) fail "unknown option: $1" ;;
esac
done
version=${version#v}
if [[ -z "$version" || "$version" == "0.1.0" ]]; then
if [[ "$version_supplied" == false ]]; then
local_csproj="$workspace/Launcher/Directory.Build.props"
if [[ -f "$local_csproj" ]]; then
detected=$(grep -m1 '<Version>' "$local_csproj" | sed -E 's/.*<Version>([^<]+)<\/Version>.*/\1/' || true)
+1
View File
@@ -11,6 +11,7 @@
<a href="#building-from-source"><img alt="PowerPC static recompilation" src="https://img.shields.io/badge/PowerPC-static%20recompilation-FF9F0A"></a>
<a href="#retro-rewind"><img alt="Retro Rewind supported" src="https://img.shields.io/badge/Retro%20Rewind-supported-FF375F"></a>
<a href="https://github.com/TeamWheelWizard/WheelWizard/releases"><img alt="Install with Wheel Wizard" src="https://img.shields.io/badge/install%20with-Wheel%20Wizard-8B5CF6"></a>
<a href="https://github.com/patchzyy/Wiicompiled/releases"><img alt="Total downloads" src="https://img.shields.io/github/downloads/patchzyy/Wiicompiled/total?color=00B8D9&amp;logo=github"></a>
<a href="LICENSE"><img alt="License: GPLv3" src="https://img.shields.io/badge/license-GPLv3-2EA44F?logo=gnu&amp;logoColor=white"></a>
</p>
+2 -2
View File
@@ -66,7 +66,7 @@ void GXCallDisplayList(const void* data, u32 nbytes) {
// Decode the display list immediately while its borrowed resources are valid.
aurora::gx::fifo::drain();
aurora::gx::fifo::process(static_cast<const u8*>(data), nbytes, true);
aurora::gx::fifo::process_all(static_cast<const u8*>(data), nbytes, true);
}
void GXCallDisplayListLE(const void* data, u32 nbytes) {
@@ -84,6 +84,6 @@ void GXCallDisplayListLE(const void* data, u32 nbytes) {
aurora::gx::fifo::drain();
// Process the display list through the command processor (little-endian)
aurora::gx::fifo::process(static_cast<const u8*>(data), nbytes, false);
aurora::gx::fifo::process_all(static_cast<const u8*>(data), nbytes, false);
}
}
+1 -1
View File
@@ -553,7 +553,7 @@ void build_index() noexcept {
continue;
}
s_replacementIndex.try_emplace(*parsed, path);
s_replacementIndex.try_emplace(*parsed, ReplacementIndexEntry{path});
}
Log.info("Indexed {} texture replacements", s_replacementIndex.size());
+8 -4
View File
@@ -82,19 +82,23 @@ void drain() {
if (detail::sBufferSize == 0) {
return;
}
process_all(detail::sBufferData, detail::sBufferSize, true);
detail::sBufferSize = 0;
}
void process_all(const uint8_t* data, uint32_t size, bool bigEndian) {
uint32_t consumed = 0;
bool retried = false;
while (consumed < detail::sBufferSize) {
const auto count = process(detail::sBufferData + consumed, detail::sBufferSize - consumed, true);
while (consumed < size) {
const auto count = process(data + consumed, size - consumed, bigEndian);
if (count == 0 && retried)
throw gfx::StagingCapacityError("FIFO draw does not fit after capacity submission");
consumed += count;
if (consumed == detail::sBufferSize) break;
if (consumed == size) break;
// process returned with its renderer lock released. No recursive drain.
gfx::split_staging_batch();
retried = true;
}
detail::sBufferSize = 0;
}
const uint8_t* get_buffer_data() { return detail::sBufferData; }
+4
View File
@@ -81,6 +81,10 @@ bool in_display_list();
// Drain the internal FIFO buffer through the command processor
void drain();
// Decode a whole command stream, submitting the staging batch whenever a draw does not fit.
// process() alone stops at that draw, so a caller that ignores its count drops the rest.
void process_all(const uint8_t* data, uint32_t size, bool bigEndian);
// Internal buffer inspection
const uint8_t* get_buffer_data();
uint32_t get_buffer_size();
+4
View File
@@ -50,6 +50,9 @@ void reset_uniform_allocations() noexcept;
const std::vector<uint8_t>& uniform_allocation(size_t index) noexcept;
void use_draw_command_tracking(bool enabled) noexcept;
void use_real_vertex_format_helpers(bool enabled) noexcept;
// Admit this many draws, then refuse one as if the staging batch were full.
void refuse_staging_admission_after(uint32_t admissions) noexcept;
void reset_staging_capacity() noexcept;
} // namespace aurora::gfx::testing
class GXFifoTest : public ::testing::Test {
@@ -65,6 +68,7 @@ protected:
aurora::gfx::testing::reset_resolve_pass_records();
aurora::gfx::testing::reset_vertex_push_record();
aurora::gfx::testing::use_real_vertex_format_helpers(false);
aurora::gfx::testing::reset_staging_capacity();
}
// Copy the internal FIFO buffer contents and clear it
+30 -2
View File
@@ -34,6 +34,9 @@ std::optional<aurora::gx::DrawData> s_lastGxDraw;
bool s_trackDrawCommands = false;
bool s_useRealVertexFormatHelpers = false;
std::deque<std::vector<uint8_t>> s_uniformAllocations;
std::optional<uint32_t> s_stagingAdmissionsBeforeRefusal;
bool s_stagingRefused = false;
uint64_t s_stagingSplitCount = 0;
} // namespace
// --- aurora::g_config ---
@@ -301,8 +304,23 @@ std::pair<ByteBuffer, Range> copy_uniform(Range source) {
uint32_t align_uniform(uint32_t value) { return (value + 255u) & ~255u; }
uint64_t staging_uniform_bytes(uint64_t value) { return staging_padded(value, 256); }
uint64_t staging_storage_bytes(uint64_t value) { return staging_padded(value, 256); }
bool staging_has_space(const StagingSizes&) { return true; }
void split_staging_batch() { throw StagingCapacityError("Unexpected split in FIFO unit test"); }
bool staging_has_space(const StagingSizes&) {
if (!s_stagingAdmissionsBeforeRefusal) return true;
if (*s_stagingAdmissionsBeforeRefusal > 0) {
--*s_stagingAdmissionsBeforeRefusal;
return true;
}
// Refuse once, as a full batch would; the split that follows makes room again.
s_stagingAdmissionsBeforeRefusal.reset();
s_stagingRefused = true;
return false;
}
void split_staging_batch() {
if (!s_stagingRefused) throw StagingCapacityError("Unexpected split in FIFO unit test");
s_stagingRefused = false;
++s_stagingSplitCount;
}
uint64_t staging_split_count() noexcept { return s_stagingSplitCount; }
Vec2<uint32_t> get_render_target_size() noexcept { return s_renderTargetSize; }
Vec2<uint32_t> get_frame_buffer_size() noexcept { return s_renderTargetSize; }
@@ -345,6 +363,15 @@ void use_draw_command_tracking(bool enabled) noexcept {
void use_real_vertex_format_helpers(bool enabled) noexcept {
s_useRealVertexFormatHelpers = enabled;
}
void refuse_staging_admission_after(uint32_t admissions) noexcept {
s_stagingAdmissionsBeforeRefusal = admissions;
s_stagingRefused = false;
}
void reset_staging_capacity() noexcept {
s_stagingAdmissionsBeforeRefusal.reset();
s_stagingRefused = false;
s_stagingSplitCount = 0;
}
} // namespace aurora::gfx::testing
// --- Pipeline/draw command stubs ---
@@ -525,6 +552,7 @@ std::optional<TextureHandle> find_replacement(const GXTexObj_&) noexcept { retur
namespace aurora::window {
AuroraWindowSize get_window_size() { return {640, 480, 640, 480, 640, 480, 1.0f}; }
void set_frame_buffer_aspect_fit(bool) {}
void set_force_aspect_16_9(bool) {}
} // namespace aurora::window
// --- WebGPU C API stubs (prevent linker errors from wgpu:: destructors) ---
@@ -88,6 +88,31 @@ TEST_F(GXFifoTest, SingleExpandedPrimitiveCannotMergeWithTriangles) {
EXPECT_EQ(aurora::gfx::testing::last_pushed_indices(), (std::vector<u16>{0, 1, 2}));
}
TEST_F(GXFifoTest, DisplayListSubmitsStagingWhenADrawDoesNotFit) {
__GXSetDirtyState();
aurora::gx::fifo::clear_buffer();
g_gxState.lastVtxFmt = GX_VTXFMT0;
g_gxState.lastVtxSize = 1;
// admitted=1 refuses the middle draw; admitted=2 refuses the last, so only a replay of the refused draw pushes 0x33.
for (const u32 admitted : {1u, 2u}) for (const bool bigEndian : {true, false}) {
std::vector<u8> list;
for (const u8 fill : {0x11, 0x22, 0x33}) {
auto bytes = draw(GX_TRIANGLES, 3);
if (!bigEndian) std::swap(bytes[1], bytes[2]);
std::fill(bytes.begin() + 3, bytes.end(), fill);
list.insert(list.end(), bytes.begin(), bytes.end());
}
const auto splitsBefore = aurora::gfx::staging_split_count();
aurora::gfx::testing::refuse_staging_admission_after(admitted);
if (bigEndian) GXCallDisplayList(list.data(), static_cast<u32>(list.size()));
else GXCallDisplayListLE(list.data(), static_cast<u32>(list.size()));
EXPECT_EQ(aurora::gfx::staging_split_count() - splitsBefore, 1u)
<< "admitted=" << admitted << " bigEndian=" << bigEndian;
EXPECT_EQ(aurora::gfx::testing::last_pushed_vertices(), (std::vector<u8>{0x33, 0x33, 0x33}))
<< "admitted=" << admitted << " bigEndian=" << bigEndian;
}
}
TEST(StagingMapping, RetiredCallbacksCannotPublishAnotherBuffersReadiness) {
using namespace aurora::gfx;
StagingMapState state;
+24 -146
View File
@@ -48,6 +48,24 @@ endif()
option(MKW_BUILD_PRODUCTS "Build translated WiiCompiled product targets" ON)
option(MKW_BUILD_PSQ_TESTS "Build focused PSQ ISA tests" OFF)
set(MKW_X86_CPU_PROFILE "v3" CACHE STRING
"Minimum x86-64 CPU profile for translated products (v2 or v3)")
set_property(CACHE MKW_X86_CPU_PROFILE PROPERTY STRINGS v2 v3)
if(CMAKE_SYSTEM_PROCESSOR MATCHES "^(AMD64|amd64|x86_64|X86_64)$" AND
NOT MKW_X86_CPU_PROFILE MATCHES "^(v2|v3)$")
message(FATAL_ERROR "MKW_X86_CPU_PROFILE must be v2 or v3 (got '${MKW_X86_CPU_PROFILE}')")
endif()
option(MKW_PLATFORM_TESTS_ONLY "Build host-platform tests without graphics dependencies or game assets" OFF)
include("${CMAKE_CURRENT_LIST_DIR}/cmake/HostLibraries.cmake")
if(MKW_PLATFORM_TESTS_ONLY)
if(MKW_BUILD_PRODUCTS)
message(FATAL_ERROR "MKW_PLATFORM_TESTS_ONLY requires MKW_BUILD_PRODUCTS=OFF")
endif()
add_compile_definitions(NOMINMAX)
include("${CMAKE_CURRENT_LIST_DIR}/cmake/PlatformTests.cmake")
return()
endif()
# Preprocessor definitions that belong to this project's own code (the runtime,
# the translated shards and the product glue) and to nothing else. They are
@@ -85,21 +103,6 @@ target_include_directories(mkw_pugixml PUBLIC third_party/pugixml)
target_compile_features(mkw_pugixml PUBLIC cxx_std_17)
set_target_properties(mkw_pugixml PROPERTIES UNITY_BUILD OFF)
# POSIX x86-64 guest-fiber scheduling (runtime/src/host_context.cpp) needs a symmetric
# stackful-coroutine primitive to stand in for Win32 Fibers. libco's co_switch() transfers
# directly to any other created coroutine, matching SwitchToFiber's semantics exactly (unlike
# asymmetric resume/yield coroutine libraries, which would need every call site restructured).
# Vendored from upstream (higan-emu/libco @ e18e09d, 2019-10-16, ISC license; valgrind.h is
# separately BSD-style licensed, see third_party/libco/LICENSE). Windows keeps native Fibers;
# Apple Silicon uses the project's x18-safe AArch64 assembly backend, while Intel macOS uses
# libco's existing System V AMD64 backend.
if(MKW_PLATFORM_LINUX OR MKW_PLATFORM_MACOS_X86_64)
add_library(mkw_libco STATIC third_party/libco/libco.c)
add_library(mkw::libco ALIAS mkw_libco)
target_include_directories(mkw_libco PUBLIC third_party/libco)
set_target_properties(mkw_libco PROPERTIES UNITY_BUILD OFF)
endif()
# Runtime configuration is real TOML, parsed by toml11 rather than a project-
# specific line parser. Keep it header-only and vendored so disconnected release
# builds have exactly the same parser as developer builds.
@@ -341,133 +344,7 @@ set(MKW_CPU_BASELINE_SOURCE "${CMAKE_CURRENT_LIST_DIR}/src/host_cpu_baseline.cpp
list(REMOVE_ITEM SOURCES ${MKW_BASE_PRODUCT_SOURCE} ${MKW_RETRO_REWIND_PRODUCT_SOURCE}
${MKW_CPU_BASELINE_SOURCE} ${MKW_PLATFORM_SOURCE})
# This deliberately small library contains host services that are safe to
# validate before guest memory and fiber work makes a full runtime build viable.
add_library(mkw_platform STATIC "${MKW_PLATFORM_SOURCE}")
target_include_directories(mkw_platform PUBLIC "${CMAKE_CURRENT_LIST_DIR}/include")
target_compile_features(mkw_platform PUBLIC cxx_std_17)
set_target_properties(mkw_platform PROPERTIES UNITY_BUILD OFF)
# Keep these independent from Aurora's BUILD_TESTING option: they validate the
# project's host-platform contracts, not Aurora's third-party test suite.
enable_testing()
if(MKW_BUILD_PSQ_TESTS)
add_executable(mkw_psq_helpers_tests
"${CMAKE_CURRENT_LIST_DIR}/tests/psq_helpers_tests.cpp"
"${CMAKE_CURRENT_LIST_DIR}/src/ppc_quantized.cpp")
target_include_directories(mkw_psq_helpers_tests PRIVATE
"${CMAKE_CURRENT_LIST_DIR}/tests/psq_memory"
"${CMAKE_CURRENT_LIST_DIR}/include/isa"
"${CMAKE_CURRENT_LIST_DIR}/include")
target_compile_features(mkw_psq_helpers_tests PRIVATE cxx_std_17)
target_compile_options(mkw_psq_helpers_tests PRIVATE
-O2 ${MKW_TRANSLATED_PPC_FP_OPTIONS} -fno-slp-vectorize)
if(CMAKE_SYSTEM_PROCESSOR MATCHES "^(AMD64|amd64|x86_64|X86_64)$")
target_compile_options(mkw_psq_helpers_tests PRIVATE -march=x86-64-v3)
endif()
set_target_properties(mkw_psq_helpers_tests PROPERTIES UNITY_BUILD OFF)
add_test(NAME mkw_psq_helpers_tests COMMAND mkw_psq_helpers_tests)
add_test(NAME mkw_psq_reserved_tests COMMAND "${CMAKE_COMMAND}"
"-DPSQ_TEST_EXECUTABLE=$<TARGET_FILE:mkw_psq_helpers_tests>"
-P "${CMAKE_CURRENT_LIST_DIR}/tests/psq_reserved_tests.cmake")
endif()
add_executable(mkw_platform_paths_tests "${CMAKE_CURRENT_LIST_DIR}/tests/platform_paths_tests.cpp")
target_link_libraries(mkw_platform_paths_tests PRIVATE mkw_platform)
target_compile_features(mkw_platform_paths_tests PRIVATE cxx_std_17)
add_test(NAME mkw_platform_paths_tests COMMAND mkw_platform_paths_tests)
add_executable(mkw_runtime_config_tests "${CMAKE_CURRENT_LIST_DIR}/tests/runtime_config_tests.cpp")
target_include_directories(mkw_runtime_config_tests PRIVATE
"${CMAKE_CURRENT_LIST_DIR}/include"
"${CMAKE_CURRENT_LIST_DIR}/third_party/toml11")
target_compile_features(mkw_runtime_config_tests PRIVATE cxx_std_20)
add_test(NAME mkw_runtime_config_tests COMMAND mkw_runtime_config_tests)
add_executable(mkw_nand_save_tests "${CMAKE_CURRENT_LIST_DIR}/tests/nand_save_tests.cpp")
target_include_directories(mkw_nand_save_tests PRIVATE "${CMAKE_CURRENT_LIST_DIR}/include")
target_compile_features(mkw_nand_save_tests PRIVATE cxx_std_17)
add_test(NAME mkw_nand_save_tests COMMAND mkw_nand_save_tests)
add_executable(mkw_nand_settings_tests "${CMAKE_CURRENT_LIST_DIR}/tests/nand_settings_tests.cpp")
find_package(Threads REQUIRED)
target_link_libraries(mkw_nand_settings_tests PRIVATE Threads::Threads)
target_include_directories(mkw_nand_settings_tests PRIVATE "${CMAKE_CURRENT_LIST_DIR}/include")
target_compile_features(mkw_nand_settings_tests PRIVATE cxx_std_17)
add_test(NAME mkw_nand_settings_tests COMMAND mkw_nand_settings_tests)
add_executable(mkw_sc_serial_tests "${CMAKE_CURRENT_LIST_DIR}/tests/sc_serial_tests.cpp")
target_include_directories(mkw_sc_serial_tests PRIVATE "${CMAKE_CURRENT_LIST_DIR}/include")
target_compile_features(mkw_sc_serial_tests PRIVATE cxx_std_17)
add_test(NAME mkw_sc_serial_tests COMMAND mkw_sc_serial_tests)
# The input expression engine is self-contained, so it can be exercised without
# linking the runtime or SDL.
add_executable(mkw_input_expr_tests
"${CMAKE_CURRENT_LIST_DIR}/tests/test_expr.cpp"
"${CMAKE_CURRENT_LIST_DIR}/src/input_expr.cpp")
target_include_directories(mkw_input_expr_tests PRIVATE "${CMAKE_CURRENT_LIST_DIR}/include")
target_compile_features(mkw_input_expr_tests PRIVATE cxx_std_17)
add_test(NAME mkw_input_expr_tests COMMAND mkw_input_expr_tests)
# HostContext deliberately keeps the platform-specific context primitive out
# of fiber_manager.cpp. Exercise the Linux libco handoff directly so future
# refactors cannot silently remove its headers, implementation, or link edge.
if(MKW_PLATFORM_LINUX)
add_executable(mkw_linux_host_context_tests
"${CMAKE_CURRENT_LIST_DIR}/tests/host_context_tests.cpp"
"${CMAKE_CURRENT_LIST_DIR}/src/host_context.cpp")
target_include_directories(mkw_linux_host_context_tests PRIVATE
"${CMAKE_CURRENT_LIST_DIR}/include"
"${CMAKE_CURRENT_LIST_DIR}/third_party/libco")
target_compile_features(mkw_linux_host_context_tests PRIVATE cxx_std_17)
target_link_libraries(mkw_linux_host_context_tests PRIVATE mkw::libco)
add_test(NAME mkw_linux_host_context_tests COMMAND mkw_linux_host_context_tests)
endif()
if(MKW_PLATFORM_MACOS)
# Exercise the public host-memory contracts separately from translated products.
if(MKW_PLATFORM_MACOS_ARM64)
# Apple Silicon's context ABI is implemented by the local assembly backend.
enable_language(ASM)
add_executable(mkw_macos_context_abi_tests
"${CMAKE_CURRENT_LIST_DIR}/tests/macos_context_abi_tests.cpp"
"${CMAKE_CURRENT_LIST_DIR}/src/platform/macos/co_switch.S")
target_compile_features(mkw_macos_context_abi_tests PRIVATE cxx_std_17)
add_test(NAME mkw_macos_context_abi_tests COMMAND mkw_macos_context_abi_tests)
add_executable(mkw_macos_host_context_tests
"${CMAKE_CURRENT_LIST_DIR}/tests/host_context_tests.cpp"
"${CMAKE_CURRENT_LIST_DIR}/src/host_context.cpp"
"${CMAKE_CURRENT_LIST_DIR}/src/platform/macos/co_switch.S")
target_include_directories(mkw_macos_host_context_tests PRIVATE "${CMAKE_CURRENT_LIST_DIR}/include")
else()
# Intel macOS follows the same System V AMD64 libco path as Linux.
add_executable(mkw_macos_host_context_tests
"${CMAKE_CURRENT_LIST_DIR}/tests/host_context_tests.cpp"
"${CMAKE_CURRENT_LIST_DIR}/src/host_context.cpp")
target_include_directories(mkw_macos_host_context_tests PRIVATE
"${CMAKE_CURRENT_LIST_DIR}/include"
"${CMAKE_CURRENT_LIST_DIR}/third_party/libco")
target_link_libraries(mkw_macos_host_context_tests PRIVATE mkw::libco)
endif()
target_compile_features(mkw_macos_host_context_tests PRIVATE cxx_std_17)
add_test(NAME mkw_macos_host_context_tests COMMAND mkw_macos_host_context_tests)
add_executable(mkw_macos_guest_flat_memory_tests
"${CMAKE_CURRENT_LIST_DIR}/tests/macos_guest_flat_memory_tests.cpp"
"${CMAKE_CURRENT_LIST_DIR}/src/guest_flat_memory_macos.cpp")
target_include_directories(mkw_macos_guest_flat_memory_tests PRIVATE "${CMAKE_CURRENT_LIST_DIR}/include")
target_compile_features(mkw_macos_guest_flat_memory_tests PRIVATE cxx_std_17)
add_test(NAME mkw_macos_guest_flat_memory_tests COMMAND mkw_macos_guest_flat_memory_tests)
add_executable(mkw_macos_external_audio_tests
"${CMAKE_CURRENT_LIST_DIR}/tests/macos_external_audio_tests.cpp"
"${CMAKE_CURRENT_LIST_DIR}/src/external_audio_macos.cpp")
target_include_directories(mkw_macos_external_audio_tests PRIVATE "${CMAKE_CURRENT_LIST_DIR}/include")
target_compile_features(mkw_macos_external_audio_tests PRIVATE cxx_std_17)
add_test(NAME mkw_macos_external_audio_tests COMMAND mkw_macos_external_audio_tests)
endif()
include("${CMAKE_CURRENT_LIST_DIR}/cmake/PlatformTests.cmake")
# The translator emits the complete, content-addressed source graph. Consuming
# this one manifest keeps configure independent of the 28k generated function
@@ -518,10 +395,11 @@ else()
mkw::pugixml mkw::toml11 mkw::cryptopp mkw::mbedtls)
if(MKW_PLATFORM_MACOS_X86_64)
target_link_libraries(mkw_macos_native_compile PRIVATE mkw::libco)
# Keep this compile-only audit on the same Haswell-era x86-64-v3
# baseline as the translated product. PPC paired FMA helpers use
# FMA intrinsics and intentionally cannot compile for plain x86-64.
target_compile_options(mkw_macos_native_compile PRIVATE -march=x86-64-v3)
target_compile_options(mkw_macos_native_compile PRIVATE
-march=x86-64-${MKW_X86_CPU_PROFILE})
if(MKW_X86_CPU_PROFILE STREQUAL "v2")
target_compile_definitions(mkw_macos_native_compile PRIVATE MKW_X86_CPU_PROFILE_V2=1)
endif()
endif()
set_target_properties(mkw_macos_native_compile PROPERTIES UNITY_BUILD OFF)
endif()
+29
View File
@@ -0,0 +1,29 @@
# Shared by the product build and the dependency-free platform test configuration.
get_filename_component(MKW_HOST_RUNTIME_DIR "${CMAKE_CURRENT_LIST_DIR}/.." ABSOLUTE)
# POSIX guest-fiber scheduling (runtime/src/host_context.cpp) needs a symmetric
# stackful-coroutine primitive to stand in for Win32 Fibers. libco's co_switch() transfers
# directly to any other created coroutine, matching SwitchToFiber's semantics exactly (unlike
# asymmetric resume/yield coroutine libraries, which would need every call site restructured).
# Vendored from upstream (higan-emu/libco @ e18e09d, 2019-10-16, ISC license; valgrind.h is
# separately BSD-style licensed, see third_party/libco/LICENSE). Windows keeps native Fibers;
# Apple Silicon uses the project's x18-safe AArch64 assembly backend, while Intel macOS uses
# libco's existing System V AMD64 backend.
if(MKW_PLATFORM_LINUX OR MKW_PLATFORM_MACOS_X86_64)
add_library(mkw_libco STATIC "${MKW_HOST_RUNTIME_DIR}/third_party/libco/libco.c")
add_library(mkw::libco ALIAS mkw_libco)
target_include_directories(mkw_libco PUBLIC "${MKW_HOST_RUNTIME_DIR}/third_party/libco")
set_target_properties(mkw_libco PROPERTIES UNITY_BUILD OFF)
endif()
# This deliberately small library contains host services that are safe to
# validate before guest memory and fiber work makes a full runtime build viable.
add_library(mkw_platform STATIC "${MKW_HOST_RUNTIME_DIR}/src/platform/host_platform.cpp")
target_include_directories(mkw_platform PUBLIC "${MKW_HOST_RUNTIME_DIR}/include")
target_compile_features(mkw_platform PUBLIC cxx_std_17)
set_target_properties(mkw_platform PROPERTIES UNITY_BUILD OFF)
if(MKW_PLATFORM_WINDOWS)
target_compile_definitions(mkw_platform PRIVATE NOMINMAX)
target_link_libraries(mkw_platform PUBLIC shell32 ole32 uuid)
endif()
+188
View File
@@ -0,0 +1,188 @@
get_filename_component(MKW_TEST_RUNTIME_DIR "${CMAKE_CURRENT_LIST_DIR}/.." ABSOLUTE)
# Keep these independent from Aurora's BUILD_TESTING option: they validate the
# project's host-platform contracts, not Aurora's third-party test suite.
enable_testing()
if(MKW_BUILD_PSQ_TESTS)
add_executable(mkw_psq_helpers_tests
"${MKW_TEST_RUNTIME_DIR}/tests/psq_helpers_tests.cpp"
"${MKW_TEST_RUNTIME_DIR}/src/ppc_quantized.cpp")
target_include_directories(mkw_psq_helpers_tests PRIVATE
"${MKW_TEST_RUNTIME_DIR}/tests/psq_memory"
"${MKW_TEST_RUNTIME_DIR}/include/isa"
"${MKW_TEST_RUNTIME_DIR}/include")
target_compile_features(mkw_psq_helpers_tests PRIVATE cxx_std_17)
target_compile_options(mkw_psq_helpers_tests PRIVATE
-O2 -fno-fast-math -ffp-contract=off -fno-slp-vectorize)
if(CMAKE_SYSTEM_PROCESSOR MATCHES "^(AMD64|amd64|x86_64|X86_64)$")
target_compile_options(mkw_psq_helpers_tests PRIVATE
-march=x86-64-${MKW_X86_CPU_PROFILE})
endif()
set_target_properties(mkw_psq_helpers_tests PROPERTIES UNITY_BUILD OFF)
add_test(NAME mkw_psq_helpers_tests COMMAND mkw_psq_helpers_tests)
add_test(NAME mkw_psq_reserved_tests COMMAND "${CMAKE_COMMAND}"
"-DPSQ_TEST_EXECUTABLE=$<TARGET_FILE:mkw_psq_helpers_tests>"
-P "${MKW_TEST_RUNTIME_DIR}/tests/psq_reserved_tests.cmake")
endif()
add_executable(mkw_platform_paths_tests "${MKW_TEST_RUNTIME_DIR}/tests/platform_paths_tests.cpp")
target_link_libraries(mkw_platform_paths_tests PRIVATE mkw_platform)
target_compile_features(mkw_platform_paths_tests PRIVATE cxx_std_17)
add_test(NAME mkw_platform_paths_tests COMMAND mkw_platform_paths_tests)
add_executable(mkw_runtime_config_tests "${MKW_TEST_RUNTIME_DIR}/tests/runtime_config_tests.cpp")
target_include_directories(mkw_runtime_config_tests PRIVATE
"${MKW_TEST_RUNTIME_DIR}/include"
"${MKW_TEST_RUNTIME_DIR}/third_party/toml11")
target_compile_features(mkw_runtime_config_tests PRIVATE cxx_std_20)
add_test(NAME mkw_runtime_config_tests COMMAND mkw_runtime_config_tests)
add_executable(mkw_nand_save_tests "${MKW_TEST_RUNTIME_DIR}/tests/nand_save_tests.cpp")
target_include_directories(mkw_nand_save_tests PRIVATE "${MKW_TEST_RUNTIME_DIR}/include")
target_compile_features(mkw_nand_save_tests PRIVATE cxx_std_17)
add_test(NAME mkw_nand_save_tests COMMAND mkw_nand_save_tests)
add_executable(mkw_nand_settings_tests "${MKW_TEST_RUNTIME_DIR}/tests/nand_settings_tests.cpp")
find_package(Threads REQUIRED)
target_link_libraries(mkw_nand_settings_tests PRIVATE Threads::Threads)
target_include_directories(mkw_nand_settings_tests PRIVATE "${MKW_TEST_RUNTIME_DIR}/include")
target_compile_features(mkw_nand_settings_tests PRIVATE cxx_std_17)
add_test(NAME mkw_nand_settings_tests COMMAND mkw_nand_settings_tests)
add_executable(mkw_sc_serial_tests "${MKW_TEST_RUNTIME_DIR}/tests/sc_serial_tests.cpp")
target_include_directories(mkw_sc_serial_tests PRIVATE "${MKW_TEST_RUNTIME_DIR}/include")
target_compile_features(mkw_sc_serial_tests PRIVATE cxx_std_17)
add_test(NAME mkw_sc_serial_tests COMMAND mkw_sc_serial_tests)
# The input expression engine is self-contained, so it can be exercised without
# linking the runtime or SDL.
add_executable(mkw_input_expr_tests
"${MKW_TEST_RUNTIME_DIR}/tests/test_expr.cpp"
"${MKW_TEST_RUNTIME_DIR}/src/input_expr.cpp")
target_include_directories(mkw_input_expr_tests PRIVATE "${MKW_TEST_RUNTIME_DIR}/include")
target_compile_features(mkw_input_expr_tests PRIVATE cxx_std_17)
add_test(NAME mkw_input_expr_tests COMMAND mkw_input_expr_tests)
# HostContext deliberately keeps the platform-specific context primitive out
# of fiber_manager.cpp. Exercise the Linux libco handoff directly so future
# refactors cannot silently remove its headers, implementation, or link edge.
if(MKW_PLATFORM_LINUX)
add_executable(mkw_linux_host_context_tests
"${MKW_TEST_RUNTIME_DIR}/tests/host_context_tests.cpp"
"${MKW_TEST_RUNTIME_DIR}/src/host_context.cpp")
target_include_directories(mkw_linux_host_context_tests PRIVATE
"${MKW_TEST_RUNTIME_DIR}/include"
"${MKW_TEST_RUNTIME_DIR}/third_party/libco")
target_compile_features(mkw_linux_host_context_tests PRIVATE cxx_std_17)
target_link_libraries(mkw_linux_host_context_tests PRIVATE mkw::libco)
add_test(NAME mkw_linux_host_context_tests COMMAND mkw_linux_host_context_tests)
endif()
if(MKW_PLATFORM_WINDOWS)
add_executable(mkw_windows_host_context_tests
"${MKW_TEST_RUNTIME_DIR}/tests/host_context_tests.cpp"
"${MKW_TEST_RUNTIME_DIR}/src/host_context.cpp")
target_include_directories(mkw_windows_host_context_tests PRIVATE "${MKW_TEST_RUNTIME_DIR}/include")
target_compile_features(mkw_windows_host_context_tests PRIVATE cxx_std_17)
add_test(NAME mkw_windows_host_context_tests COMMAND mkw_windows_host_context_tests)
endif()
if(CMAKE_SYSTEM_PROCESSOR MATCHES "^(AMD64|amd64|x86_64|X86_64)$")
# Probe at the base ISA, including OS support for AVX register state. Cross
# builds execute target code only through a configured emulator.
if(NOT CMAKE_CROSSCOMPILING OR CMAKE_CROSSCOMPILING_EMULATOR)
include(CheckCXXSourceRuns)
include(CMakePushCheckState)
cmake_push_check_state(RESET)
set(CMAKE_REQUIRED_FLAGS "-march=x86-64")
check_cxx_source_runs([=[
#include <cpuid.h>
int main() {
unsigned a, b, c, d;
if (__get_cpuid_max(0, nullptr) < 7 ||
__get_cpuid_max(0x80000000u, nullptr) < 0x80000001u)
return 1;
__cpuid_count(1, 0, a, b, c, d);
// v2 plus FMA, MOVBE, XSAVE, OSXSAVE, AVX and F16C.
const unsigned leaf1 = (1u << 0) | (1u << 9) | (1u << 12) |
(1u << 13) | (1u << 19) | (1u << 20) | (1u << 22) |
(1u << 23) | (1u << 26) | (1u << 27) | (1u << 28) | (1u << 29);
if ((c & leaf1) != leaf1) return 1;
__cpuid_count(7, 0, a, b, c, d);
const unsigned leaf7 = (1u << 3) | (1u << 5) | (1u << 8);
if ((b & leaf7) != leaf7) return 1;
__cpuid_count(0x80000001u, 0, a, b, c, d);
if ((c & 0x21u) != 0x21u) return 1; // LAHF-SAHF and LZCNT.
__asm__ __volatile__("xgetbv" : "=a"(a), "=d"(d) : "c"(0));
return (a & 0x6u) == 0x6u ? 0 : 1;
}
]=] MKW_HOST_SUPPORTS_X86_V3)
cmake_pop_check_state()
endif()
# Compile the same oracle cases for both paths. The v2 binary proves the
# scalar fallback stays free of FMA; the v3 binary checks the existing
# vector intrinsic path against the same strict result.
foreach(profile IN ITEMS v2 v3)
add_executable(mkw_ppc_pair_fma_${profile}_tests
"${MKW_TEST_RUNTIME_DIR}/tests/ppc_pair_fma_tests.cpp")
target_include_directories(mkw_ppc_pair_fma_${profile}_tests PRIVATE
"${MKW_TEST_RUNTIME_DIR}/include")
target_compile_features(mkw_ppc_pair_fma_${profile}_tests PRIVATE cxx_std_17)
target_compile_options(mkw_ppc_pair_fma_${profile}_tests PRIVATE
-march=x86-64-${profile} -fno-fast-math -ffp-contract=off)
if((NOT CMAKE_CROSSCOMPILING OR CMAKE_CROSSCOMPILING_EMULATOR) AND
(profile STREQUAL "v2" OR MKW_HOST_SUPPORTS_X86_V3))
add_test(NAME mkw_ppc_pair_fma_${profile}_tests COMMAND mkw_ppc_pair_fma_${profile}_tests)
endif()
endforeach()
add_library(mkw_cpu_baseline_v2_compile OBJECT
"${MKW_TEST_RUNTIME_DIR}/src/host_cpu_baseline.cpp")
target_compile_features(mkw_cpu_baseline_v2_compile PRIVATE cxx_std_17)
target_compile_definitions(mkw_cpu_baseline_v2_compile PRIVATE MKW_X86_CPU_PROFILE_V2=1)
target_compile_options(mkw_cpu_baseline_v2_compile PRIVATE -march=x86-64-v2 -w)
endif()
if(MKW_PLATFORM_MACOS)
# Exercise the public host-memory contracts separately from translated products.
if(MKW_PLATFORM_MACOS_ARM64)
# Apple Silicon's context ABI is implemented by the local assembly backend.
enable_language(ASM)
add_executable(mkw_macos_context_abi_tests
"${MKW_TEST_RUNTIME_DIR}/tests/macos_context_abi_tests.cpp"
"${MKW_TEST_RUNTIME_DIR}/src/platform/macos/co_switch.S")
target_compile_features(mkw_macos_context_abi_tests PRIVATE cxx_std_17)
add_test(NAME mkw_macos_context_abi_tests COMMAND mkw_macos_context_abi_tests)
add_executable(mkw_macos_host_context_tests
"${MKW_TEST_RUNTIME_DIR}/tests/host_context_tests.cpp"
"${MKW_TEST_RUNTIME_DIR}/src/host_context.cpp"
"${MKW_TEST_RUNTIME_DIR}/src/platform/macos/co_switch.S")
target_include_directories(mkw_macos_host_context_tests PRIVATE "${MKW_TEST_RUNTIME_DIR}/include")
else()
# Intel macOS follows the same System V AMD64 libco path as Linux.
add_executable(mkw_macos_host_context_tests
"${MKW_TEST_RUNTIME_DIR}/tests/host_context_tests.cpp"
"${MKW_TEST_RUNTIME_DIR}/src/host_context.cpp")
target_include_directories(mkw_macos_host_context_tests PRIVATE
"${MKW_TEST_RUNTIME_DIR}/include"
"${MKW_TEST_RUNTIME_DIR}/third_party/libco")
target_link_libraries(mkw_macos_host_context_tests PRIVATE mkw::libco)
endif()
target_compile_features(mkw_macos_host_context_tests PRIVATE cxx_std_17)
add_test(NAME mkw_macos_host_context_tests COMMAND mkw_macos_host_context_tests)
add_executable(mkw_macos_guest_flat_memory_tests
"${MKW_TEST_RUNTIME_DIR}/tests/macos_guest_flat_memory_tests.cpp"
"${MKW_TEST_RUNTIME_DIR}/src/guest_flat_memory_macos.cpp")
target_include_directories(mkw_macos_guest_flat_memory_tests PRIVATE "${MKW_TEST_RUNTIME_DIR}/include")
target_compile_features(mkw_macos_guest_flat_memory_tests PRIVATE cxx_std_17)
add_test(NAME mkw_macos_guest_flat_memory_tests COMMAND mkw_macos_guest_flat_memory_tests)
add_executable(mkw_macos_external_audio_tests
"${MKW_TEST_RUNTIME_DIR}/tests/macos_external_audio_tests.cpp"
"${MKW_TEST_RUNTIME_DIR}/src/external_audio_macos.cpp")
target_include_directories(mkw_macos_external_audio_tests PRIVATE "${MKW_TEST_RUNTIME_DIR}/include")
target_compile_features(mkw_macos_external_audio_tests PRIVATE cxx_std_17)
add_test(NAME mkw_macos_external_audio_tests COMMAND mkw_macos_external_audio_tests)
endif()
+17 -11
View File
@@ -85,9 +85,9 @@ target_link_libraries(mkw_runtime_common PRIVATE
target_link_libraries(mkw_runtime_common PRIVATE mkw_platform mkw::pugixml mkw::toml11 mkw::cryptopp mkw::mbedtls)
if(MKW_PLATFORM_WINDOWS)
target_link_libraries(mkw_runtime_common PRIVATE shell32 windowsapp)
elseif(MKW_PLATFORM_LINUX)
elseif(MKW_PLATFORM_LINUX OR MKW_PLATFORM_MACOS_X86_64)
# ${CMAKE_DL_LIBS} for music_attenuation.cpp's dlopen of libdbus-1 (MPRIS
# media monitoring).
# media monitoring). Empty on platforms where dl* is already in libc/libSystem.
target_link_libraries(mkw_runtime_common PRIVATE mkw::libco ${CMAKE_DL_LIBS})
endif()
@@ -135,15 +135,18 @@ set_target_properties(mkw_runtime_common PROPERTIES UNITY_BUILD ON UNITY_BUILD_M
target_precompile_headers(mkw_runtime_common PRIVATE "${MKW_RUNTIME_SOURCE_DIR}/include/mkw_pch.h")
mkw_apply_common_compile_options(mkw_runtime_common)
# Host ISA guard. Windows and Linux x86_64 product targets use x86-64-v3, so
# this object deliberately keeps the plain baseline ISA and checks the CPU
# before any AVX2/FMA code can execute. AArch64 has no equivalent optional ISA
# floor to probe: NEON/FMA are architectural requirements.
# Host ISA guard deliberately keeps the plain baseline ISA and checks the
# selected x86 profile before optional instructions can execute. AArch64 has
# no equivalent optional ISA floor to probe: NEON/FMA are architectural
# requirements.
if(CMAKE_SYSTEM_PROCESSOR MATCHES "^(AMD64|amd64|x86_64|X86_64)$")
add_library(mkw_cpu_baseline OBJECT "${MKW_CPU_BASELINE_SOURCE}")
target_compile_features(mkw_cpu_baseline PRIVATE cxx_std_17)
set_target_properties(mkw_cpu_baseline PROPERTIES UNITY_BUILD OFF)
target_compile_options(mkw_cpu_baseline PRIVATE -w)
if(MKW_X86_CPU_PROFILE STREQUAL "v2")
target_compile_definitions(mkw_cpu_baseline PRIVATE MKW_X86_CPU_PROFILE_V2=1)
endif()
endif()
if(NOT MKW_BASE_COMMON_SHARDS)
@@ -333,12 +336,12 @@ else()
message(STATUS "RetroRewind target disabled (run translate-mod and emit-build-shards)")
endif()
# Windows and Linux x86_64 share the x86-64-v3 floor that the CPU baseline
# object above checks. AArch64 builds are compiled locally for the host that
# will run them, so both Linux and Apple Silicon use the compiler's native CPU
# tuning rather than leaving target-specific performance on the table.
# Windows, Linux and Intel macOS x86_64 use the selected profile. AArch64
# builds are compiled locally for the host that will run them, so both Linux
# and Apple Silicon use the compiler's native CPU tuning rather than leaving
# target-specific performance on the table.
if(CMAKE_SYSTEM_PROCESSOR MATCHES "^(AMD64|amd64|x86_64|X86_64)$")
set(MKW_BASELINE_ARCH_FLAG -march=x86-64-v3)
set(MKW_BASELINE_ARCH_FLAG -march=x86-64-${MKW_X86_CPU_PROFILE})
elseif(CMAKE_SYSTEM_PROCESSOR MATCHES "^(aarch64|arm64|ARM64)$")
set(MKW_BASELINE_ARCH_FLAG -mcpu=native)
else()
@@ -351,5 +354,8 @@ set(MKW_ALL_BUILD_TARGETS
foreach(target IN LISTS MKW_ALL_BUILD_TARGETS)
if(TARGET ${target} AND MKW_BASELINE_ARCH_FLAG)
target_compile_options(${target} PRIVATE ${MKW_BASELINE_ARCH_FLAG})
if(MKW_X86_CPU_PROFILE STREQUAL "v2")
target_compile_definitions(${target} PRIVATE MKW_X86_CPU_PROFILE_V2=1)
endif()
endif()
endforeach()
+1 -1
View File
@@ -4,7 +4,7 @@
// HostContext is the deliberately small boundary between the guest scheduler
// and the host's cooperative-context facility. Windows uses native Fibers;
// Linux and Intel macOS use libco's System V x86-64 backend. macOS AArch64 uses
// Linux uses libco's host backend; Intel macOS uses its System V x86-64 backend. macOS AArch64 uses
// the local assembly backend because it must preserve Darwin's platform-reserved
// x18 register, which libco's AArch64 backend does not save. Its handles are
// only valid on the thread that initialized the scheduler.
+20
View File
@@ -607,7 +607,18 @@ inline double PPC_PsMulNoNiInline(double lhs, double rhs)
inline PpcPairVec PpcFmaddPairInline(PpcPairVec multiplicand, PpcPairVec multiplier, PpcPairVec addend)
{
#if defined(__x86_64__)
#if defined(__FMA__)
return _mm_fmadd_ps(multiplicand, multiplier, addend);
#else
// x86-64-v2 has SSE4.2 but no FMA. Preserve the one rounding point per
// lane through the scalar C++ FMA rather than decomposing into mul+add.
const double a = PpcM128ToPsInline(multiplicand);
const double c = PpcM128ToPsInline(multiplier);
const double b = PpcM128ToPsInline(addend);
return PpcPsToM128Inline(PpcPackPairedInline(
PpcAccuratePsMaddLaneNoNiInline<false>(PpcGetPs0Inline(a), PpcGetPs0Inline(c), PpcGetPs0Inline(b)),
PpcAccuratePsMaddLaneNoNiInline<false>(PpcGetPs1Inline(a), PpcGetPs1Inline(c), PpcGetPs1Inline(b))));
#endif
#elif defined(__aarch64__)
const PpcPairVec result = vfma_f32(addend, multiplicand, multiplier);
if (PpcPairNanLaneBitsInline(result) != 0) [[unlikely]]
@@ -619,7 +630,16 @@ inline PpcPairVec PpcFmaddPairInline(PpcPairVec multiplicand, PpcPairVec multipl
inline PpcPairVec PpcFmsubPairInline(PpcPairVec multiplicand, PpcPairVec multiplier, PpcPairVec subtractor)
{
#if defined(__x86_64__)
#if defined(__FMA__)
return _mm_fmsub_ps(multiplicand, multiplier, subtractor);
#else
const double a = PpcM128ToPsInline(multiplicand);
const double c = PpcM128ToPsInline(multiplier);
const double b = PpcM128ToPsInline(subtractor);
return PpcPsToM128Inline(PpcPackPairedInline(
PpcAccuratePsMaddLaneNoNiInline<true>(PpcGetPs0Inline(a), PpcGetPs0Inline(c), PpcGetPs0Inline(b)),
PpcAccuratePsMaddLaneNoNiInline<true>(PpcGetPs1Inline(a), PpcGetPs1Inline(c), PpcGetPs1Inline(b))));
#endif
#elif defined(__aarch64__)
const PpcPairVec result = vfma_f32(vneg_f32(subtractor), multiplicand, multiplier);
if (PpcPairNanLaneBitsInline(result) != 0) [[unlikely]]
-1
View File
@@ -242,7 +242,6 @@ void WritePollResults(uint32_t outAddress,
const std::vector<NetworkPollContract::CopiedDescriptor>& descriptors);
// network_socket.cpp
int32_t DeleteWiiSocket(uint32_t fd);
void CleanupAllWiiSockets();
sockaddr_in ReadWiiSockAddr(uint32_t addr);
int32_t HandleIpTopIoctl(uint32_t cmd, uint32_t inBuf, uint32_t inLen, uint32_t outBuf,
+21 -3
View File
@@ -54,6 +54,9 @@ constexpr int kMaxSslSessions = 4;
struct SslSession {
bool active = false;
bool handshaked = false;
// A failed POSIX TLS write cannot be resumed with a new guest buffer.
// Keep the slot and socket ownership intact until explicit teardown.
bool failed = false;
bool plaintextWfc = false;
uint32_t socketFd = UINT32_MAX;
NativeSocket native = kInvalidSocket;
@@ -761,6 +764,14 @@ static int32_t SslHandshakeImpl(SslSession& ssl) {
return SSL_OK;
}
static int32_t FailSslWrite(SslSession& ssl) {
ssl.failed = true;
// Stop transport I/O without deleting the guest descriptor or freeing a
// session still referenced by the IOCTLV_NET_SSL_WRITE caller.
::shutdown(ssl.native, SHUT_RDWR);
return SSL_ERR_FAILED;
}
static int32_t SslWrite(SslSession& ssl, const uint8_t* data, uint32_t size) {
if (!data || size == 0) {
return SSL_ERR_ZERO;
@@ -795,12 +806,11 @@ static int32_t SslWrite(SslSession& ssl, const uint8_t* data, uint32_t size) {
}
if (ret == MBEDTLS_ERR_SSL_WANT_READ || ret == MBEDTLS_ERR_SSL_WANT_WRITE) {
if (std::chrono::steady_clock::now() >= writeDeadline) {
DeleteWiiSocket(ssl.socketFd);
return SSL_ERR_FAILED;
return FailSslWrite(ssl);
}
continue;
}
return SSL_ERR_FAILED;
return FailSslWrite(ssl);
}
return static_cast<int32_t>(totalWritten);
}
@@ -842,6 +852,9 @@ static int32_t SslRead(SslSession& ssl, uint8_t* out, uint32_t size) {
// The handshake runs on every SSL read/write, so a failure repeats for as long
// as the session lives; report only the first one.
static int32_t SslHandshake(SslSession& ssl) {
if (ssl.failed) {
return SSL_ERR_FAILED;
}
const int32_t result = SslHandshakeImpl(ssl);
if (result != SSL_OK && !ssl.loggedHandshakeFail) {
ssl.loggedHandshakeFail = true;
@@ -947,6 +960,11 @@ int32_t HandleSslIoctlv(uint32_t cmd, const std::vector<IoVector>& in, const std
WriteSslReturn(in, SSL_ERR_ID);
return 0;
}
// Reject before NAS buffering can acknowledge data on a failed session.
if (g_sslSessions[sslId].failed) {
WriteSslReturn(in, SSL_ERR_FAILED);
return 0;
}
if (out.size() < 2 || !out[1].address) {
WriteSslReturn(in, SSL_ERR_FAILED);
return 0;
+46 -23
View File
@@ -1,14 +1,12 @@
// Host ISA guard. Every other product target builds with -march=x86-64-v3, so a pre-Haswell
// Intel or pre-Excavator AMD machine would otherwise die on an illegal-instruction fault with no
// Host ISA guard. Product targets build with an explicit x86-64-v2 or x86-64-v3 profile, so an
// unsupported machine would otherwise die on an illegal-instruction fault with no
// explanation. This TU alone skips that flag (own CMake object library, excluded from unity
// build/PCH) and runs from a priority-101 C initializer, ahead of every C++ dynamic initializer
// and thus the first AVX2 code that could execute. Keep it free of anything that could pull in
// vectorized code: no iostreams, no std::string, no runtime-wide headers.
//
// x86-64-v3 is an x86-specific optional-feature baseline (AVX2/BMI2/FMA and friends are not
// guaranteed present on every x86_64 chip); nothing here applies on AArch64, where ASIMD/NEON is
// mandatory in the base architecture and PublicProducts.cmake never applies an -march=x86-64-v3
// equivalent flag to begin with. That branch below is a no-op stub, not a port of this check.
// x86-64-v2/v3 are x86-specific optional-feature baselines; nothing here applies on AArch64,
// where ASIMD/NEON is mandatory in the base architecture.
#if defined(__x86_64__)
@@ -58,27 +56,27 @@ struct CpuFeature {
bool isOsXsave;
};
// Everything x86-64-v3 implies, which includes all of x86-64-v2. Spelled out so
// the error message can name the exact instruction sets the machine lacks
// rather than only "AVX2", which is merely the best known member of the set.
// x86-64-v2 requirements. The v3-only extension below retains a useful
// feature-by-feature diagnostic rather than reducing a failed v3 check to AVX2.
constexpr CpuFeature kRequiredFeatures[] = {
{"SSE3", 1, 0, 2, 0, false},
{"SSSE3", 1, 0, 2, 9, false},
{"FMA", 1, 0, 2, 12, false},
{"CMPXCHG16B", 1, 0, 2, 13, false},
{"SSE4.1", 1, 0, 2, 19, false},
{"SSE4.2", 1, 0, 2, 20, false},
{"MOVBE", 1, 0, 2, 22, false},
{"POPCNT", 1, 0, 2, 23, false},
{"OSXSAVE", 1, 0, 2, 27, true},
{"AVX", 1, 0, 2, 28, false},
{"F16C", 1, 0, 2, 29, false},
{"BMI1", 7, 0, 1, 3, false},
{"AVX2", 7, 0, 1, 5, false},
{"BMI2", 7, 0, 1, 8, false},
{"LAHF-SAHF", 0x80000001u, 0, 2, 0, false},
};
#if !defined(MKW_X86_CPU_PROFILE_V2)
constexpr CpuFeature kV3RequiredFeatures[] = {
{"FMA", 1, 0, 2, 12, false}, {"MOVBE", 1, 0, 2, 22, false},
{"OSXSAVE", 1, 0, 2, 27, true}, {"AVX", 1, 0, 2, 28, false},
{"F16C", 1, 0, 2, 29, false}, {"BMI1", 7, 0, 1, 3, false},
{"AVX2", 7, 0, 1, 5, false}, {"BMI2", 7, 0, 1, 8, false},
{"LZCNT", 0x80000001u, 0, 2, 5, false},
};
#endif
// Fixed-capacity text accumulation: no allocation, no exceptions, nothing that
// could route through code this file is trying to stay ahead of.
@@ -130,6 +128,27 @@ bool CollectMissingBaselineFeatures(TextBuffer& missing) {
ok = false;
}
#if !defined(MKW_X86_CPU_PROFILE_V2)
for (const CpuFeature& feature : kV3RequiredFeatures) {
const bool leafAvailable = (feature.leaf & 0x80000000u) != 0
? feature.leaf <= maxExtended
: feature.leaf <= maxBasic;
bool present = false;
if (leafAvailable) {
unsigned regs[4] = {0, 0, 0, 0};
HostCpuId(feature.leaf, feature.subleaf, regs);
present = (regs[feature.reg] & (1u << feature.bit)) != 0;
}
if (present) {
haveOsXsave = haveOsXsave || feature.isOsXsave;
continue;
}
if (!ok) missing.Append(", ");
missing.Append(feature.name);
ok = false;
}
#endif
// CPUID reporting AVX is not sufficient: the OS also has to have enabled
// XMM and YMM state saving or every VEX-encoded instruction faults. This is
// the same guard a compiler's own runtime feature dispatch applies.
@@ -175,13 +194,17 @@ void WriteStdErrEarly(const char* text) {
[[noreturn]] void ReportUnsupportedCpu(const char* missing) {
TextBuffer message;
message.Append(
"This build needs a processor that supports AVX2 and the rest of the "
"x86-64-v3 instruction set.\n\nMissing on this machine: ");
#if defined(MKW_X86_CPU_PROFILE_V2)
message.Append("This build needs a processor that supports the x86-64-v2 instruction set.\n\nMissing on this machine: ");
#else
message.Append("This build needs a processor that supports the x86-64-v3 instruction set.\n\nMissing on this machine: ");
#endif
message.Append(missing);
message.Append(
"\n\nx86-64-v3 covers Intel Core processors from Haswell (4th "
"generation, 2013) onward and AMD processors from Excavator (2015) onward.");
#if defined(MKW_X86_CPU_PROFILE_V2)
message.Append("\n\nx86-64-v2 covers Intel Core processors from Nehalem (2008) onward and AMD processors from Jaguar (2013) onward.");
#else
message.Append("\n\nx86-64-v3 covers Intel Core processors from Haswell (4th generation, 2013) onward and AMD processors from Excavator (2015) onward.");
#endif
// The tag matches RT_TAG_RUNTIME in runtime_log.h. It is spelled out here
// because this translation unit must not include runtime-wide headers (see
+13
View File
@@ -1,6 +1,7 @@
#include "platform/host_platform.h"
#include <cstdlib>
#include <vector>
#if defined(_WIN32)
#ifndef WIN32_LEAN_AND_MEAN
@@ -46,6 +47,18 @@ std::optional<std::filesystem::path> ExecutableDirectory() noexcept {
std::error_code ec;
const auto resolved = std::filesystem::weakly_canonical(path, ec);
return (ec ? std::filesystem::path(path) : resolved).parent_path();
#elif defined(__linux__)
std::vector<char> buffer(256);
for (;;) {
const auto length = ::readlink("/proc/self/exe", buffer.data(), buffer.size());
if (length < 0) {
return std::nullopt;
}
if (static_cast<std::size_t>(length) < buffer.size()) {
return std::filesystem::path(std::string(buffer.data(), length)).parent_path();
}
buffer.resize(buffer.size() * 2);
}
#else
return std::nullopt;
#endif
+58
View File
@@ -0,0 +1,58 @@
#include "isa/ppc_isa_float.h"
#include <cmath>
#include <cstdint>
#include <cstdio>
namespace {
bool SameBits(double lhs, double rhs)
{
return PpcBitCastToU64Inline(lhs) == PpcBitCastToU64Inline(rhs);
}
bool CheckPair(double actual, float a0, float a1, float c0, float c1, float b0, float b1,
bool subtract, const char* operation)
{
const float expected0 = subtract ? std::fma(a0, c0, -b0) : std::fma(a0, c0, b0);
const float expected1 = subtract ? std::fma(a1, c1, -b1) : std::fma(a1, c1, b1);
const double expected = PpcPackPairedInline(expected0, expected1);
if (SameBits(actual, expected)) {
return true;
}
std::fprintf(stderr, "%s produced the wrong paired lanes\n", operation);
return false;
}
bool CheckNegatedPair(double actual, float a0, float a1, float c0, float c1, float b0, float b1,
bool subtract, const char* operation)
{
const float expected0 = -std::fma(a0, c0, subtract ? -b0 : b0);
const float expected1 = -std::fma(a1, c1, subtract ? -b1 : b1);
const double expected = PpcPackPairedInline(expected0, expected1);
if (SameBits(actual, expected)) {
return true;
}
std::fprintf(stderr, "%s produced the wrong paired lanes\n", operation);
return false;
}
} // namespace
int main()
{
const double a = PpcPackPairedInline(1.000000119f, -123.25f);
const double c = PpcPackPairedInline(33554431.0f, 0.0625f);
const double b = PpcPackPairedInline(-33554430.0f, 7.75f);
bool ok = true;
ok &= CheckPair(PPC_PsMaddNoNiInline(a, c, b), 1.000000119f, -123.25f,
33554431.0f, 0.0625f, -33554430.0f, 7.75f, false, "ps_madd");
ok &= CheckPair(PPC_PsMsubNoNiInline(a, c, b), 1.000000119f, -123.25f,
33554431.0f, 0.0625f, -33554430.0f, 7.75f, true, "ps_msub");
ok &= CheckNegatedPair(PPC_PsNmaddInline(a, c, b), 1.000000119f, -123.25f,
33554431.0f, 0.0625f, -33554430.0f, 7.75f, false, "ps_nmadd");
ok &= CheckNegatedPair(PPC_PsNmsubNoNiInline(a, c, b), 1.000000119f, -123.25f,
33554431.0f, 0.0625f, -33554430.0f, 7.75f, true, "ps_nmsub");
return ok ? 0 : 1;
}
@@ -225,13 +225,37 @@ public sealed partial class CxxLinearCodeGenerator
var labelNames = func.Blocks.ToDictionary(b => b.Label, b => SanitizeLabel(b.Label), StringComparer.OrdinalIgnoreCase);
var instructionContinuationLabels = new Dictionary<uint, string>();
var continuationCallCount = func.Blocks
var continuationCallCount = func.Blocks.Sum(block =>
{
var term = block.Instructions.LastOrDefault();
var emittedInstructionCount = term is IrBranch or IrJump or IrReturn or IrJumpTable or IrUndefined
? Math.Max(0, block.Instructions.Count - 1)
: block.Instructions.Count;
var suppressed = suppressedInstructionMasks.TryGetValue(block.Label, out var mask) ? mask : null;
return Enumerable.Range(0, emittedInstructionCount).Count(index =>
{
if ((suppressed is not null && index < suppressed.Length && suppressed[index]) ||
block.Instructions[index] is not IrCall call ||
!TryParseAddress(call.Target, out var target) ||
TryGetInlineGuestThunkSpec(target, out _))
{
return false;
}
return nonReturningCallTargets.Contains(target) ||
(lrContinuationCallTargets.Contains(target) &&
TryGetLocalFallthroughLr(block.Instructions, index, nonReturningCallTargets, lrContinuationCallTargets).HasValue);
});
});
// Labels remain available for every recognized continuation target.
// Sharing the terminal dispatcher is narrower: it only applies when
// more than one emitted call can actually jump there.
var needsInstructionContinuationLabels = func.Blocks
.SelectMany(static block => block.Instructions)
.OfType<IrCall>()
.Count(call =>
.Any(call =>
TryParseAddress(call.Target, out var target) &&
(nonReturningCallTargets.Contains(target) || lrContinuationCallTargets.Contains(target)));
var needsInstructionContinuationLabels = continuationCallCount > 0;
var shareLrContinuationDispatch = continuationCallCount > 1;
if (needsInstructionContinuationLabels)
{
@@ -500,7 +524,7 @@ public sealed partial class CxxLinearCodeGenerator
// Every call site has already reloaded the callee's state.
// Keep the complete local target set, but emit it only once.
body.AppendLine(" return;");
body.AppendLine("[[maybe_unused]] lr_continuation_dispatch:");
body.AppendLine("lr_continuation_dispatch:");
EmitLocalLrContinuationDispatch(body, " ", labelNames);
body.AppendLine(" if (TranslatedFunctionRegistry::FindByAddressPtr(ctx->lr) != nullptr) {");
body.AppendLine(" InvokeIndirectCpu(ctx->lr, ctx);");
@@ -1943,21 +1943,6 @@ public sealed partial class PpcLifter
};
case "bltl":
{
var crField = "cr0";
if (ops.Count > 0 && ops[0] is PpcConditionRegisterOperand crOp)
{
crField = NormalizeRegister(crOp.Name);
}
return new IrInstruction[]
{
new IrAssign("lr", IrValue.Imm((int)ins.EndAddress)),
new IrBranch("blt", TargetLabel(ins, validAddresses, preferFallthrough: false),
$"0x{ins.EndAddress:X8}", crField)
};
}
case "bcl":
{
var rawInstr = ReadRawInstruction(ins);
@@ -1971,6 +1956,7 @@ public sealed partial class PpcLifter
{
instructions.Add(new IrBinary("ctr", IrValue.Register("ctr"), IrValue.Imm(-1), "add"));
}
instructions.Add(new IrAssign("lr", IrValue.Imm((int)ins.EndAddress)));
instructions.Add(new IrBranch("raw", linkedTarget, fallthrough, BuildBoConditionExpression(bo, bi, allowCtr: true)));
return instructions;
}
@@ -184,4 +184,3 @@ public class EmittedOutputShapeTests
Assert.DoesNotContain("loc_800E77A0:\n}", code.Replace("\r\n", "\n"), StringComparison.Ordinal);
}
}
@@ -21,14 +21,20 @@ public class PpcLifterAdditionalCoverageTests
var ir = Assert.Single(new PpcLifter().Lift(new[] { branch })).Ir;
Assert.Equal("bltl", branch.Mnemonic);
var lr = Assert.IsType<IrAssign>(ir[0]);
Assert.Equal("lr", lr.Destination);
Assert.Equal(unchecked((int)0x80004398u), lr.Value.Constant);
Assert.Equal(2, ir.Count);
var lrAssign = Assert.IsType<IrAssign>(ir[0]);
Assert.Equal("lr", lrAssign.Destination);
Assert.Equal(unchecked((int)0x80004398u), lrAssign.Value.Constant);
Assert.Equal(unchecked((int)branch.EndAddress), lrAssign.Value.Constant);
var decision = Assert.IsType<IrBranch>(ir[1]);
Assert.Equal("blt", decision.Condition);
Assert.Equal("0x800043BC", decision.TrueLabel);
Assert.Equal("raw", decision.Condition);
Assert.Equal("link_branch_80004398_800043BC", decision.TrueLabel);
Assert.Equal("0x80004398", decision.FalseLabel);
Assert.Equal($"link_branch_{branch.EndAddress:X8}_{branch.BranchTargets.First():X8}", decision.TrueLabel);
Assert.Equal($"0x{branch.EndAddress:X8}", decision.FalseLabel);
}
[Fact]
@@ -9,6 +9,33 @@ namespace Translator.Tests;
public class SharedLrContinuationCodeGenTests
{
[Fact]
public void OnlyLocalLinkRegisterContinuationsCountTowardSharedDispatch()
{
var function = new IrFunction("single_lr_continuation", "0x80001000", new[]
{
new IrBasicBlock("0x80001000", new IrInstruction[]
{
new IrCall(string.Empty, "0x81800000", Array.Empty<IrValue>()),
new IrAssign("lr", IrValue.Imm(unchecked((int)0x80001004u))),
new IrCall(string.Empty, "0x81800000", Array.Empty<IrValue>()),
new IrReturn(null)
})
});
var types = new RepresentationEnvironment(new Dictionary<string, ValueRepresentation>
{
["lr"] = ValueRepresentation.UInt32
});
var code = new CxxLinearCodeGenerator().Emit(0x80001000,
new SsaTransformer().Convert(function),
new FunctionAbiClassification("single_lr_continuation", ValueRepresentation.Void), types,
lrContinuationCallTargets: new HashSet<uint> { 0x81800000u });
Assert.DoesNotContain("lr_continuation_dispatch:", code);
Assert.DoesNotContain("goto lr_continuation_dispatch;", code);
}
[Theory]
[InlineData(2)]
[InlineData(20)]