Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions .github/workflows/build.yml
Original file line number Diff line number Diff line change
Expand Up @@ -61,6 +61,7 @@ jobs:
run: |
cmake --build build-macos --target \
mkw_platform_paths_tests \
mkw_runtime_config_tests \
mkw_nand_save_tests \
mkw_nand_settings_tests \
mkw_sc_serial_tests \
Expand Down
1 change: 1 addition & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -45,6 +45,7 @@ rather it didn't. All audio that shows in your display media controls on your wi
Press **F10** while the game window has focus:
- Internal resolution
- FPS counter
- MetalFX spatial upscaling on supported macOS GPUs
- Controller assignment for all four ports
- Full per-controller button mapping, including the bumpers
- Dolphin-syntax input expressions and GCPadNew.ini import
Expand Down
8 changes: 8 additions & 0 deletions aurora-main/CMakeLists.txt
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,7 @@ cmake_minimum_required(VERSION 3.25)
project(aurora LANGUAGES C CXX)
if (APPLE)
enable_language(OBJC)
enable_language(OBJCXX)
endif()
set(CMAKE_C_STANDARD 11)
set(CMAKE_CXX_STANDARD 20)
Expand Down Expand Up @@ -83,3 +84,10 @@ if (CMAKE_SOURCE_DIR STREQUAL CMAKE_CURRENT_SOURCE_DIR AND NOT CMAKE_CROSSCOMPIL
enable_testing()
add_subdirectory(tests)
endif ()

option(AURORA_BUILD_METALFX_PRESENTATION_TEST "Build the macOS MetalFX presentation test" OFF)
if (AURORA_BUILD_METALFX_PRESENTATION_TEST AND APPLE AND AURORA_ENABLE_GX AND DAWN_ENABLE_METAL AND AURORA_METALFX_FRAMEWORK)
add_executable(metalfx_presentation_test tests/metalfx_interop/presentation_test.cpp)
target_include_directories(metalfx_presentation_test PRIVATE lib)
target_link_libraries(metalfx_presentation_test PRIVATE aurora::core aurora::gx aurora::vi dawn::webgpu_dawn)
endif ()
11 changes: 11 additions & 0 deletions aurora-main/cmake/aurora_core.cmake
Original file line number Diff line number Diff line change
Expand Up @@ -34,6 +34,17 @@ if (AURORA_ENABLE_GX)
target_compile_definitions(aurora_core PUBLIC AURORA_ENABLE_GX WEBGPU_DAWN)
target_sources(aurora_core PRIVATE lib/webgpu/gpu.cpp lib/webgpu/gpu_cache.cpp lib/dawn/BackendBinding.cpp)
target_link_libraries(aurora_core PRIVATE dawn::webgpu_dawn)
if (APPLE AND DAWN_ENABLE_METAL)
find_library(AURORA_METALFX_FRAMEWORK MetalFX)
endif ()
if (APPLE AND DAWN_ENABLE_METAL AND AURORA_METALFX_FRAMEWORK)
target_sources(aurora_core PRIVATE lib/webgpu/metalfx.mm)
set_source_files_properties(lib/webgpu/metalfx.mm PROPERTIES COMPILE_FLAGS -fobjc-arc)
target_link_options(aurora_core PUBLIC "LINKER:-weak_framework,MetalFX")
target_link_libraries(aurora_core PRIVATE "-framework IOSurface")
else ()
target_sources(aurora_core PRIVATE lib/webgpu/metalfx_stub.cpp)
endif ()
if (DAWN_ENABLE_VULKAN)
target_compile_definitions(aurora_core PRIVATE DAWN_ENABLE_BACKEND_VULKAN)
endif ()
Expand Down
15 changes: 15 additions & 0 deletions aurora-main/include/aurora/aurora.h
Original file line number Diff line number Diff line change
Expand Up @@ -162,6 +162,21 @@ void aurora_set_background_input(bool value);
void aurora_set_display_mode(AuroraDisplayMode mode);
AuroraDisplayMode aurora_get_display_mode();

typedef enum {
AURORA_METALFX_DISABLED,
AURORA_METALFX_UNSUPPORTED,
AURORA_METALFX_NOT_UPSCALING,
AURORA_METALFX_ACTIVE,
AURORA_METALFX_ERROR,
} AuroraMetalFXStatus;

// Changes are consumed at the next sealed frame boundary. MetalFX only applies
// when both source dimensions are smaller than the aspect-fitted output.
void aurora_set_metalfx_spatial(bool enabled);
bool aurora_get_metalfx_spatial();
bool aurora_is_metalfx_spatial_supported();
AuroraMetalFXStatus aurora_get_metalfx_status();

AuroraBackend aurora_get_backend();
const AuroraBackend* aurora_get_available_backends(size_t* count);

Expand Down
159 changes: 150 additions & 9 deletions aurora-main/lib/aurora.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,7 @@
#include "gx/shader_info.hpp"
#include "imgui.hpp"
#include "webgpu/gpu.hpp"
#include "webgpu/metalfx.hpp"
#include <webgpu/webgpu_cpp.h>
#endif

Expand Down Expand Up @@ -60,6 +61,9 @@ std::atomic<AuroraFrameWorkerWaitCallback> g_frameWorkerWaitCallback{nullptr};
// deadlines derived from it, so the presenter cannot drift. Zero means present when ready.
std::atomic<uint64_t> g_presentScheduleBaseNanos{0};
std::atomic<uint64_t> g_presentScheduleIntervalNanos{0};
std::atomic<bool> g_metalfxRequested{false};
std::atomic<bool> g_metalfxSupported{false};
std::atomic<AuroraMetalFXStatus> g_metalfxStatus{AURORA_METALFX_DISABLED};

namespace {
Module Log("aurora");
Expand Down Expand Up @@ -747,6 +751,9 @@ AuroraInfo initialize(int argc, char* argv[], const AuroraConfig& config) noexce
#ifdef AURORA_ENABLE_GX
gfx::initialize();

g_metalfxSupported.store(webgpu::metalfx::supported(g_device, webgpu::g_backendType));
g_metalfxStatus.store(AURORA_METALFX_DISABLED);

imgui::create_context();
#endif
const auto size = window::get_window_size();
Expand Down Expand Up @@ -1169,12 +1176,114 @@ void stop_presenter() noexcept {
g_presenterStarted.store(false, std::memory_order_release);
}

struct MetalFXSlot {
webgpu::metalfx::Size size{};
std::unique_ptr<webgpu::metalfx::SpatialScaler> scaler;
wgpu::BindGroup bindGroup;
};
std::array<MetalFXSlot, gx::MaxInterpolatedFrames + 1> g_metalfxSlots;
size_t g_metalfxNextSlot = 0;
webgpu::metalfx::SpatialScaler* g_metalfxPendingOutput = nullptr;
bool g_metalfxFailed = false;

void metalfx_failed(const std::string& reason) {
Log.warn("MetalFX spatial upscaling disabled: {}; using normal presentation", reason);
g_metalfxFailed = true;
g_metalfxStatus.store(AURORA_METALFX_ERROR);
g_metalfxPendingOutput = nullptr;
g_metalfxSlots = {};
}

wgpu::BindGroup upscale_presentation(wgpu::CommandEncoder& encoder,
const webgpu::PresentSource& source,
const webgpu::Viewport& viewport, bool enabled) {
if (!enabled) {
g_metalfxSlots = {};
g_metalfxFailed = false;
g_metalfxStatus.store(AURORA_METALFX_DISABLED);
return {};
}
if (!g_metalfxSupported.load()) {
g_metalfxStatus.store(AURORA_METALFX_UNSUPPORTED);
return {};
}
if (g_metalfxFailed) return {};
const webgpu::metalfx::Size size{
source.size.width, source.size.height,
static_cast<uint32_t>(viewport.width), static_cast<uint32_t>(viewport.height),
webgpu::g_graphicsConfig.surfaceConfiguration.format,
};
// The existing copy path samples perceptual values from unorm game images.
// Do not introduce implicit sRGB decoding or downscaling into MetalFX.
if (!size.inputWidth || !size.inputHeight || size.inputWidth >= size.outputWidth ||
size.inputHeight >= size.outputHeight ||
(source.format != wgpu::TextureFormat::RGBA8Unorm && source.format != wgpu::TextureFormat::BGRA8Unorm)) {
g_metalfxStatus.store(AURORA_METALFX_NOT_UPSCALING);
return {};
}
auto& slot = g_metalfxSlots[g_metalfxNextSlot++ % g_metalfxSlots.size()];
if (!slot.scaler || !(slot.size == size)) {
slot = {};
std::string error;
slot.scaler = webgpu::metalfx::create(g_instance, g_device, size, error);
if (!slot.scaler) {
if (error.empty()) g_metalfxStatus.store(AURORA_METALFX_NOT_UPSCALING);
else metalfx_failed(error);
return {};
}
slot.size = size;
wgpu::SamplerDescriptor samplerDescriptor{};
samplerDescriptor.magFilter = wgpu::FilterMode::Linear;
samplerDescriptor.minFilter = wgpu::FilterMode::Linear;
slot.bindGroup = webgpu::create_copy_bind_group(slot.scaler->output_view(),
g_device.CreateSampler(&samplerDescriptor));
Log.info("MetalFX spatial slot: {}x{} -> {}x{}", size.inputWidth, size.inputHeight,
size.outputWidth, size.outputHeight);
}
if (!slot.scaler->begin_input()) {
metalfx_failed(slot.scaler->error());
return {};
}
const wgpu::RenderPassColorAttachment attachment{
.view = slot.scaler->input_view(),
.loadOp = wgpu::LoadOp::Clear,
.storeOp = wgpu::StoreOp::Store,
};
const wgpu::RenderPassDescriptor descriptor{
.label = "MetalFX input copy",
.colorAttachmentCount = 1,
.colorAttachments = &attachment,
};
auto pass = encoder.BeginRenderPass(&descriptor);
pass.SetPipeline(webgpu::g_CopyPipeline);
pass.SetBindGroup(0, source.bindGroup);
pass.SetViewport(0, 0, static_cast<float>(size.inputWidth), static_cast<float>(size.inputHeight), 0, 1);
pass.Draw(3);
pass.End();
// Submit the sealed scene and input copy before crossing to the native queue.
// The replacement encoder composites the upscaled image and ImGui normally.
auto buffer = encoder.Finish();
{
std::lock_guard submitLock(g_queueSubmitMutex);
g_queue.Submit(1, &buffer);
}
encoder = g_device.CreateCommandEncoder();
if (!slot.scaler->upscale()) {
metalfx_failed(slot.scaler->error());
return {};
}
g_metalfxPendingOutput = slot.scaler.get();
g_metalfxStatus.store(AURORA_METALFX_ACTIVE);
return slot.bindGroup;
}

// `presentSource` is latched in the seal prologue: by the time this encodes, the producer's next
// gfx::begin_frame() may already have cleared the display-copy override.
void encode_presentation_snapshot(const wgpu::CommandEncoder& encoder,
const webgpu::PresentSource& presentSource,
const PresentationImage& image,
bool includeImGui) {
wgpu::BindGroup encode_presentation_snapshot(wgpu::CommandEncoder& encoder,
const webgpu::PresentSource& presentSource,
const PresentationImage& image,
bool includeImGui, bool metalfxEnabled,
const wgpu::BindGroup* cachedMetalFXOutput = nullptr) {
ZoneScoped;
auto viewport = webgpu::calculate_present_viewport(
image.texture.size.width, image.texture.size.height, presentSource.size.width,
Expand All @@ -1185,6 +1294,13 @@ void encode_presentation_snapshot(const wgpu::CommandEncoder& encoder,
image.texture.size.width, image.texture.size.height, presentAspect);
}
wgpu::BindGroup presentBindGroup = presentSource.bindGroup;
wgpu::BindGroup newMetalFXOutput;
if (cachedMetalFXOutput && *cachedMetalFXOutput) {
presentBindGroup = *cachedMetalFXOutput;
} else if (auto upscaled = upscale_presentation(encoder, presentSource, viewport, metalfxEnabled)) {
presentBindGroup = std::move(upscaled);
newMetalFXOutput = presentBindGroup;
}
{
const std::array attachments{
wgpu::RenderPassColorAttachment{
Expand Down Expand Up @@ -1225,6 +1341,7 @@ void encode_presentation_snapshot(const wgpu::CommandEncoder& encoder,
imgui::render(pass);
pass.End();
}
return newMetalFXOutput;
}
#endif

Expand All @@ -1233,6 +1350,13 @@ void shutdown() noexcept {
#ifdef AURORA_ENABLE_GX
stop_presenter();
g_presentationImagePools = {};
g_metalfxSlots = {};
g_metalfxPendingOutput = nullptr;
g_metalfxNextSlot = 0;
g_metalfxFailed = false;
g_metalfxRequested.store(false);
g_metalfxSupported.store(false);
g_metalfxStatus.store(AURORA_METALFX_DISABLED);
imgui::shutdown();
gfx::shutdown();
webgpu::shutdown();
Expand Down Expand Up @@ -1345,6 +1469,7 @@ struct SealedFrameContext {
uint32_t logicalFrame = 0;
bool interpolationActive = false;
bool replayInterpolatedFrames = false;
bool metalfxEnabled = false;
};

// Phase 1: everything that touches producer-shared renderer state. Needs g_rendererGpuMutex and
Expand Down Expand Up @@ -1372,6 +1497,7 @@ void seal_frame_locked(gfx::SealedFrame& sealedFrame, SealedFrameContext& ctx) {
ctx.snapshotWidth = (std::max)(windowSize.native_fb_width, 1u);
ctx.snapshotHeight = (std::max)(windowSize.native_fb_height, 1u);
ctx.logicalFrame = gfx::current_frame();
ctx.metalfxEnabled = g_metalfxRequested.load();
// Latched before webgpu::clear_present_source_override() in the producer's
// next gfx::begin_frame().
ctx.presentSource = webgpu::current_present_source();
Expand Down Expand Up @@ -1419,10 +1545,14 @@ std::vector<PresentationJob> encode_sealed_frame(gfx::SealedFrame& sealedFrame,
const wgpu::CommandBufferDescriptor cmdBufDescriptor{
.label = "Presentation slot command buffer",
};
const auto submitEncodedSlot = [&](wgpu::CommandEncoder& target) {
const auto submitEncodedSlot = [&](wgpu::CommandEncoder& target, bool releaseMetalFXOutput = true) {
const auto buffer = target.Finish(&cmdBufDescriptor);
std::lock_guard submitLock(g_queueSubmitMutex);
g_queue.Submit(1, &buffer);
if (releaseMetalFXOutput && g_metalfxPendingOutput) {
if (!g_metalfxPendingOutput->end_output()) metalfx_failed(g_metalfxPendingOutput->error());
g_metalfxPendingOutput = nullptr;
}
};

if (ctx.replayInterpolatedFrames) {
Expand All @@ -1431,7 +1561,7 @@ std::vector<PresentationJob> encode_sealed_frame(gfx::SealedFrame& sealedFrame,
gfx::render(sealedFrame, encoder, static_cast<int32_t>(interpolatedFrame), false);
auto image =
acquire_presentation_image(interpolatedFrame, ctx.snapshotWidth, ctx.snapshotHeight);
encode_presentation_snapshot(encoder, ctx.presentSource, *image, true);
encode_presentation_snapshot(encoder, ctx.presentSource, *image, true, ctx.metalfxEnabled);
presentationJobs.push_back({
.image = std::move(image),
.logicalFrame = ctx.logicalFrame,
Expand All @@ -1449,26 +1579,33 @@ std::vector<PresentationJob> encode_sealed_frame(gfx::SealedFrame& sealedFrame,
// The copy targets now hold this frame's resolves, so queue their readbacks on the same encoder;
// completion is harvested in gfx::after_submit, never waited on here.
gfx::efb_ram::encode_async_downloads(encoder);
wgpu::BindGroup duplicatedMetalFXOutput;
if (!ctx.replayInterpolatedFrames) {
for (uint32_t interpolatedFrame = 0; interpolatedFrame < ctx.interpolatedFrameCount;
++interpolatedFrame) {
auto image =
acquire_presentation_image(interpolatedFrame, ctx.snapshotWidth, ctx.snapshotHeight);
encode_presentation_snapshot(encoder, ctx.presentSource, *image, true);
const auto newMetalFXOutput = encode_presentation_snapshot(
encoder, ctx.presentSource, *image, true, ctx.metalfxEnabled,
duplicatedMetalFXOutput ? &duplicatedMetalFXOutput : nullptr);
if (!duplicatedMetalFXOutput && newMetalFXOutput) {
duplicatedMetalFXOutput = newMetalFXOutput;
}
presentationJobs.push_back({
.image = std::move(image),
.logicalFrame = ctx.logicalFrame,
.presentAt = slotPresentDeadline(interpolatedFrame),
.interpolated = true,
.duplicated = true,
});
submitEncodedSlot(encoder);
submitEncodedSlot(encoder, false);
encoder = g_device.CreateCommandEncoder(&encoderDescriptor);
}
}
auto finalImage =
acquire_presentation_image(ctx.interpolatedFrameCount, ctx.snapshotWidth, ctx.snapshotHeight);
encode_presentation_snapshot(encoder, ctx.presentSource, *finalImage, true);
encode_presentation_snapshot(encoder, ctx.presentSource, *finalImage, true, ctx.metalfxEnabled,
duplicatedMetalFXOutput ? &duplicatedMetalFXOutput : nullptr);
auto pendingFrameCapture = encode_frame_capture(encoder, ctx.presentSource);
presentationJobs.push_back({
.image = std::move(finalImage),
Expand Down Expand Up @@ -1948,3 +2085,7 @@ void aurora_set_background_input(bool value) {
}
void aurora_set_display_mode(AuroraDisplayMode mode) { aurora::window::set_display_mode(mode); }
AuroraDisplayMode aurora_get_display_mode() { return aurora::window::get_display_mode(); }
void aurora_set_metalfx_spatial(bool enabled) { aurora::g_metalfxRequested.store(enabled); }
bool aurora_get_metalfx_spatial() { return aurora::g_metalfxRequested.load(); }
bool aurora_is_metalfx_spatial_supported() { return aurora::g_metalfxSupported.load(); }
AuroraMetalFXStatus aurora_get_metalfx_status() { return aurora::g_metalfxStatus.load(); }
8 changes: 8 additions & 0 deletions aurora-main/lib/webgpu/gpu.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -638,6 +638,14 @@ bool initialize(AuroraBackend auroraBackend) {
requiredLimits.maxDynamicStorageBuffersPerPipelineLayout, requiredLimits.maxStorageBuffersPerShaderStage,
requiredLimits.minUniformBufferOffsetAlignment, requiredLimits.minStorageBufferOffsetAlignment);
std::vector<wgpu::FeatureName> requiredFeatures;
// Optional native sharing for MetalFX. Devices without either feature keep
// the normal renderer; the upscaler checks the enabled pair at runtime.
if (backend == wgpu::BackendType::Metal &&
g_adapter.HasFeature(wgpu::FeatureName::SharedTextureMemoryIOSurface) &&
g_adapter.HasFeature(wgpu::FeatureName::SharedFenceMTLSharedEvent)) {
requiredFeatures.push_back(wgpu::FeatureName::SharedTextureMemoryIOSurface);
requiredFeatures.push_back(wgpu::FeatureName::SharedFenceMTLSharedEvent);
}
bool implicitDeviceSynchronizationSupported = false;
wgpu::SupportedFeatures supportedFeatures;
g_adapter.GetFeatures(&supportedFeatures);
Expand Down
Loading