Compare commits

...
110 changed files with 21263 additions and 21578 deletions
+2
View File
@@ -329,6 +329,7 @@ add_kyty_full_emulator_test(shader_cfg_tests ../tests/shaderCfgTests.cpp)
add_executable(scalar_provenance_tests EXCLUDE_FROM_ALL add_executable(scalar_provenance_tests EXCLUDE_FROM_ALL
../tests/ScalarProvenanceTests.cpp ../tests/ScalarProvenanceTests.cpp
graphics/host_gpu/hostMemory.cpp graphics/host_gpu/hostMemory.cpp
graphics/shader/recompiler/ir/ReadLaneElimination.cpp
graphics/shader/recompiler/ir/ScalarProvenance.cpp graphics/shader/recompiler/ir/ScalarProvenance.cpp
graphics/shader/recompiler/ir/SrtWalker.cpp graphics/shader/recompiler/ir/SrtWalker.cpp
) )
@@ -441,6 +442,7 @@ if(NOT KYTY_CLANG_CL)
endif() endif()
if(BUILD_TESTING) if(BUILD_TESTING)
add_test(NAME scalar_provenance COMMAND $<TARGET_FILE:scalar_provenance_tests>)
add_test(NAME image_page_table COMMAND $<TARGET_FILE:image_page_table_tests>) add_test(NAME image_page_table COMMAND $<TARGET_FILE:image_page_table_tests>)
add_test(NAME memory_tracker COMMAND $<TARGET_FILE:memory_tracker_tests>) add_test(NAME memory_tracker COMMAND $<TARGET_FILE:memory_tracker_tests>)
add_test(NAME page_manager COMMAND $<TARGET_FILE:page_manager_tests>) add_test(NAME page_manager COMMAND $<TARGET_FILE:page_manager_tests>)
+6 -6
View File
@@ -175,9 +175,9 @@ static void SignalHandler(int sig, siginfo_t* si, void* uctx) {
} }
g_in_exception_filter = true; g_in_exception_filter = true;
auto* uc = static_cast<ucontext_t*>(uctx); auto* uc = static_cast<ucontext_t*>(uctx);
const auto* mc = uc->uc_mcontext; const auto* mc = uc->uc_mcontext;
const auto& ss = mc->__ss; const auto& ss = mc->__ss;
ExceptionInfo info {}; ExceptionInfo info {};
info.exception_address = ss.__rip; info.exception_address = ss.__rip;
@@ -214,7 +214,7 @@ static void SignalHandler(int sig, siginfo_t* si, void* uctx) {
FailFast("host exception callback is null"); FailFast("host exception callback is null");
} }
const bool resolved = handler(info); const bool resolved = handler(info);
g_in_exception_filter = false; g_in_exception_filter = false;
if (resolved) { if (resolved) {
@@ -255,8 +255,8 @@ static void SignalHandler(int signal_number, siginfo_t* signal_info, void* nativ
info.native_context = context; info.native_context = context;
if (signal_number == SIGSEGV || signal_number == SIGBUS) { if (signal_number == SIGSEGV || signal_number == SIGBUS) {
info.type = ExceptionType::AccessViolation; info.type = ExceptionType::AccessViolation;
const auto error_code = static_cast<uint64_t>(gregs[REG_ERR]); const auto error_code = static_cast<uint64_t>(gregs[REG_ERR]);
if ((error_code & PAGE_FAULT_ERROR_INSTRUCTION) != 0) { if ((error_code & PAGE_FAULT_ERROR_INSTRUCTION) != 0) {
info.access_violation_type = AccessViolationType::Execute; info.access_violation_type = AccessViolationType::Execute;
} else if ((error_code & PAGE_FAULT_ERROR_WRITE) != 0) { } else if ((error_code & PAGE_FAULT_ERROR_WRITE) != 0) {
+7 -8
View File
@@ -19,10 +19,10 @@ class LeastRecentlyUsedCache {
public: public:
[[nodiscard]] size_t Insert(Object object, Tick tick) { [[nodiscard]] size_t Insert(Object object, Tick tick) {
const auto id = Build(); const auto id = Build();
auto& item = m_items[id]; auto& item = m_items[id];
item.object = std::move(object); item.object = std::move(object);
item.tick = tick; item.tick = tick;
Attach(item); Attach(item);
return id; return id;
} }
@@ -49,8 +49,7 @@ public:
template <typename Function> template <typename Function>
void ForEachItemBelow(Tick tick, Function&& function) { void ForEachItemBelow(Tick tick, Function&& function) {
constexpr bool ReturnsBool = constexpr bool ReturnsBool = std::is_same_v<std::invoke_result_t<Function, Object>, bool>;
std::is_same_v<std::invoke_result_t<Function, Object>, bool>;
for (auto* item = m_first; item != nullptr;) { for (auto* item = m_first; item != nullptr;) {
if (item->tick > tick) { if (item->tick > tick) {
return; return;
@@ -87,10 +86,10 @@ private:
m_last = &item; m_last = &item;
return; return;
} }
item.prev = m_last; item.prev = m_last;
m_last->next = &item; m_last->next = &item;
item.next = nullptr; item.next = nullptr;
m_last = &item; m_last = &item;
} }
void Detach(Item& item) { void Detach(Item& item) {
+3 -4
View File
@@ -31,10 +31,9 @@ static bool OnOwnStack() {
if (pthread_getattr_np(pthread_self(), &attr) != 0) { if (pthread_getattr_np(pthread_self(), &attr) != 0) {
return false; return false;
} }
void* base = nullptr; void* base = nullptr;
size_t size = 0; size_t size = 0;
const bool ok = const bool ok = pthread_attr_getstack(&attr, &base, &size) == 0 && base != nullptr && size != 0;
pthread_attr_getstack(&attr, &base, &size) == 0 && base != nullptr && size != 0;
pthread_attr_destroy(&attr); pthread_attr_destroy(&attr);
if (!ok) { if (!ok) {
return false; return false;
+3 -5
View File
@@ -172,8 +172,7 @@ sys_file_t* SysFileCreate(const std::filesystem::path& file_name) {
return ret; return ret;
} }
sys_file_t* SysFileOpenR(const std::filesystem::path& file_name, sys_file_t* SysFileOpenR(const std::filesystem::path& file_name, sys_file_cache_type_t cache_type) {
sys_file_cache_type_t cache_type) {
auto* ret = new sys_file_t; auto* ret = new sys_file_t;
ret->type = SYS_FILE_FILE; ret->type = SYS_FILE_FILE;
@@ -218,8 +217,7 @@ sys_file_t* SysFileCreate() {
return ret; return ret;
} }
sys_file_t* SysFileOpenW(const std::filesystem::path& file_name, sys_file_t* SysFileOpenW(const std::filesystem::path& file_name, sys_file_cache_type_t cache_type) {
sys_file_cache_type_t cache_type) {
auto* ret = new sys_file_t; auto* ret = new sys_file_t;
auto real_name = get_internal_name(file_name); auto real_name = get_internal_name(file_name);
@@ -241,7 +239,7 @@ sys_file_t* SysFileOpenW(const std::filesystem::path& file_name,
} }
sys_file_t* SysFileOpenRw(const std::filesystem::path& file_name, sys_file_t* SysFileOpenRw(const std::filesystem::path& file_name,
sys_file_cache_type_t cache_type) { sys_file_cache_type_t cache_type) {
auto* ret = new sys_file_t; auto* ret = new sys_file_t;
auto real_name = get_internal_name(file_name); auto real_name = get_internal_name(file_name);
+13 -13
View File
@@ -136,8 +136,8 @@ static void* map_anonymous(uintptr_t addr, size_t size, int protect, int flags)
break; break;
} }
const auto hint = (top - step) & ~(LOW_ARENA_GRAIN - 1); const auto hint = (top - step) & ~(LOW_ARENA_GRAIN - 1);
void* ptr = mmap(reinterpret_cast<void*>(hint), size, protect, void* ptr = mmap(reinterpret_cast<void*>(hint), size, protect, flags | MAP_FIXED_NOREPLACE,
flags | MAP_FIXED_NOREPLACE, -1, 0); // NOLINT -1, 0); // NOLINT
if (ptr != MAP_FAILED) { if (ptr != MAP_FAILED) {
return ptr; return ptr;
} }
@@ -161,8 +161,8 @@ uint64_t SysVirtualAlloc(uint64_t address, uint64_t size, VirtualMemory::Mode mo
if (ptr != MAP_FAILED) { if (ptr != MAP_FAILED) {
pthread_mutex_lock(&g_virtual_mutex); pthread_mutex_lock(&g_virtual_mutex);
record_alloc(ret_addr, size); record_alloc(ret_addr, size);
uintptr_t page_start = ret_addr >> 12u; uintptr_t page_start = ret_addr >> 12u;
uintptr_t page_end = (ret_addr + size - 1) >> 12u; uintptr_t page_end = (ret_addr + size - 1) >> 12u;
for (uintptr_t page = page_start; page <= page_end; page++) { for (uintptr_t page = page_start; page <= page_end; page++) {
(*g_protects)[page] = protect; (*g_protects)[page] = protect;
} }
@@ -194,8 +194,8 @@ uint64_t SysVirtualAllocAligned(uint64_t address, uint64_t size, VirtualMemory::
if (ptr != MAP_FAILED && ((ret_addr & (alignment - 1)) != 0)) { if (ptr != MAP_FAILED && ((ret_addr & (alignment - 1)) != 0)) {
munmap(ptr, size); munmap(ptr, size);
ptr = map_anonymous(addr, size + alignment, protect, ptr =
MAP_PRIVATE | MAP_ANON | MAP_NORESERVE); map_anonymous(addr, size + alignment, protect, MAP_PRIVATE | MAP_ANON | MAP_NORESERVE);
ret_addr = reinterpret_cast<uintptr_t>(ptr); ret_addr = reinterpret_cast<uintptr_t>(ptr);
if (ptr != MAP_FAILED) { if (ptr != MAP_FAILED) {
#if defined(__APPLE__) #if defined(__APPLE__)
@@ -251,8 +251,8 @@ uint64_t SysVirtualAllocAligned(uint64_t address, uint64_t size, VirtualMemory::
pthread_mutex_lock(&g_virtual_mutex); pthread_mutex_lock(&g_virtual_mutex);
record_alloc(ret_addr, size); record_alloc(ret_addr, size);
uintptr_t page_start = ret_addr >> 12u; uintptr_t page_start = ret_addr >> 12u;
uintptr_t page_end = (ret_addr + size - 1) >> 12u; uintptr_t page_end = (ret_addr + size - 1) >> 12u;
for (uintptr_t page = page_start; page <= page_end; page++) { for (uintptr_t page = page_start; page <= page_end; page++) {
(*g_protects)[page] = protect; (*g_protects)[page] = protect;
} }
@@ -266,9 +266,9 @@ uint64_t SysVirtualAllocAligned(uint64_t address, uint64_t size, VirtualMemory::
// the first mapped region at or above `region_addr`; if it begins before the end of the // the first mapped region at or above `region_addr`; if it begins before the end of the
// requested range, the range overlaps an existing mapping. // requested range, the range overlaps an existing mapping.
static bool is_mapped(void* ptr, size_t length) { static bool is_mapped(void* ptr, size_t length) {
auto query_addr = reinterpret_cast<mach_vm_address_t>(ptr); auto query_addr = reinterpret_cast<mach_vm_address_t>(ptr);
mach_vm_address_t region_addr = query_addr; mach_vm_address_t region_addr = query_addr;
mach_vm_size_t region_size = 0; mach_vm_size_t region_size = 0;
vm_region_basic_info_data_64_t info {}; vm_region_basic_info_data_64_t info {};
mach_msg_type_number_t count = VM_REGION_BASIC_INFO_COUNT_64; mach_msg_type_number_t count = VM_REGION_BASIC_INFO_COUNT_64;
mach_port_t object_name = MACH_PORT_NULL; mach_port_t object_name = MACH_PORT_NULL;
@@ -337,8 +337,8 @@ bool SysVirtualAllocFixed(uint64_t address, uint64_t size, VirtualMemory::Mode m
if (ptr != MAP_FAILED) { if (ptr != MAP_FAILED) {
pthread_mutex_lock(&g_virtual_mutex); pthread_mutex_lock(&g_virtual_mutex);
record_alloc(ret_addr, size); record_alloc(ret_addr, size);
uintptr_t page_start = ret_addr >> 12u; uintptr_t page_start = ret_addr >> 12u;
uintptr_t page_end = (ret_addr + size - 1) >> 12u; uintptr_t page_end = (ret_addr + size - 1) >> 12u;
for (uintptr_t page = page_start; page <= page_end; page++) { for (uintptr_t page = page_start; page <= page_end; page++) {
(*g_protects)[page] = protect; (*g_protects)[page] = protect;
} }
+1 -1
View File
@@ -5,9 +5,9 @@
#include <algorithm> #include <algorithm>
#include <atomic> #include <atomic>
#include <cerrno>
#include <chrono> // IWYU pragma: keep #include <chrono> // IWYU pragma: keep
#include <condition_variable> // IWYU pragma: keep #include <condition_variable> // IWYU pragma: keep
#include <cerrno>
#include <mutex> #include <mutex>
#include <vector> #include <vector>
+2 -4
View File
@@ -11,7 +11,7 @@ template <typename Result, typename... Args>
class UniqueFunction { class UniqueFunction {
class CallableBase { class CallableBase {
public: public:
virtual ~CallableBase() = default; virtual ~CallableBase() = default;
virtual Result Invoke(Args&&... args) = 0; virtual Result Invoke(Args&&... args) = 0;
}; };
@@ -20,9 +20,7 @@ class UniqueFunction {
public: public:
explicit Callable(Function function): m_function(std::move(function)) {} explicit Callable(Function function): m_function(std::move(function)) {}
Result Invoke(Args&&... args) override { Result Invoke(Args&&... args) override { return m_function(std::forward<Args>(args)...); }
return m_function(std::forward<Args>(args)...);
}
private: private:
Function m_function; Function m_function;
@@ -158,7 +158,7 @@ private:
void CheckBuffer() const { GetScheduler().CheckActive(); } void CheckBuffer() const { GetScheduler().CheckActive(); }
GpuResourceManager& GetGpuResources() const { return m_renderer.GetGpuResources(); } GpuResourceManager& GetGpuResources() const { return m_renderer.GetGpuResources(); }
RenderContext& m_renderer; RenderContext& m_renderer;
HW::Context m_ctx; HW::Context m_ctx;
HW::UserConfig m_ucfg; HW::UserConfig m_ucfg;
HW::Shader m_sh_ctx; HW::Shader m_sh_ctx;
@@ -170,9 +170,9 @@ private:
uint64_t m_dispatch_indirect_args_base_addr = 0; uint64_t m_dispatch_indirect_args_base_addr = 0;
uint32_t m_num_instances = 1; uint32_t m_num_instances = 1;
uint32_t m_de_count = 0; uint32_t m_de_count = 0;
uint32_t m_ce_count = 0; uint32_t m_ce_count = 0;
bool m_ce_complete = false; bool m_ce_complete = false;
bool m_readback_active = false; bool m_readback_active = false;
uint32_t m_const_ram[0x3000] = {0}; uint32_t m_const_ram[0x3000] = {0};
+2
View File
@@ -374,6 +374,8 @@ enum class BufferFormat : uint32_t {
k32_32_32_32UInt = 75, k32_32_32_32UInt = 75,
k32_32_32_32SInt = 76, k32_32_32_32SInt = 76,
k32_32_32_32Float = 77, k32_32_32_32Float = 77,
k8Srgb = 128,
k8_8Srgb = 129,
k8_8_8_8Srgb = 130, k8_8_8_8Srgb = 130,
k9_9_9_5Float = 132, k9_9_9_5Float = 132,
k5_6_5UNorm = 133, k5_6_5UNorm = 133,
+2
View File
@@ -57,6 +57,8 @@ constexpr FormatInfo kFormatInfo[] = {
{GpuEnumValue(BufferFormat::k32_32_32_32UInt), 16, 0, 16, true, true}, {GpuEnumValue(BufferFormat::k32_32_32_32UInt), 16, 0, 16, true, true},
{GpuEnumValue(BufferFormat::k32_32_32_32SInt), 16, 0, 16, false, false}, {GpuEnumValue(BufferFormat::k32_32_32_32SInt), 16, 0, 16, false, false},
{GpuEnumValue(BufferFormat::k32_32_32_32Float), 16, 0, 16, true, false}, {GpuEnumValue(BufferFormat::k32_32_32_32Float), 16, 0, 16, true, false},
{GpuEnumValue(BufferFormat::k8Srgb), 1, 0, 0, true, false},
{GpuEnumValue(BufferFormat::k8_8Srgb), 2, 0, 0, true, false},
{GpuEnumValue(BufferFormat::k8_8_8_8Srgb), 4, 0, 4, true, false}, {GpuEnumValue(BufferFormat::k8_8_8_8Srgb), 4, 0, 4, true, false},
{GpuEnumValue(BufferFormat::k9_9_9_5Float), 4, 0, 0, true, false}, {GpuEnumValue(BufferFormat::k9_9_9_5Float), 4, 0, 0, true, false},
{GpuEnumValue(BufferFormat::k5_6_5UNorm), 2, 0, 2, true, false}, {GpuEnumValue(BufferFormat::k5_6_5UNorm), 2, 0, 2, true, false},
+11 -13
View File
@@ -962,9 +962,8 @@ void CommandProcessor::DrawIndexOffset(uint32_t index_offset, uint32_t index_cou
auto* index_addr = reinterpret_cast<const void*>( auto* index_addr = reinterpret_cast<const void*>(
m_index_base_addr + static_cast<uint64_t>(index_offset) * index_size); m_index_base_addr + static_cast<uint64_t>(index_offset) * index_size);
m_renderer.GetRenderExecutor().DrawIndex(m_submit_id, CurrentBuffer(), m_renderer.GetRenderExecutor().DrawIndex(m_submit_id, CurrentBuffer(), m_index_type_and_size,
m_index_type_and_size, index_count, index_addr, index_count, index_addr, flags, 1, m_num_instances);
flags, 1, m_num_instances);
} }
void CommandProcessor::DrawIndirect(uint32_t data_offset, uint32_t draw_initiator, bool indexed) { void CommandProcessor::DrawIndirect(uint32_t data_offset, uint32_t draw_initiator, bool indexed) {
@@ -1190,8 +1189,8 @@ void CommandProcessor::DispatchDirect(uint32_t thread_group_x, uint32_t thread_g
} }
} }
m_renderer.GetRenderExecutor().DispatchDirect( m_renderer.GetRenderExecutor().DispatchDirect(m_submit_id, CurrentBuffer(), thread_group_x,
m_submit_id, CurrentBuffer(), thread_group_x, thread_group_y, thread_group_z, mode); thread_group_y, thread_group_z, mode);
} }
constexpr uint32_t DispatchInitiatorUseThreadDimensions = 1u << 5u; constexpr uint32_t DispatchInitiatorUseThreadDimensions = 1u << 5u;
@@ -1237,16 +1236,16 @@ void CommandProcessor::DrawIndexAuto(uint32_t index_count, uint32_t flags,
uint32_t first_vertex, uint32_t first_instance) { uint32_t first_vertex, uint32_t first_instance) {
CheckBuffer(); CheckBuffer();
m_renderer.GetRenderExecutor().DrawAuto( m_renderer.GetRenderExecutor().DrawAuto(m_submit_id, CurrentBuffer(), index_count, flags,
m_submit_id, CurrentBuffer(), index_count, flags, render_target_slice_offset, render_target_slice_offset, instance_count,
instance_count, first_vertex, first_instance); first_vertex, first_instance);
} }
void CommandProcessor::WaitFlipDone(uint32_t video_out_handle, uint32_t display_buffer_index) { void CommandProcessor::WaitFlipDone(uint32_t video_out_handle, uint32_t display_buffer_index) {
BufferFlush(); BufferFlush();
m_renderer.GetVideoOut().WaitFlipDone(static_cast<int>(video_out_handle), m_renderer.GetVideoOut().WaitFlipDone(static_cast<int>(video_out_handle),
static_cast<int>(display_buffer_index)); static_cast<int>(display_buffer_index));
} }
template <typename T> template <typename T>
@@ -1317,8 +1316,8 @@ void CommandProcessor::WriteAtEndOfPipe(uint32_t cache_policy, uint32_t event_wr
if (eop_event_type == 0x2f && cache_action == 0x00 && event_index == 0x06) { if (eop_event_type == 0x2f && cache_action == 0x00 && event_index == 0x06) {
auto* dst = static_cast<uint32_t*>(dst_gpu_addr); auto* dst = static_cast<uint32_t*>(dst_gpu_addr);
SynchronizeGpu(); SynchronizeGpu();
Sync::ReadGds(m_renderer.GetBufferCache().GetGdsBuffer(), dst, Sync::ReadGds(m_renderer.GetBufferCache().GetGdsBuffer(), dst, value & 0xffffu,
value & 0xffffu, value >> 16u); value >> 16u);
Sync::WriteAtEndOfPipeGds32(m_submit_id, CurrentBuffer(), dst, value & 0xffffu, Sync::WriteAtEndOfPipeGds32(m_submit_id, CurrentBuffer(), dst, value & 0xffffu,
value >> 16u); value >> 16u);
return; return;
@@ -1486,8 +1485,7 @@ void CommandProcessor::EmitGlobalBarrier() {
barrier.srcStageMask = vk::PipelineStageFlagBits2::eAllCommands; barrier.srcStageMask = vk::PipelineStageFlagBits2::eAllCommands;
barrier.srcAccessMask = vk::AccessFlagBits2::eMemoryWrite; barrier.srcAccessMask = vk::AccessFlagBits2::eMemoryWrite;
barrier.dstStageMask = vk::PipelineStageFlagBits2::eAllCommands; barrier.dstStageMask = vk::PipelineStageFlagBits2::eAllCommands;
barrier.dstAccessMask = barrier.dstAccessMask = vk::AccessFlagBits2::eMemoryRead | vk::AccessFlagBits2::eMemoryWrite;
vk::AccessFlagBits2::eMemoryRead | vk::AccessFlagBits2::eMemoryWrite;
vk::DependencyInfo dependency {}; vk::DependencyInfo dependency {};
dependency.memoryBarrierCount = 1; dependency.memoryBarrierCount = 1;
+11 -11
View File
@@ -65,41 +65,41 @@ struct TileVolumeLayout {
}; };
bool TileGetBlockLayout(TileBlockFamily family, uint32_t bytes_per_element, bool TileGetBlockLayout(TileBlockFamily family, uint32_t bytes_per_element,
TileBlockLayout& layout); TileBlockLayout& layout);
bool TileGetBlockOffset(const TileBlockLayout& layout, uint32_t x, uint32_t y, uint32_t z, bool TileGetBlockOffset(const TileBlockLayout& layout, uint32_t x, uint32_t y, uint32_t z,
uint32_t& byte_offset); uint32_t& byte_offset);
bool TileGetBlockXor(const TileBlockLayout& layout, uint32_t block_x, uint32_t block_y, bool TileGetBlockXor(const TileBlockLayout& layout, uint32_t block_x, uint32_t block_y,
uint32_t& byte_offset); uint32_t& byte_offset);
bool TileGetBlockXor(const TileBlockLayout& layout, uint32_t block_x, uint32_t block_y, bool TileGetBlockXor(const TileBlockLayout& layout, uint32_t block_x, uint32_t block_y,
uint32_t block_z, uint32_t& byte_offset); uint32_t block_z, uint32_t& byte_offset);
bool TileIsStandard256BTextureSupported(uint32_t format); bool TileIsStandard256BTextureSupported(uint32_t format);
bool TileIsStandard4KBTextureSupported(uint32_t format); bool TileIsStandard4KBTextureSupported(uint32_t format);
bool TileIsStandard64KBTextureSupported(uint32_t format); bool TileIsStandard64KBTextureSupported(uint32_t format);
bool TileGetTextureVolumeLayout(uint32_t format, uint32_t width, uint32_t height, uint32_t depth, bool TileGetTextureVolumeLayout(uint32_t format, uint32_t width, uint32_t height, uint32_t depth,
uint32_t levels, uint32_t tile, TileVolumeLayout& layout); uint32_t levels, uint32_t tile, TileVolumeLayout& layout);
bool TileGetHtileSize(uint32_t width, uint32_t height, TileSizeAlign& htile_size); bool TileGetHtileSize(uint32_t width, uint32_t height, TileSizeAlign& htile_size);
bool TileGetDepthSize(uint32_t width, uint32_t height, uint32_t pitch, uint32_t z_format, bool TileGetDepthSize(uint32_t width, uint32_t height, uint32_t pitch, uint32_t z_format,
uint32_t stencil_format, bool htile, TileSizeAlign& stencil_size, uint32_t stencil_format, bool htile, TileSizeAlign& stencil_size,
TileSizeAlign& htile_size, TileSizeAlign& depth_size, TileSizeAlign& htile_size, TileSizeAlign& depth_size,
uint32_t num_fragments_log2 = 0); uint32_t num_fragments_log2 = 0);
uint32_t TileGetRenderTargetPitch(uint32_t width, uint32_t bytes_per_element, uint32_t TileGetRenderTargetPitch(uint32_t width, uint32_t bytes_per_element,
uint32_t num_fragments_log2 = 0); uint32_t num_fragments_log2 = 0);
uint32_t TileGetDepthPitch(uint32_t width, uint32_t bytes_per_element, uint32_t TileGetDepthPitch(uint32_t width, uint32_t bytes_per_element,
uint32_t num_fragments_log2 = 0); uint32_t num_fragments_log2 = 0);
bool TileGetRenderTargetSize(uint32_t width, uint32_t height, uint32_t pitch, bool TileGetRenderTargetSize(uint32_t width, uint32_t height, uint32_t pitch,
uint32_t bytes_per_element, TileSizeAlign& total_size, uint32_t bytes_per_element, TileSizeAlign& total_size,
uint32_t num_fragments_log2 = 0); uint32_t num_fragments_log2 = 0);
bool TileGetRenderTargetMipLayout(uint32_t width, uint32_t height, uint32_t pitch, bool TileGetRenderTargetMipLayout(uint32_t width, uint32_t height, uint32_t pitch,
uint32_t bytes_per_element, uint32_t levels, uint32_t bytes_per_element, uint32_t levels,
TileSizeAlign& total_size, TileSizeOffset* level_sizes, TileSizeAlign& total_size, TileSizeOffset* level_sizes,
TilePaddedSize* padded_size); TilePaddedSize* padded_size);
void TileGetTextureSize(uint32_t format, uint32_t width, uint32_t height, uint32_t pitch, void TileGetTextureSize(uint32_t format, uint32_t width, uint32_t height, uint32_t pitch,
uint32_t levels, uint32_t tile, TileSizeAlign* total_size, uint32_t levels, uint32_t tile, TileSizeAlign* total_size,
TileSizeOffset* level_sizes, TilePaddedSize* padded_size); TileSizeOffset* level_sizes, TilePaddedSize* padded_size);
void TileGetTextureTotalSize(uint32_t format, uint32_t width, uint32_t height, uint32_t depth, void TileGetTextureTotalSize(uint32_t format, uint32_t width, uint32_t height, uint32_t depth,
uint32_t pitch, uint32_t levels, uint32_t tile, bool volume_texture, uint32_t pitch, uint32_t levels, uint32_t tile, bool volume_texture,
TileSizeAlign& total_size); TileSizeAlign& total_size);
uint32_t TileGetTexturePitch(uint32_t format, uint32_t width, uint32_t levels, uint32_t tile); uint32_t TileGetTexturePitch(uint32_t format, uint32_t width, uint32_t levels, uint32_t tile);
} // namespace Libs::Graphics } // namespace Libs::Graphics
+12 -12
View File
@@ -60,19 +60,19 @@ struct VulkanImage {
VulkanImage() = default; VulkanImage() = default;
KYTY_CLASS_NO_COPY(VulkanImage); KYTY_CLASS_NO_COPY(VulkanImage);
vk::Format format = vk::Format::eUndefined; vk::Format format = vk::Format::eUndefined;
vk::ImageType image_type = vk::ImageType::e2D; vk::ImageType image_type = vk::ImageType::e2D;
vk::Extent3D extent = {1, 1, 1}; vk::Extent3D extent = {1, 1, 1};
uint32_t guest_pitch = 0; uint32_t guest_pitch = 0;
uint32_t layers = 1; uint32_t layers = 1;
uint32_t mip_levels = 1; uint32_t mip_levels = 1;
uint32_t samples = 1; uint32_t samples = 1;
vk::ImageUsageFlags usage = {}; vk::ImageUsageFlags usage = {};
vk::ImageCreateFlags flags = {}; vk::ImageCreateFlags flags = {};
vk::Image image = nullptr; vk::Image image = nullptr;
VulkanImageState state; VulkanImageState state;
std::vector<VulkanImageState> subresource_states; std::vector<VulkanImageState> subresource_states;
Graphics::VulkanMemory memory; Graphics::VulkanMemory memory;
}; };
struct VulkanBuffer { struct VulkanBuffer {
+1 -1
View File
@@ -30,7 +30,7 @@ bool IsAccessible(DWORD protect, HostMemoryAccess access) {
} // namespace } // namespace
bool HostMemoryQueryRange(uint64_t addr, uint64_t requested_size, HostMemoryAccess access, bool HostMemoryQueryRange(uint64_t addr, uint64_t requested_size, HostMemoryAccess access,
uint64_t& accessible_size) { uint64_t& accessible_size) {
accessible_size = 0; accessible_size = 0;
if (addr == 0 || requested_size == 0) { if (addr == 0 || requested_size == 0) {
return false; return false;
+1 -1
View File
@@ -8,7 +8,7 @@ namespace Libs::Graphics {
enum class HostMemoryAccess { Read, Mapped }; enum class HostMemoryAccess { Read, Mapped };
bool HostMemoryQueryRange(uint64_t addr, uint64_t requested_size, HostMemoryAccess access, bool HostMemoryQueryRange(uint64_t addr, uint64_t requested_size, HostMemoryAccess access,
uint64_t& accessible_size); uint64_t& accessible_size);
bool HostMemoryQueryReadable(uint64_t addr, uint64_t requested_size, uint64_t& readable_size); bool HostMemoryQueryReadable(uint64_t addr, uint64_t requested_size, uint64_t& readable_size);
bool HostMemoryIsReadable(uint64_t addr); bool HostMemoryIsReadable(uint64_t addr);
bool HostMemoryRangeIsReadable(uint64_t addr, uint64_t size); bool HostMemoryRangeIsReadable(uint64_t addr, uint64_t size);
+10 -12
View File
@@ -135,8 +135,8 @@ void Buffer::Write(uint64_t offset, const void* source, uint64_t size) {
void Buffer::Flush(uint64_t offset, uint64_t size) { void Buffer::Flush(uint64_t offset, uint64_t size) {
EXIT_IF(m_mapped.empty() || offset > m_size || size > m_size - offset); EXIT_IF(m_mapped.empty() || offset > m_size || size > m_size - offset);
if (!m_is_coherent && size != 0) { if (!m_is_coherent && size != 0) {
const auto result = vmaFlushAllocation(m_graphics->allocator, m_buffer->memory.allocation, const auto result =
offset, size); vmaFlushAllocation(m_graphics->allocator, m_buffer->memory.allocation, offset, size);
EXIT_NOT_IMPLEMENTED(static_cast<vk::Result>(result) != vk::Result::eSuccess); EXIT_NOT_IMPLEMENTED(static_cast<vk::Result>(result) != vk::Result::eSuccess);
} }
} }
@@ -144,8 +144,8 @@ void Buffer::Flush(uint64_t offset, uint64_t size) {
vk::BufferMemoryBarrier Buffer::Barrier(uint64_t offset, uint64_t size, vk::AccessFlags source, vk::BufferMemoryBarrier Buffer::Barrier(uint64_t offset, uint64_t size, vk::AccessFlags source,
vk::AccessFlags destination) const { vk::AccessFlags destination) const {
if (Handle() == nullptr || size == 0 || offset > m_size || size > m_size - offset) { if (Handle() == nullptr || size == 0 || offset > m_size || size > m_size - offset) {
EXIT("Buffer: invalid DMA barrier, handle=%p offset=0x%016" PRIx64 EXIT("Buffer: invalid DMA barrier, handle=%p offset=0x%016" PRIx64 " size=0x%016" PRIx64
" size=0x%016" PRIx64 " capacity=0x%016" PRIx64 "\n", " capacity=0x%016" PRIx64 "\n",
static_cast<const void*>(Handle()), offset, size, m_size); static_cast<const void*>(Handle()), offset, size, m_size);
} }
vk::BufferMemoryBarrier barrier {}; vk::BufferMemoryBarrier barrier {};
@@ -175,10 +175,9 @@ void Buffer::CopyFrom(CommandBuffer& command, const Buffer& source, uint64_t sou
command.EndRendering(); command.EndRendering();
const vk::BufferMemoryBarrier before[] = { const vk::BufferMemoryBarrier before[] = {
source.Barrier(source_offset, size, source_before, vk::AccessFlagBits::eTransferRead), source.Barrier(source_offset, size, source_before, vk::AccessFlagBits::eTransferRead),
Barrier(destination_offset, size, destination_before, Barrier(destination_offset, size, destination_before, vk::AccessFlagBits::eTransferWrite),
vk::AccessFlagBits::eTransferWrite),
}; };
const auto host_access = vk::AccessFlagBits::eHostRead | vk::AccessFlagBits::eHostWrite; const auto host_access = vk::AccessFlagBits::eHostRead | vk::AccessFlagBits::eHostWrite;
auto before_stage = vk::PipelineStageFlags {vk::PipelineStageFlagBits::eAllCommands}; auto before_stage = vk::PipelineStageFlags {vk::PipelineStageFlagBits::eAllCommands};
if (static_cast<bool>((source_before | destination_before) & host_access)) { if (static_cast<bool>((source_before | destination_before) & host_access)) {
before_stage |= vk::PipelineStageFlagBits::eHost; before_stage |= vk::PipelineStageFlagBits::eHost;
@@ -214,9 +213,8 @@ void Buffer::Fill(uint64_t offset, uint64_t size, uint32_t value) {
vk::PipelineStageFlagBits::eTransfer, vk::DependencyFlagBits::eByRegion, vk::PipelineStageFlagBits::eTransfer, vk::DependencyFlagBits::eByRegion,
0, nullptr, 1, &before, 0, nullptr); 0, nullptr, 1, &before, 0, nullptr);
native.fillBuffer(Handle(), offset, size, value); native.fillBuffer(Handle(), offset, size, value);
const auto after = const auto after = Barrier(offset, size, vk::AccessFlagBits::eTransferWrite,
Barrier(offset, size, vk::AccessFlagBits::eTransferWrite, vk::AccessFlagBits::eMemoryRead | vk::AccessFlagBits::eMemoryWrite);
vk::AccessFlagBits::eMemoryRead | vk::AccessFlagBits::eMemoryWrite);
native.pipelineBarrier(vk::PipelineStageFlagBits::eTransfer, native.pipelineBarrier(vk::PipelineStageFlagBits::eTransfer,
vk::PipelineStageFlagBits::eAllCommands, vk::PipelineStageFlagBits::eAllCommands,
vk::DependencyFlagBits::eByRegion, 0, nullptr, 1, &after, 0, nullptr); vk::DependencyFlagBits::eByRegion, 0, nullptr, 1, &after, 0, nullptr);
@@ -250,8 +248,8 @@ std::pair<uint8_t*, uint64_t> StreamBuffer::Map(uint64_t size, uint64_t alignmen
if (Mapped().empty()) { if (Mapped().empty()) {
return {nullptr, 0}; return {nullptr, 0};
} }
uint64_t mapped_size = size; uint64_t mapped_size = size;
const auto atom = Graphics().physical_device_properties.limits.nonCoherentAtomSize; const auto atom = Graphics().physical_device_properties.limits.nonCoherentAtomSize;
if (!NormalizeReservation(IsCoherent(), atom, mapped_size, alignment)) { if (!NormalizeReservation(IsCoherent(), atom, mapped_size, alignment)) {
return {nullptr, 0}; return {nullptr, 0};
} }
+14 -15
View File
@@ -54,16 +54,15 @@ public:
[[nodiscard]] bool IsInBounds(uint64_t address, uint64_t size) const noexcept; [[nodiscard]] bool IsInBounds(uint64_t address, uint64_t size) const noexcept;
void Write(uint64_t offset, const void* source, uint64_t size); void Write(uint64_t offset, const void* source, uint64_t size);
void Flush(uint64_t offset, uint64_t size); void Flush(uint64_t offset, uint64_t size);
void CopyFrom( void CopyFrom(CommandBuffer& command, const Buffer& source, uint64_t source_offset,
CommandBuffer& command, const Buffer& source, uint64_t source_offset, uint64_t destination_offset, uint64_t size,
uint64_t destination_offset, uint64_t size, vk::AccessFlags source_before = vk::AccessFlagBits::eMemoryWrite,
vk::AccessFlags source_before = vk::AccessFlagBits::eMemoryWrite, vk::AccessFlags destination_before = vk::AccessFlagBits::eMemoryRead |
vk::AccessFlags destination_before = vk::AccessFlagBits::eMemoryWrite,
vk::AccessFlagBits::eMemoryRead | vk::AccessFlagBits::eMemoryWrite, vk::AccessFlags source_after = vk::AccessFlagBits::eMemoryRead |
vk::AccessFlags source_after = vk::AccessFlagBits::eMemoryWrite,
vk::AccessFlagBits::eMemoryRead | vk::AccessFlagBits::eMemoryWrite, vk::AccessFlags destination_after = vk::AccessFlagBits::eMemoryRead |
vk::AccessFlags destination_after = vk::AccessFlagBits::eMemoryWrite);
vk::AccessFlagBits::eMemoryRead | vk::AccessFlagBits::eMemoryWrite);
void Fill(uint64_t offset, uint64_t size, uint32_t value); void Fill(uint64_t offset, uint64_t size, uint32_t value);
protected: protected:
@@ -107,13 +106,13 @@ private:
uint64_t upper_bound = 0; uint64_t upper_bound = 0;
}; };
void ReserveWatches(std::vector<Watch>& watches, size_t grow_size); void ReserveWatches(std::vector<Watch>& watches, size_t grow_size);
[[nodiscard]] static bool NormalizeReservation(bool coherent, uint64_t atom, uint64_t& size, [[nodiscard]] static bool NormalizeReservation(bool coherent, uint64_t atom, uint64_t& size,
uint64_t& alignment); uint64_t& alignment);
[[nodiscard]] bool WaitPendingOperations(const std::vector<Watch>& watches, [[nodiscard]] bool WaitPendingOperations(const std::vector<Watch>& watches,
std::optional<size_t> invalidation_mark, std::optional<size_t> invalidation_mark,
uint64_t requested_upper_bound, bool allow_wait, uint64_t requested_upper_bound, bool allow_wait,
size_t& wait_cursor, uint64_t& wait_bound); size_t& wait_cursor, uint64_t& wait_bound);
uint64_t m_offset = 0; uint64_t m_offset = 0;
uint64_t m_mapped_size = 0; uint64_t m_mapped_size = 0;
@@ -7,8 +7,8 @@
#include "graphics/guest_gpu/hardwareContext.h" #include "graphics/guest_gpu/hardwareContext.h"
#include "graphics/guest_gpu/tile.h" #include "graphics/guest_gpu/tile.h"
#include "graphics/host_gpu/graphicContext.h" #include "graphics/host_gpu/graphicContext.h"
#include "graphics/host_gpu/renderer/image/textureCommon.h"
#include "graphics/host_gpu/renderer/debug.h" #include "graphics/host_gpu/renderer/debug.h"
#include "graphics/host_gpu/renderer/image/textureCommon.h"
#include "graphics/host_gpu/renderer/pipeline/descriptorCache.h" #include "graphics/host_gpu/renderer/pipeline/descriptorCache.h"
#include "graphics/host_gpu/renderer/render.h" #include "graphics/host_gpu/renderer/render.h"
#include "graphics/host_gpu/renderer/renderContext.h" #include "graphics/host_gpu/renderer/renderContext.h"
@@ -23,10 +23,10 @@ static std::atomic<uint32_t> g_render_color_log_count = 0;
// NOLINTNEXTLINE(readability-function-cognitive-complexity) // NOLINTNEXTLINE(readability-function-cognitive-complexity)
void RenderExecutor::ResolveRenderColorTarget(uint64_t submit_id, RenderCommandBuffer& buffer, void RenderExecutor::ResolveRenderColorTarget(uint64_t submit_id, RenderCommandBuffer& buffer,
RenderColorInfo& r, RenderColorInfo& r,
uint32_t render_target_slice_offset, uint32_t render_target_slice_offset,
uint32_t render_target_slot, bool ignore_target_mask, uint32_t render_target_slot, bool ignore_target_mask,
bool exact_format) { bool exact_format) {
KYTY_PROFILER_FUNCTION(); KYTY_PROFILER_FUNCTION();
const auto& hw = buffer.GetRegisters(); const auto& hw = buffer.GetRegisters();
@@ -79,10 +79,8 @@ void RenderExecutor::ResolveRenderColorTarget(uint64_t submit_id, RenderCommandB
const auto view = ResolveTargetViewInfo( const auto view = ResolveTargetViewInfo(
rt.view.base_array_slice_index, rt.view.last_array_slice_index, render_target_slice_offset); rt.view.base_array_slice_index, rt.view.last_array_slice_index, render_target_slice_offset);
switch (view.type) { switch (view.type) {
case TargetViewType::Image2D: break; case TargetViewType::Image2D:
case TargetViewType::Image2DArray: case TargetViewType::Image2DArray: break;
EXIT("layered render-target views are unsupported: base=%u count=%u\n", view.base_layer,
view.layer_count);
case TargetViewType::Unsupported: case TargetViewType::Unsupported:
EXIT("invalid render-target view: base=%u last=%u draw_offset=%u\n", EXIT("invalid render-target view: base=%u last=%u draw_offset=%u\n",
rt.view.base_array_slice_index, rt.view.last_array_slice_index, rt.view.base_array_slice_index, rt.view.last_array_slice_index,
@@ -241,12 +239,12 @@ void RenderExecutor::ResolveRenderColorTarget(uint64_t submit_id, RenderCommandB
} }
TextureCache::ImageDesc desc {}; TextureCache::ImageDesc desc {};
desc.type = TextureCache::BindingType::RenderTarget; desc.type = TextureCache::BindingType::RenderTarget;
desc.info.data = {rt.base.addr, backing_size}; desc.info.data = {rt.base.addr, backing_size};
desc.info.pixel_format = target_format.format; desc.info.pixel_format = target_format.format;
desc.info.guest_format = ImageOps::RenderTargetTransferFormat(bytes_per_element); desc.info.guest_format = ImageOps::RenderTargetTransferFormat(bytes_per_element);
desc.info.type = Prospero::ImageType::kColor2D; desc.info.type = Prospero::ImageType::kColor2D;
desc.info.extent = {width, height, 1}; desc.info.extent = {width, height, 1};
desc.info.resources = {levels, view.image_layers}; desc.info.resources = {levels, view.image_layers};
desc.info.pitch = pitch; desc.info.pitch = pitch;
desc.info.bytes_per_block = bytes_per_element; desc.info.bytes_per_block = bytes_per_element;
@@ -275,20 +273,20 @@ void RenderExecutor::ResolveRenderColorTarget(uint64_t submit_id, RenderCommandB
desc.view_info.base_layer = view.base_layer; desc.view_info.base_layer = view.base_layer;
desc.view_info.layer_count = view.layer_count; desc.view_info.layer_count = view.layer_count;
desc.view_info.usage = vk::ImageUsageFlagBits::eColorAttachment; desc.view_info.usage = vk::ImageUsageFlagBits::eColorAttachment;
auto& texture_cache = m_context.GetTextureCache(); auto& texture_cache = m_context.GetTextureCache();
r.desc = std::move(desc); r.desc = std::move(desc);
r.image_id = texture_cache.FindImage(r.desc, exact_format); r.image_id = texture_cache.FindImage(r.desc, exact_format);
r.type = RenderColorType::RenderTexture; r.type = RenderColorType::RenderTexture;
r.base_addr = rt.base.addr; r.base_addr = rt.base.addr;
r.image_view = nullptr; r.image_view = nullptr;
r.format = r.desc.view_info.format; r.format = r.desc.view_info.format;
r.extent = view_extent; r.extent = view_extent;
r.base_mip_level = rt.view.current_mip_level; r.base_mip_level = rt.view.current_mip_level;
r.buffer_size = backing_size; r.buffer_size = backing_size;
r.samples = samples; r.samples = samples;
r.export_mapping = target_format.export_mapping; r.export_mapping = target_format.export_mapping;
r.color_clear_enable = false; r.color_clear_enable = false;
r.color_clear_value = {}; r.color_clear_value = {};
BindRenderTarget(r.image_id); BindRenderTarget(r.image_id);
} }
@@ -2,8 +2,8 @@
#define EMULATOR_SRC_GRAPHICS_HOST_GPU_RENDERER_COLORRENDERTARGET_H_ #define EMULATOR_SRC_GRAPHICS_HOST_GPU_RENDERER_COLORRENDERTARGET_H_
#include "graphics/guest_gpu/gpu_defs.h" #include "graphics/guest_gpu/gpu_defs.h"
#include "graphics/host_gpu/renderer/renderTarget.h"
#include "graphics/host_gpu/renderer/cache/textureCache.h" #include "graphics/host_gpu/renderer/cache/textureCache.h"
#include "graphics/host_gpu/renderer/renderTarget.h"
#include "graphics/host_gpu/vulkanCommon.h" #include "graphics/host_gpu/vulkanCommon.h"
#include <cstdint> #include <cstdint>
@@ -44,13 +44,13 @@ CommandSlot* CommandScheduler::CommandPool::CreateSlot() {
allocate.commandPool = m_pool; allocate.commandPool = m_pool;
allocate.level = vk::CommandBufferLevel::ePrimary; allocate.level = vk::CommandBufferLevel::ePrimary;
allocate.commandBufferCount = 1; allocate.commandBufferCount = 1;
vk::CommandBuffer buffer = nullptr; vk::CommandBuffer buffer = nullptr;
EXIT_IF(graphics.device.allocateCommandBuffers(&allocate, &buffer) != vk::Result::eSuccess); EXIT_IF(graphics.device.allocateCommandBuffers(&allocate, &buffer) != vk::Result::eSuccess);
vk::FenceCreateInfo fence_create {}; vk::FenceCreateInfo fence_create {};
fence_create.sType = vk::StructureType::eFenceCreateInfo; fence_create.sType = vk::StructureType::eFenceCreateInfo;
fence_create.flags = vk::FenceCreateFlagBits::eSignaled; fence_create.flags = vk::FenceCreateFlagBits::eSignaled;
vk::Fence fence = nullptr; vk::Fence fence = nullptr;
if (graphics.device.createFence(&fence_create, nullptr, &fence) != vk::Result::eSuccess) { if (graphics.device.createFence(&fence_create, nullptr, &fence) != vk::Result::eSuccess) {
graphics.device.freeCommandBuffers(m_pool, 1, &buffer); graphics.device.freeCommandBuffers(m_pool, 1, &buffer);
EXIT("failed to create command-buffer fence\n"); EXIT("failed to create command-buffer fence\n");
@@ -70,9 +70,9 @@ CommandSlot* CommandScheduler::CommandPool::Allocate(GraphicContext& graphics) {
Create(graphics); Create(graphics);
} }
EXIT_IF(m_graphics != &graphics); EXIT_IF(m_graphics != &graphics);
auto found = std::ranges::find_if(m_slots, [](const auto& slot) { return !slot.busy; }); auto found = std::ranges::find_if(m_slots, [](const auto& slot) { return !slot.busy; });
auto* slot = found != m_slots.end() ? &*found : CreateSlot(); auto* slot = found != m_slots.end() ? &*found : CreateSlot();
slot->busy = true; slot->busy = true;
slot->Reset(); slot->Reset();
return slot; return slot;
} }
@@ -331,8 +331,7 @@ void CommandScheduler::WaitPriorityOperations(uint64_t tick) {
EXIT_IF(g_deferred_callback_scheduler == this); EXIT_IF(g_deferred_callback_scheduler == this);
std::unique_lock lock(m_operation_mutex); std::unique_lock lock(m_operation_mutex);
m_operation_available.wait(lock, [this, tick] { m_operation_available.wait(lock, [this, tick] {
const bool active_before_or_at = const bool active_before_or_at = m_priority_active && m_priority_active_tick <= tick;
m_priority_active && m_priority_active_tick <= tick;
const bool queued_before_or_at = const bool queued_before_or_at =
!m_priority_operations.empty() && m_priority_operations.front().tick <= tick; !m_priority_operations.empty() && m_priority_operations.front().tick <= tick;
return !active_before_or_at && !queued_before_or_at; return !active_before_or_at && !queued_before_or_at;
@@ -47,21 +47,21 @@ public:
void FinishCurrent(); void FinishCurrent();
// Deferred callbacks can observe an externally owned drain, but cannot initiate shutdown: // Deferred callbacks can observe an externally owned drain, but cannot initiate shutdown:
// the priority runner cannot join itself. // the priority runner cannot join itself.
void Shutdown(); void Shutdown();
void Wait(uint64_t tick); void Wait(uint64_t tick);
void PopPendingOperations(); void PopPendingOperations();
void DrainPriorityOperations(); void DrainPriorityOperations();
void WaitPriorityOperations(uint64_t tick); void WaitPriorityOperations(uint64_t tick);
void DeferOperation(Common::UniqueFunction<void>&& operation); void DeferOperation(Common::UniqueFunction<void>&& operation);
void DeferPriorityOperation(Common::UniqueFunction<void>&& operation); void DeferPriorityOperation(Common::UniqueFunction<void>&& operation);
[[nodiscard]] static bool InDeferredOperation() noexcept; [[nodiscard]] static bool InDeferredOperation() noexcept;
[[nodiscard]] bool Active() const noexcept { return m_current >= 0; } [[nodiscard]] bool Active() const noexcept { return m_current >= 0; }
void CheckActive() const; void CheckActive() const;
RenderCommandBuffer& Current() const; RenderCommandBuffer& Current() const;
[[nodiscard]] uint64_t CurrentTick() const noexcept { return m_master.CurrentTick(); } [[nodiscard]] uint64_t CurrentTick() const noexcept { return m_master.CurrentTick(); }
[[nodiscard]] bool IsFree(uint64_t tick); [[nodiscard]] bool IsFree(uint64_t tick);
[[nodiscard]] RenderContext& Context() const noexcept { return m_context; } [[nodiscard]] RenderContext& Context() const noexcept { return m_context; }
[[nodiscard]] GraphicContext& Graphics() const noexcept { return m_graphics; } [[nodiscard]] GraphicContext& Graphics() const noexcept { return m_graphics; }
private: private:
@@ -91,11 +91,11 @@ private:
uint64_t tick = 0; uint64_t tick = 0;
}; };
void BindCurrent() const; void BindCurrent() const;
CommandBuffer& SubmitCurrent(SubmitInfo& submit); CommandBuffer& SubmitCurrent(SubmitInfo& submit);
void BeginNext(); void BeginNext();
void PriorityOperationsThread(std::stop_token stop); void PriorityOperationsThread(std::stop_token stop);
void RunOperation(Common::UniqueFunction<void>&& operation); void RunOperation(Common::UniqueFunction<void>&& operation);
[[nodiscard]] CommandSlot* AllocateCommandBuffer(); [[nodiscard]] CommandSlot* AllocateCommandBuffer();
[[nodiscard]] uint64_t NextSubmitSequence() noexcept; [[nodiscard]] uint64_t NextSubmitSequence() noexcept;
@@ -109,14 +109,14 @@ private:
std::mutex m_operation_mutex; std::mutex m_operation_mutex;
std::condition_variable m_operation_available; std::condition_variable m_operation_available;
std::jthread m_priority_thread; std::jthread m_priority_thread;
bool m_priority_active = false; bool m_priority_active = false;
uint64_t m_priority_active_tick = 0; uint64_t m_priority_active_tick = 0;
OperationState m_operation_state = OperationState::Open; OperationState m_operation_state = OperationState::Open;
int m_current = -1; int m_current = -1;
bool m_recording = false; bool m_recording = false;
HW::Context* m_registers = nullptr; HW::Context* m_registers = nullptr;
HW::UserConfig* m_user_config = nullptr; HW::UserConfig* m_user_config = nullptr;
HW::Shader* m_shaders = nullptr; HW::Shader* m_shaders = nullptr;
std::atomic<uint64_t> m_submit_sequence = 0; std::atomic<uint64_t> m_submit_sequence = 0;
friend class CommandBuffer; friend class CommandBuffer;
+13 -13
View File
@@ -8,8 +8,8 @@
#include "graphics/host_gpu/renderer/colorRenderTarget.h" #include "graphics/host_gpu/renderer/colorRenderTarget.h"
#include "graphics/host_gpu/renderer/debug.h" #include "graphics/host_gpu/renderer/debug.h"
#include "graphics/host_gpu/renderer/depthRenderTarget.h" #include "graphics/host_gpu/renderer/depthRenderTarget.h"
#include "graphics/host_gpu/renderer/pipeline/descriptorCache.h"
#include "graphics/host_gpu/renderer/image/imageView.h" #include "graphics/host_gpu/renderer/image/imageView.h"
#include "graphics/host_gpu/renderer/pipeline/descriptorCache.h"
#include "graphics/host_gpu/renderer/render.h" #include "graphics/host_gpu/renderer/render.h"
#include "graphics/host_gpu/renderer/renderContext.h" #include "graphics/host_gpu/renderer/renderContext.h"
#include "graphics/host_gpu/vma.h" #include "graphics/host_gpu/vma.h"
@@ -270,30 +270,30 @@ void CommandBuffer::BeginRendering(const RenderState& state) const {
colors[i].sType = vk::StructureType::eRenderingAttachmentInfo; colors[i].sType = vk::StructureType::eRenderingAttachmentInfo;
colors[i].imageView = attachment.image_view; colors[i].imageView = attachment.image_view;
colors[i].imageLayout = attachment.image_layout; colors[i].imageLayout = attachment.image_layout;
colors[i].loadOp = attachment.is_clear ? vk::AttachmentLoadOp::eClear colors[i].loadOp =
: vk::AttachmentLoadOp::eLoad; attachment.is_clear ? vk::AttachmentLoadOp::eClear : vk::AttachmentLoadOp::eLoad;
colors[i].storeOp = vk::AttachmentStoreOp::eStore; colors[i].storeOp = vk::AttachmentStoreOp::eStore;
colors[i].clearValue.color.uint32 = attachment.clear_value; colors[i].clearValue.color.uint32 = attachment.clear_value;
} }
const auto& depth_stencil = state.depth_stencil_attachment; const auto& depth_stencil = state.depth_stencil_attachment;
vk::RenderingAttachmentInfo depth {}; vk::RenderingAttachmentInfo depth {};
depth.sType = vk::StructureType::eRenderingAttachmentInfo; depth.sType = vk::StructureType::eRenderingAttachmentInfo;
depth.imageView = depth_stencil.image_view; depth.imageView = depth_stencil.image_view;
depth.imageLayout = depth_stencil.image_layout; depth.imageLayout = depth_stencil.image_layout;
depth.loadOp = depth_stencil.depth_clear ? vk::AttachmentLoadOp::eClear depth.loadOp =
: vk::AttachmentLoadOp::eLoad; depth_stencil.depth_clear ? vk::AttachmentLoadOp::eClear : vk::AttachmentLoadOp::eLoad;
depth.storeOp = vk::AttachmentStoreOp::eStore; depth.storeOp = vk::AttachmentStoreOp::eStore;
depth.clearValue.depthStencil.depth = std::bit_cast<float>(depth_stencil.clear_value[0]); depth.clearValue.depthStencil.depth = std::bit_cast<float>(depth_stencil.clear_value[0]);
vk::RenderingAttachmentInfo stencil {}; vk::RenderingAttachmentInfo stencil {};
stencil.sType = vk::StructureType::eRenderingAttachmentInfo; stencil.sType = vk::StructureType::eRenderingAttachmentInfo;
stencil.imageView = depth_stencil.image_view; stencil.imageView = depth_stencil.image_view;
stencil.imageLayout = depth_stencil.image_layout; stencil.imageLayout = depth_stencil.image_layout;
stencil.loadOp = depth_stencil.stencil_clear ? vk::AttachmentLoadOp::eClear stencil.loadOp =
: vk::AttachmentLoadOp::eLoad; depth_stencil.stencil_clear ? vk::AttachmentLoadOp::eClear : vk::AttachmentLoadOp::eLoad;
stencil.storeOp = vk::AttachmentStoreOp::eStore; stencil.storeOp = vk::AttachmentStoreOp::eStore;
stencil.clearValue.depthStencil.stencil = depth_stencil.clear_value[1]; stencil.clearValue.depthStencil.stencil = depth_stencil.clear_value[1];
vk::RenderingInfo rendering {}; vk::RenderingInfo rendering {};
rendering.sType = vk::StructureType::eRenderingInfo; rendering.sType = vk::StructureType::eRenderingInfo;
-8
View File
@@ -548,14 +548,6 @@ static void ZCheck(const HW::DepthRenderTarget& z) {
EXIT_NOT_IMPLEMENTED(z.htile_surface.prefetch_height != 0x00000000); EXIT_NOT_IMPLEMENTED(z.htile_surface.prefetch_height != 0x00000000);
EXIT_NOT_IMPLEMENTED(z.htile_surface.dst_outside_zero_to_one != 0x00000000); EXIT_NOT_IMPLEMENTED(z.htile_surface.dst_outside_zero_to_one != 0x00000000);
if (z.depth_view.slice_start != 0x00000000 || z.depth_view.slice_max != 0x00000000) {
static std::atomic<uint32_t> log_count {0};
if (log_count.fetch_add(1, std::memory_order_relaxed) < 16) {
LOGF("DepthTarget: temporary: ignoring PS5 array slice view start=0x%08" PRIx32
", max=0x%08" PRIx32 "\n",
z.depth_view.slice_start, z.depth_view.slice_max);
}
}
if (z.depth_view.current_mip_level != 0x00000000) { if (z.depth_view.current_mip_level != 0x00000000) {
static std::atomic<uint32_t> log_count {0}; static std::atomic<uint32_t> log_count {0};
if (log_count.fetch_add(1, std::memory_order_relaxed) < 16) { if (log_count.fetch_add(1, std::memory_order_relaxed) < 16) {
@@ -10,10 +10,10 @@
#include "graphics/guest_gpu/hardwareContext.h" #include "graphics/guest_gpu/hardwareContext.h"
#include "graphics/guest_gpu/tile.h" #include "graphics/guest_gpu/tile.h"
#include "graphics/host_gpu/graphicContext.h" #include "graphics/host_gpu/graphicContext.h"
#include "graphics/host_gpu/renderer/image/textureCommon.h"
#include "graphics/host_gpu/renderer/debug.h" #include "graphics/host_gpu/renderer/debug.h"
#include "graphics/host_gpu/renderer/pipeline/descriptorCache.h"
#include "graphics/host_gpu/renderer/image/imageView.h" #include "graphics/host_gpu/renderer/image/imageView.h"
#include "graphics/host_gpu/renderer/image/textureCommon.h"
#include "graphics/host_gpu/renderer/pipeline/descriptorCache.h"
#include "graphics/host_gpu/renderer/render.h" #include "graphics/host_gpu/renderer/render.h"
#include "graphics/host_gpu/renderer/renderContext.h" #include "graphics/host_gpu/renderer/renderContext.h"
#include "graphics/host_gpu/vulkanCommon.h" #include "graphics/host_gpu/vulkanCommon.h"
@@ -150,10 +150,8 @@ void RenderExecutor::ResolveRenderDepthTarget(uint64_t submit_id, RenderCommandB
has_stencil, has_htile, z.stencil_info.htile_stencil_disabled); has_stencil, has_htile, z.stencil_info.htile_stencil_disabled);
const auto view = ResolveTargetViewInfo(z.depth_view.slice_start, z.depth_view.slice_max); const auto view = ResolveTargetViewInfo(z.depth_view.slice_start, z.depth_view.slice_max);
switch (view.type) { switch (view.type) {
case TargetViewType::Image2D: break; case TargetViewType::Image2D:
case TargetViewType::Image2DArray: case TargetViewType::Image2DArray: break;
DepthFatal("layered depth views are unsupported: base=%u count=%u", view.base_layer,
view.layer_count);
case TargetViewType::Unsupported: case TargetViewType::Unsupported:
DepthFatal("invalid depth view: base=%u last=%u", z.depth_view.slice_start, DepthFatal("invalid depth view: base=%u last=%u", z.depth_view.slice_start,
z.depth_view.slice_max); z.depth_view.slice_max);
@@ -2,9 +2,9 @@
#define EMULATOR_SRC_GRAPHICS_HOST_GPU_RENDERER_DEPTHRENDERTARGET_H_ #define EMULATOR_SRC_GRAPHICS_HOST_GPU_RENDERER_DEPTHRENDERTARGET_H_
#include "common/assert.h" #include "common/assert.h"
#include "graphics/host_gpu/renderer/cache/textureCache.h"
#include "graphics/host_gpu/renderer/image/imageView.h" #include "graphics/host_gpu/renderer/image/imageView.h"
#include "graphics/host_gpu/renderer/renderTarget.h" #include "graphics/host_gpu/renderer/renderTarget.h"
#include "graphics/host_gpu/renderer/cache/textureCache.h"
#include "graphics/host_gpu/vulkanCommon.h" #include "graphics/host_gpu/vulkanCommon.h"
#include <cstdint> #include <cstdint>
@@ -182,8 +182,8 @@ void BlitHelper::ReinterpretColorAsMsDepth(Image& source, Image& destination) {
auto command = command_buffer.Handle(); auto command = command_buffer.Handle();
source.Transit(vk::ImageLayout::eShaderReadOnlyOptimal, vk::AccessFlagBits2::eShaderRead, {}, source.Transit(vk::ImageLayout::eShaderReadOnlyOptimal, vk::AccessFlagBits2::eShaderRead, {},
command); command);
destination.Transit(ColorToMsDepthLayout, destination.Transit(ColorToMsDepthLayout, vk::AccessFlagBits2::eDepthStencilAttachmentWrite, {},
vk::AccessFlagBits2::eDepthStencilAttachmentWrite, {}, command); command);
vk::RenderingAttachmentInfo depth_attachment {}; vk::RenderingAttachmentInfo depth_attachment {};
depth_attachment.sType = vk::StructureType::eRenderingAttachmentInfo; depth_attachment.sType = vk::StructureType::eRenderingAttachmentInfo;
@@ -19,10 +19,9 @@ struct GuestRange {
uint64_t address = 0; uint64_t address = 0;
uint64_t size = 0; uint64_t size = 0;
[[nodiscard]] constexpr bool Empty() const noexcept { return address == 0 || size == 0; } [[nodiscard]] constexpr bool Empty() const noexcept { return address == 0 || size == 0; }
[[nodiscard]] constexpr bool Valid() const noexcept { [[nodiscard]] constexpr bool Valid() const noexcept {
return !Empty() && address < TRACKER_ADDRESS_SIZE && return !Empty() && address < TRACKER_ADDRESS_SIZE && size <= TRACKER_ADDRESS_SIZE - address;
size <= TRACKER_ADDRESS_SIZE - address;
} }
[[nodiscard]] constexpr uint64_t End() const noexcept { return address + size; } [[nodiscard]] constexpr uint64_t End() const noexcept { return address + size; }
auto operator<=>(const GuestRange&) const = default; auto operator<=>(const GuestRange&) const = default;
@@ -47,10 +46,10 @@ struct ImageSubresources {
}; };
struct ImageSubresourceRange { struct ImageSubresourceRange {
uint32_t base_level = 0; uint32_t base_level = 0;
uint32_t level_count = 1; uint32_t level_count = 1;
uint32_t base_layer = 0; uint32_t base_layer = 0;
uint32_t layer_count = 1; uint32_t layer_count = 1;
auto operator<=>(const ImageSubresourceRange&) const = default; auto operator<=>(const ImageSubresourceRange&) const = default;
}; };
@@ -67,10 +66,10 @@ struct ImageInfo {
GuestRange stencil; GuestRange stencil;
ImageMetadataInfo metadata; ImageMetadataInfo metadata;
uint32_t htile_clear_mask = UINT32_MAX; uint32_t htile_clear_mask = UINT32_MAX;
vk::Format pixel_format = vk::Format::eUndefined; vk::Format pixel_format = vk::Format::eUndefined;
uint32_t guest_format = 0; uint32_t guest_format = 0;
Prospero::ImageType type = Prospero::ImageType::kColor2D; Prospero::ImageType type = Prospero::ImageType::kColor2D;
vk::Extent3D extent = {1, 1, 1}; vk::Extent3D extent = {1, 1, 1};
ImageSubresources resources; ImageSubresources resources;
uint32_t pitch = 0; uint32_t pitch = 0;
uint32_t bytes_per_block = 0; uint32_t bytes_per_block = 0;
@@ -352,8 +351,7 @@ inline bool ImageInfo::IsDepth() const noexcept {
} }
const auto transfer_bytes = DepthAspectTransferBytes(info.pixel_format); const auto transfer_bytes = DepthAspectTransferBytes(info.pixel_format);
return transfer_bytes == info.bytes_per_block || return transfer_bytes == info.bytes_per_block ||
(info.bytes_per_block == sizeof(uint16_t) && (info.bytes_per_block == sizeof(uint16_t) && transfer_bytes == sizeof(uint32_t));
transfer_bytes == sizeof(uint32_t));
} }
[[nodiscard]] inline VideoOutCompression [[nodiscard]] inline VideoOutCompression
@@ -470,18 +468,13 @@ IsSupportedDisplayRenderTargetTileMode(uint32_t tile_mode) noexcept {
vk::ClearColorValue& clear) { vk::ClearColorValue& clear) {
vk::ClearColorValue next {}; vk::ClearColorValue next {};
const auto unorm8 = [](uint32_t value) { return static_cast<float>(value & 0xffu) / 255.0f; }; const auto unorm8 = [](uint32_t value) { return static_cast<float>(value & 0xffu) / 255.0f; };
const auto srgb8 = [](uint32_t value) { const auto srgb8 = [](uint32_t value) {
const auto encoded = static_cast<float>(value & 0xffu) / 255.0f; const auto encoded = static_cast<float>(value & 0xffu) / 255.0f;
return encoded <= 0.04045f ? encoded / 12.92f return encoded <= 0.04045f ? encoded / 12.92f : std::pow((encoded + 0.055f) / 1.055f, 2.4f);
: std::pow((encoded + 0.055f) / 1.055f, 2.4f);
}; };
switch (format) { switch (format) {
case vk::Format::eR32Uint: case vk::Format::eR32Uint: next.uint32[0] = packed; break;
next.uint32[0] = packed; case vk::Format::eR32Sint: next.int32[0] = static_cast<int32_t>(packed); break;
break;
case vk::Format::eR32Sint:
next.int32[0] = static_cast<int32_t>(packed);
break;
case vk::Format::eR8G8B8A8Srgb: case vk::Format::eR8G8B8A8Srgb:
next.float32[0] = srgb8(packed); next.float32[0] = srgb8(packed);
next.float32[1] = srgb8(packed >> 8u); next.float32[1] = srgb8(packed >> 8u);
@@ -70,15 +70,14 @@ namespace {
} }
case vk::ImageType::e3D: case vk::ImageType::e3D:
switch (info.type) { switch (info.type) {
case vk::ImageViewType::e3D: case vk::ImageViewType::e3D: return info.base_layer == 0 && info.layer_count == 1;
return info.base_layer == 0 && info.layer_count == 1;
case vk::ImageViewType::e2D: case vk::ImageViewType::e2D:
return static_cast<bool>( return static_cast<bool>(image.flags &
image.flags & vk::ImageCreateFlagBits::e2DArrayCompatible) && vk::ImageCreateFlagBits::e2DArrayCompatible) &&
info.level_count == 1 && info.layer_count == 1; info.level_count == 1 && info.layer_count == 1;
case vk::ImageViewType::e2DArray: case vk::ImageViewType::e2DArray:
return static_cast<bool>( return static_cast<bool>(image.flags &
image.flags & vk::ImageCreateFlagBits::e2DArrayCompatible) && vk::ImageCreateFlagBits::e2DArrayCompatible) &&
info.level_count == 1; info.level_count == 1;
default: return false; default: return false;
} }
@@ -325,11 +324,10 @@ bool FormatsCompatible(vk::Format base, vk::Format view) noexcept {
} // namespace ImageViewOps } // namespace ImageViewOps
vk::ImageView Image::FindView(const ImageViewInfo& view_info) { vk::ImageView Image::FindView(const ImageViewInfo& view_info) {
const auto& image = backing; const auto& image = backing;
auto normalized = view_info; auto normalized = view_info;
const bool is_storage = const bool is_storage = static_cast<bool>(normalized.usage & vk::ImageUsageFlagBits::eStorage);
static_cast<bool>(normalized.usage & vk::ImageUsageFlagBits::eStorage); normalized.aspect = FullAspectMask(image.format);
normalized.aspect = FullAspectMask(image.format);
if (normalized.aspect & vk::ImageAspectFlagBits::eDepth && if (normalized.aspect & vk::ImageAspectFlagBits::eDepth &&
IsDepthViewFormat(normalized.format)) { IsDepthViewFormat(normalized.format)) {
normalized.format = image.format; normalized.format = image.format;
@@ -340,28 +338,26 @@ vk::ImageView Image::FindView(const ImageViewInfo& view_info) {
normalized.format = image.format; normalized.format = image.format;
normalized.aspect = vk::ImageAspectFlagBits::eStencil; normalized.aspect = vk::ImageAspectFlagBits::eStencil;
} }
normalized.usage = normalized.usage = is_storage ? vk::ImageUsageFlagBits::eStorage : vk::ImageUsageFlags {};
is_storage ? vk::ImageUsageFlagBits::eStorage : vk::ImageUsageFlags {};
const bool format_compatible = normalized.format != vk::Format::eUndefined && const bool format_compatible = normalized.format != vk::Format::eUndefined &&
IsCompatibleViewFormat(image.format, normalized.format); IsCompatibleViewFormat(image.format, normalized.format);
const bool slice_view = image.image_type == vk::ImageType::e3D && const bool slice_view =
(normalized.type == vk::ImageViewType::e2D || image.image_type == vk::ImageType::e3D && (normalized.type == vk::ImageViewType::e2D ||
normalized.type == vk::ImageViewType::e2DArray); normalized.type == vk::ImageViewType::e2DArray);
const bool levels_valid = normalized.level_count != 0 && const bool levels_valid = normalized.level_count != 0 &&
normalized.base_level < image.mip_levels && normalized.base_level < image.mip_levels &&
normalized.level_count <= image.mip_levels - normalized.base_level; normalized.level_count <= image.mip_levels - normalized.base_level;
const auto view_layers = slice_view && levels_valid const auto view_layers = slice_view && levels_valid
? std::max(image.extent.depth >> normalized.base_level, 1u) ? std::max(image.extent.depth >> normalized.base_level, 1u)
: image.layers; : image.layers;
const bool ranges_valid = levels_valid && const bool ranges_valid = levels_valid && normalized.layer_count != 0 &&
normalized.layer_count != 0 && normalized.base_layer < view_layers && normalized.base_layer < view_layers &&
normalized.layer_count <= view_layers - normalized.base_layer; normalized.layer_count <= view_layers - normalized.base_layer;
const bool mapping_valid = const bool mapping_valid =
IsComponentSwizzle(normalized.mapping.r) && IsComponentSwizzle(normalized.mapping.g) && IsComponentSwizzle(normalized.mapping.r) && IsComponentSwizzle(normalized.mapping.g) &&
IsComponentSwizzle(normalized.mapping.b) && IsComponentSwizzle(normalized.mapping.a); IsComponentSwizzle(normalized.mapping.b) && IsComponentSwizzle(normalized.mapping.a);
if (image.image == nullptr || !format_compatible || !ranges_valid || !mapping_valid || if (image.image == nullptr || !format_compatible || !ranges_valid || !mapping_valid ||
!IsValidViewType(image, normalized) || !IsValidViewType(image, normalized) || !IsValidAspect(image, normalized.aspect)) {
!IsValidAspect(image, normalized.aspect)) {
EXIT("invalid image view: image_format=%d view_format=%d type=%d aspect=0x%x " EXIT("invalid image view: image_format=%d view_format=%d type=%d aspect=0x%x "
"mip=%u+%u layer=%u+%u usage=0x%x image_levels=%u image_layers=%u\n", "mip=%u+%u layer=%u+%u usage=0x%x image_levels=%u image_layers=%u\n",
static_cast<int>(image.format), static_cast<int>(normalized.format), static_cast<int>(image.format), static_cast<int>(normalized.format),
@@ -88,7 +88,9 @@ SelectSampledDepthView(vk::Format image_format, vk::Format view_format, uint32_t
IsSupportedSampledDepthResource(const ShaderRecompiler::IR::ImageResource& resource) noexcept { IsSupportedSampledDepthResource(const ShaderRecompiler::IR::ImageResource& resource) noexcept {
return resource.kind == ShaderRecompiler::IR::ResourceKind::Image && return resource.kind == ShaderRecompiler::IR::ResourceKind::Image &&
(resource.dimension == ShaderRecompiler::Decoder::ImageDimension::Dim2D || (resource.dimension == ShaderRecompiler::Decoder::ImageDimension::Dim2D ||
resource.dimension == ShaderRecompiler::Decoder::ImageDimension::Dim2DArray) && resource.dimension == ShaderRecompiler::Decoder::ImageDimension::Dim2DArray ||
resource.dimension == ShaderRecompiler::Decoder::ImageDimension::Dim2DMsaa ||
resource.dimension == ShaderRecompiler::Decoder::ImageDimension::Dim2DMsaaArray) &&
resource.mip_mode == ShaderRecompiler::IR::ImageMipMode::None && resource.read && resource.mip_mode == ShaderRecompiler::IR::ImageMipMode::None && resource.read &&
!resource.written && !resource.atomic; !resource.written && !resource.atomic;
} }
@@ -397,10 +397,10 @@ TextureUploadLayout TextureCalcUploadLayout(uint32_t fmt, uint64_t width, uint64
return layout; return layout;
} }
std::vector<vk::BufferImageCopy> std::vector<vk::BufferImageCopy> TextureBuildImageCopies(const TextureUploadLayout& layout,
TextureBuildImageCopies(const TextureUploadLayout& layout, uint32_t width, uint32_t height, uint32_t width, uint32_t height,
uint32_t depth, uint64_t levels, bool array_texture, uint32_t depth, uint64_t levels,
bool volume_texture) { bool array_texture, bool volume_texture) {
uint32_t mip_width = width; uint32_t mip_width = width;
uint32_t mip_height = height; uint32_t mip_height = height;
uint32_t mip_pitch = volume_texture && static_cast<Prospero::TileMode>(layout.tile) != uint32_t mip_pitch = volume_texture && static_cast<Prospero::TileMode>(layout.tile) !=
@@ -416,14 +416,13 @@ TextureBuildImageCopies(const TextureUploadLayout& layout, uint32_t width, uint3
const auto mip_depth = GetTextureLevelDepth(depth, i, volume_texture); const auto mip_depth = GetTextureLevelDepth(depth, i, volume_texture);
for (uint32_t z = 0; z < mip_depth; z++) { for (uint32_t z = 0; z < mip_depth; z++) {
const auto slice_offset = z * layout.slice_stride; const auto slice_offset = z * layout.slice_stride;
vk::BufferImageCopy region {}; vk::BufferImageCopy region {};
region.bufferOffset = region.bufferOffset = layout.level_sizes[i].offset + slice_offset;
layout.level_sizes[i].offset + slice_offset; region.imageSubresource = {vk::ImageAspectFlagBits::eColor, i, array_texture ? z : 0,
region.imageSubresource = {vk::ImageAspectFlagBits::eColor, i, 1};
array_texture ? z : 0, 1}; region.imageOffset.z = volume_texture ? static_cast<int>(z) : 0;
region.imageOffset.z = volume_texture ? static_cast<int>(z) : 0; region.imageExtent = {mip_width, mip_height, 1};
region.imageExtent = {mip_width, mip_height, 1};
const bool linear = const bool linear =
static_cast<Prospero::TileMode>(layout.tile) == Prospero::TileMode::kLinear; static_cast<Prospero::TileMode>(layout.tile) == Prospero::TileMode::kLinear;
if (linear) { if (linear) {
@@ -433,9 +432,8 @@ TextureBuildImageCopies(const TextureUploadLayout& layout, uint32_t width, uint3
const auto align = [](uint32_t value, uint32_t block) { const auto align = [](uint32_t value, uint32_t block) {
return ((value + block - 1u) / block) * block; return ((value + block - 1u) / block) * block;
}; };
const auto pitch = align(mip_pitch, layout.texel_block); const auto pitch = align(mip_pitch, layout.texel_block);
region.bufferRowLength = region.bufferRowLength = pitch > align(mip_width, layout.texel_block) ? pitch : 0;
pitch > align(mip_width, layout.texel_block) ? pitch : 0;
} }
regions.push_back(region); regions.push_back(region);
} }
@@ -480,8 +478,7 @@ static bool SetGpuTileSize(uint64_t offset, uint64_t length, uint64_t capacity,
return true; return true;
} }
bool TextureBuildGpuTileInfos(uint64_t size, bool TextureBuildGpuTileInfos(uint64_t size, const std::vector<vk::BufferImageCopy>& regions,
const std::vector<vk::BufferImageCopy>& regions,
const TextureUploadLayout& layout, uint32_t fmt, uint32_t depth, const TextureUploadLayout& layout, uint32_t fmt, uint32_t depth,
uint64_t levels, std::vector<GpuTileInfo>& out_infos) { uint64_t levels, std::vector<GpuTileInfo>& out_infos) {
if (size == 0 || levels == 0 || levels > 16 || depth == 0 || if (size == 0 || levels == 0 || levels > 16 || depth == 0 ||
@@ -522,13 +519,12 @@ bool TextureBuildGpuTileInfos(uint64_t size,
for (uint32_t z = 0; z < mip_depth; z += block.block_depth) { for (uint32_t z = 0; z < mip_depth; z += block.block_depth) {
const uint32_t copy_depth = std::min(block.block_depth, mip_depth - z); const uint32_t copy_depth = std::min(block.block_depth, mip_depth - z);
const auto& region = regions[region_base + z]; const auto& region = regions[region_base + z];
const auto pitch = region.bufferRowLength != 0 const auto pitch =
? region.bufferRowLength region.bufferRowLength != 0 ? region.bufferRowLength : region.imageExtent.width;
: region.imageExtent.width; const auto logical_height = region.bufferImageHeight != 0
const auto logical_height = region.bufferImageHeight != 0 ? region.bufferImageHeight
? region.bufferImageHeight : region.imageExtent.height;
: region.imageExtent.height; GpuTileInfo info {};
GpuTileInfo info {};
info.family = block.family; info.family = block.family;
info.bytes_per_element = block.bytes_per_element; info.bytes_per_element = block.bytes_per_element;
info.linear_offset = region.bufferOffset; info.linear_offset = region.bufferOffset;
@@ -544,20 +540,17 @@ bool TextureBuildGpuTileInfos(uint64_t size,
return false; return false;
} }
info.linear_slice_stride = linear_stride; info.linear_slice_stride = linear_stride;
info.width = std::max( info.width =
(region.imageExtent.width + element.wide - 1u) / element.wide, 1u); std::max((region.imageExtent.width + element.wide - 1u) / element.wide, 1u);
info.height = std::max( info.height = std::max((logical_height + element.tall - 1u) / element.tall, 1u);
(logical_height + element.tall - 1u) / element.tall, 1u); info.depth = copy_depth;
info.depth = copy_depth; info.surface_z =
info.surface_z = block.block_depth == 1 block.block_depth == 1 ? static_cast<uint32_t>(region.imageOffset.z) : 0;
? static_cast<uint32_t>(region.imageOffset.z) info.pitch = std::max((pitch + element.wide - 1u) / element.wide, 1u);
: 0; info.tail_x = tail ? volume.tail_x[level] : 0;
info.pitch = info.tail_y = tail ? volume.tail_y[level] : 0;
std::max((pitch + element.wide - 1u) / element.wide, 1u); info.tail = tail;
info.tail_x = tail ? volume.tail_x[level] : 0; info.tiled_width = volume.level_widths[level];
info.tail_y = tail ? volume.tail_y[level] : 0;
info.tail = tail;
info.tiled_width = volume.level_widths[level];
info.tiled_height = volume.level_heights[level]; info.tiled_height = volume.level_heights[level];
infos.push_back(info); infos.push_back(info);
} }
@@ -581,12 +574,11 @@ bool TextureBuildGpuTileInfos(uint64_t size,
const auto level_depth = GetTextureLevelDepth(depth, level, layout.volume_texture); const auto level_depth = GetTextureLevelDepth(depth, level, layout.volume_texture);
for (uint32_t z = 0; z < level_depth; z++) { for (uint32_t z = 0; z < level_depth; z++) {
const auto& region = regions[region_index++]; const auto& region = regions[region_index++];
const auto pitch = region.bufferRowLength != 0 const auto pitch =
? region.bufferRowLength region.bufferRowLength != 0 ? region.bufferRowLength : region.imageExtent.width;
: region.imageExtent.width; const auto logical_height = region.bufferImageHeight != 0
const auto logical_height = region.bufferImageHeight != 0 ? region.bufferImageHeight
? region.bufferImageHeight : region.imageExtent.height;
: region.imageExtent.height;
GpuTileInfo info {}; GpuTileInfo info {};
info.family = block.family; info.family = block.family;
info.bytes_per_element = block.bytes_per_element; info.bytes_per_element = block.bytes_per_element;
@@ -597,16 +589,14 @@ bool TextureBuildGpuTileInfos(uint64_t size,
info.tiled_size)) { info.tiled_size)) {
return false; return false;
} }
info.width = std::max( info.width =
(region.imageExtent.width + element.wide - 1u) / element.wide, 1u); std::max((region.imageExtent.width + element.wide - 1u) / element.wide, 1u);
info.height = std::max( info.height = std::max((logical_height + element.tall - 1u) / element.tall, 1u);
(logical_height + element.tall - 1u) / element.tall, 1u);
info.surface_z = base_family == TileBlockFamily::RenderTarget64KB || info.surface_z = base_family == TileBlockFamily::RenderTarget64KB ||
base_family == TileBlockFamily::Depth64KB base_family == TileBlockFamily::Depth64KB
? region.imageSubresource.baseArrayLayer ? region.imageSubresource.baseArrayLayer
: 0; : 0;
info.pitch = info.pitch = std::max((pitch + element.wide - 1u) / element.wide, 1u);
std::max((pitch + element.wide - 1u) / element.wide, 1u);
info.tail = tail; info.tail = tail;
info.tail_x = tail ? level_size.x : 0; info.tail_x = tail ? level_size.x : 0;
info.tail_y = tail ? level_size.y : 0; info.tail_y = tail ? level_size.y : 0;
@@ -32,20 +32,19 @@ struct TextureUploadLayout {
TilePaddedSize padded_sizes[16] = {}; TilePaddedSize padded_sizes[16] = {};
}; };
vk::ComponentMapping TextureGetComponentMapping(uint32_t swizzle); vk::ComponentMapping TextureGetComponentMapping(uint32_t swizzle);
vk::Format TextureGetFormat(uint32_t fmt); vk::Format TextureGetFormat(uint32_t fmt);
RenderTargetFormatInfo TextureGetRenderTargetFormat(uint32_t layout, uint32_t type, uint32_t order); RenderTargetFormatInfo TextureGetRenderTargetFormat(uint32_t layout, uint32_t type, uint32_t order);
TextureUploadLayout TextureCalcUploadLayout(uint32_t fmt, uint64_t width, uint64_t height, TextureUploadLayout TextureCalcUploadLayout(uint32_t fmt, uint64_t width, uint64_t height,
uint64_t levels, uint32_t depth, uint64_t pitch, uint64_t levels, uint32_t depth, uint64_t pitch,
uint64_t tile, uint64_t upload_size, uint64_t tile, uint64_t upload_size,
bool allow_depth_tile, bool volume_texture, bool allow_depth_tile, bool volume_texture,
const char* owner); const char* owner);
std::vector<vk::BufferImageCopy> std::vector<vk::BufferImageCopy> TextureBuildImageCopies(const TextureUploadLayout& layout,
TextureBuildImageCopies(const TextureUploadLayout& layout, uint32_t width, uint32_t height, uint32_t width, uint32_t height,
uint32_t depth, uint64_t levels, bool array_texture, uint32_t depth, uint64_t levels,
bool volume_texture); bool array_texture, bool volume_texture);
bool TextureBuildGpuTileInfos(uint64_t size, bool TextureBuildGpuTileInfos(uint64_t size, const std::vector<vk::BufferImageCopy>& regions,
const std::vector<vk::BufferImageCopy>& regions,
const TextureUploadLayout& layout, uint32_t fmt, uint32_t depth, const TextureUploadLayout& layout, uint32_t fmt, uint32_t depth,
uint64_t levels, std::vector<GpuTileInfo>& infos); uint64_t levels, std::vector<GpuTileInfo>& infos);
@@ -14,9 +14,9 @@
#include "gpu_tiler_shaders/gpu_tiler_standard64_spv.h" #include "gpu_tiler_shaders/gpu_tiler_standard64_spv.h"
#include "gpu_tiler_shaders/gpu_tiler_swap_bgra16_spv.h" #include "gpu_tiler_shaders/gpu_tiler_swap_bgra16_spv.h"
#include "graphics/host_gpu/graphicContext.h" #include "graphics/host_gpu/graphicContext.h"
#include "graphics/host_gpu/renderer/cache/streamBuffer.h"
#include "graphics/host_gpu/renderer/commandScheduler.h" #include "graphics/host_gpu/renderer/commandScheduler.h"
#include "graphics/host_gpu/renderer/image/image.h" #include "graphics/host_gpu/renderer/image/image.h"
#include "graphics/host_gpu/renderer/cache/streamBuffer.h"
#include <algorithm> #include <algorithm>
#include <array> #include <array>
@@ -26,8 +26,8 @@ MasterSemaphore::~MasterSemaphore() {
} }
void MasterSemaphore::Refresh() { void MasterSemaphore::Refresh() {
uint64_t counter = 0; uint64_t counter = 0;
const auto result = m_graphics.device.getSemaphoreCounterValue(m_semaphore, &counter); const auto result = m_graphics.device.getSemaphoreCounterValue(m_semaphore, &counter);
EXIT_NOT_IMPLEMENTED(result != vk::Result::eSuccess); EXIT_NOT_IMPLEMENTED(result != vk::Result::eSuccess);
auto known = m_gpu_tick.load(std::memory_order_acquire); auto known = m_gpu_tick.load(std::memory_order_acquire);
@@ -22,7 +22,7 @@ public:
[[nodiscard]] uint64_t KnownGpuTick() const noexcept { [[nodiscard]] uint64_t KnownGpuTick() const noexcept {
return m_gpu_tick.load(std::memory_order_acquire); return m_gpu_tick.load(std::memory_order_acquire);
} }
[[nodiscard]] bool IsFree(uint64_t tick) const noexcept { return KnownGpuTick() >= tick; } [[nodiscard]] bool IsFree(uint64_t tick) const noexcept { return KnownGpuTick() >= tick; }
[[nodiscard]] uint64_t NextTick() noexcept { [[nodiscard]] uint64_t NextTick() noexcept {
return m_current_tick.fetch_add(1, std::memory_order_release); return m_current_tick.fetch_add(1, std::memory_order_release);
} }
@@ -26,11 +26,15 @@ bool IsSampledImage(BindingKind kind) {
case BindingKind::Sampled1DArray: case BindingKind::Sampled1DArray:
case BindingKind::Sampled2D: case BindingKind::Sampled2D:
case BindingKind::Sampled2DArray: case BindingKind::Sampled2DArray:
case BindingKind::Sampled2DMsaa:
case BindingKind::Sampled2DMsaaArray:
case BindingKind::Sampled3D: case BindingKind::Sampled3D:
case BindingKind::SampledUint1D: case BindingKind::SampledUint1D:
case BindingKind::SampledUint1DArray: case BindingKind::SampledUint1DArray:
case BindingKind::SampledUint2D: case BindingKind::SampledUint2D:
case BindingKind::SampledUint2DArray: case BindingKind::SampledUint2DArray:
case BindingKind::SampledUint2DMsaa:
case BindingKind::SampledUint2DMsaaArray:
case BindingKind::SampledUint3D: return true; case BindingKind::SampledUint3D: return true;
default: return false; default: return false;
} }
@@ -95,7 +95,7 @@ private:
}; };
static vk::DescriptorImageInfo MakeImageInfo(const TextureBinding& texture); static vk::DescriptorImageInfo MakeImageInfo(const TextureBinding& texture);
void CreatePool(); void CreatePool();
VulkanDescriptorSet* Allocate(Stage stage, const ShaderRecompiler::IR::Program& program); VulkanDescriptorSet* Allocate(Stage stage, const ShaderRecompiler::IR::Program& program);
vk::DescriptorSetLayout vk::DescriptorSetLayout
GetDescriptorSetLayoutInternal(Stage stage, const ShaderRecompiler::IR::Program& program); GetDescriptorSetLayoutInternal(Stage stage, const ShaderRecompiler::IR::Program& program);
@@ -73,6 +73,11 @@ static Prospero::ImageType TextureBaseType(Prospero::ImageType type) {
} }
} }
static bool IsMultisampledTexture(Prospero::ImageType type) {
return type == Prospero::ImageType::kColor2DMsaa ||
type == Prospero::ImageType::kColor2DMsaaArray;
}
static BufferView NativeStorageBuffer(RenderContext& context, CommandBuffer& command_buffer, static BufferView NativeStorageBuffer(RenderContext& context, CommandBuffer& command_buffer,
const ShaderBufferResource& descriptor, const ShaderBufferResource& descriptor,
const ShaderRecompiler::IR::BufferResource& resource, const ShaderRecompiler::IR::BufferResource& resource,
@@ -159,6 +164,8 @@ static bool IsSupportedSampledColorResource(const ShaderRecompiler::IR::ImageRes
case ShaderRecompiler::Decoder::ImageDimension::Dim1DArray: case ShaderRecompiler::Decoder::ImageDimension::Dim1DArray:
case ShaderRecompiler::Decoder::ImageDimension::Dim2D: case ShaderRecompiler::Decoder::ImageDimension::Dim2D:
case ShaderRecompiler::Decoder::ImageDimension::Dim2DArray: case ShaderRecompiler::Decoder::ImageDimension::Dim2DArray:
case ShaderRecompiler::Decoder::ImageDimension::Dim2DMsaa:
case ShaderRecompiler::Decoder::ImageDimension::Dim2DMsaaArray:
supported_dimension = true; supported_dimension = true;
break; break;
default: break; default: break;
@@ -195,6 +202,22 @@ TargetTextureViewInfo ResolveTargetTextureView(const ShaderRecompiler::IR::Image
? TargetTextureViewInfo {vk::ImageViewType::e2DArray, base_layer, ? TargetTextureViewInfo {vk::ImageViewType::e2DArray, base_layer,
image_layers - base_layer} image_layers - base_layer}
: TargetTextureViewInfo {}; : TargetTextureViewInfo {};
case Prospero::ImageType::kColor2DMsaa:
return resource.dimension == ShaderRecompiler::Decoder::ImageDimension::Dim2DMsaa &&
base_layer == 0 && image_layers == 1
? TargetTextureViewInfo {vk::ImageViewType::e2D, 0, 1}
: TargetTextureViewInfo {};
case Prospero::ImageType::kColor2DMsaaArray:
if (resource.dimension == ShaderRecompiler::Decoder::ImageDimension::Dim2DMsaa &&
base_layer == 0 && image_layers == 1) {
return {vk::ImageViewType::e2D, 0, 1};
}
return resource.dimension ==
ShaderRecompiler::Decoder::ImageDimension::Dim2DMsaaArray &&
base_layer < image_layers
? TargetTextureViewInfo {vk::ImageViewType::e2DArray, base_layer,
image_layers - base_layer}
: TargetTextureViewInfo {};
default: return {}; default: return {};
} }
} }
@@ -209,41 +232,64 @@ bool IsSupportedSampledVideoOutView(const ShaderRecompiler::IR::ImageResource& r
} }
bool IsSupportedDepthTargetDescriptor(const ShaderTextureResource& descriptor, const Image& image) { bool IsSupportedDepthTargetDescriptor(const ShaderTextureResource& descriptor, const Image& image) {
const auto width = static_cast<uint32_t>(descriptor.Width5()) + 1u; const auto width = static_cast<uint32_t>(descriptor.Width5()) + 1u;
const auto height = static_cast<uint32_t>(descriptor.Height5()) + 1u; const auto height = static_cast<uint32_t>(descriptor.Height5()) + 1u;
const auto pitch = TileGetTexturePitch(descriptor.Format(), width, 1, descriptor.TileMode()); const auto type = static_cast<Prospero::ImageType>(descriptor.Type());
const auto type = static_cast<Prospero::ImageType>(descriptor.Type()); const bool multisampled = IsMultisampledTexture(type);
const bool supported_single_layer = const auto samples = multisampled ? 1u << descriptor.LastLevel() : 1u;
image.info.resources.layers == 1 && descriptor.Depth() == 0 && const auto pitch =
descriptor.BaseArray5() == 0 && multisampled ? TileGetDepthPitch(width, image.info.bytes_per_block, descriptor.LastLevel())
(type == Prospero::ImageType::kColor2D || type == Prospero::ImageType::kColor2DArray); : TileGetTexturePitch(descriptor.Format(), width, 1, descriptor.TileMode());
const bool supported_2d = type == Prospero::ImageType::kColor2D &&
image.info.resources.layers == 1 && descriptor.Depth() == 0 &&
descriptor.BaseArray5() == 0;
const bool supported_array = type == Prospero::ImageType::kColor2DArray &&
descriptor.BaseArray5() <= descriptor.Depth() &&
descriptor.Depth() < image.info.resources.layers;
const bool supported_cube = const bool supported_cube =
type == Prospero::ImageType::kCube && width == height && image.info.resources.layers >= 6 && type == Prospero::ImageType::kCube && width == height && image.info.resources.layers >= 6 &&
image.info.resources.layers % 6u == 0 && image.info.resources.layers % 6u == 0 &&
static_cast<uint32_t>(descriptor.Depth()) + 1u == image.info.resources.layers && static_cast<uint32_t>(descriptor.Depth()) + 1u == image.info.resources.layers &&
descriptor.BaseArray5() == 0; descriptor.BaseArray5() == 0;
const bool supported_msaa_2d = type == Prospero::ImageType::kColor2DMsaa &&
image.info.resources.layers == 1 && descriptor.Depth() == 0 &&
descriptor.BaseArray5() == 0;
const bool supported_msaa_array = type == Prospero::ImageType::kColor2DMsaaArray &&
descriptor.BaseArray5() <= descriptor.Depth() &&
descriptor.Depth() < image.info.resources.layers;
const bool levels_ok =
multisampled
? descriptor.BaseLevel() == 0 && descriptor.LastLevel() >= 1 &&
descriptor.LastLevel() <= 3 && descriptor.MaxMip() == descriptor.LastLevel() &&
image.info.resources.levels == 1 && image.info.samples == samples
: descriptor.BaseLevel() == 0 && descriptor.LastLevel() == 0 &&
descriptor.MaxMip() == 0 && image.info.samples == 1;
return image.info.IsDepth() && width == image.info.extent.width && return image.info.IsDepth() && width == image.info.extent.width &&
height == image.info.extent.height && (supported_single_layer || supported_cube) && height == image.info.extent.height &&
descriptor.BaseLevel() == 0 && descriptor.LastLevel() == 0 && descriptor.MaxMip() == 0 && (supported_2d || supported_array || supported_cube || supported_msaa_2d ||
descriptor.MinLod() == 0 && descriptor.BaseArray5() == 0 && supported_msaa_array) &&
levels_ok && descriptor.MinLod() == 0 &&
descriptor.TileMode() == Prospero::GpuEnumValue(Prospero::TileMode::kDepth) && descriptor.TileMode() == Prospero::GpuEnumValue(Prospero::TileMode::kDepth) &&
descriptor.BCSwizzle() == 0 && !descriptor.MsaaDepth() && pitch >= width && descriptor.BCSwizzle() == 0 && descriptor.MsaaDepth() == multisampled &&
pitch == image.info.pitch; pitch >= width && pitch == image.info.pitch;
} }
bool IsSupportedDepthTextureEncoding(const ShaderTextureResource& descriptor, const Image& image) { bool IsSupportedDepthTextureEncoding(const ShaderTextureResource& descriptor, const Image& image) {
constexpr uint32_t field1_reserved_mask = 0x200fff00u; constexpr uint32_t field1_reserved_mask = 0x200fff00u;
constexpr uint32_t field2_reserved_mask = 0xf0003000u; constexpr uint32_t field2_reserved_mask = 0xf0003000u;
constexpr uint32_t field3_common = 0x01800000u; const uint32_t field3_expected = descriptor.DstSelXYZW() |
constexpr uint32_t field5_expected = 0x00700000u; (static_cast<uint32_t>(descriptor.BaseLevel()) << 12u) |
const uint32_t field3_expected = (static_cast<uint32_t>(descriptor.LastLevel()) << 16u) |
(descriptor.Type() << 28u) | field3_common | descriptor.DstSelXYZW(); (static_cast<uint32_t>(descriptor.TileMode()) << 20u) |
const uint32_t field4_expected = descriptor.Depth() | (descriptor.BaseArray5() << 16u); (static_cast<uint32_t>(descriptor.Type()) << 28u);
const bool common = (descriptor.fields[1] & field1_reserved_mask) == 0 && const uint32_t field4_expected = descriptor.Depth() | (descriptor.BaseArray5() << 16u);
(descriptor.fields[2] & field2_reserved_mask) == 0 && const uint32_t field5_expected =
descriptor.fields[3] == field3_expected && 0x00700000u | (static_cast<uint32_t>(descriptor.MaxMip()) << 4u);
descriptor.fields[4] == field4_expected && const bool common = (descriptor.fields[1] & field1_reserved_mask) == 0 &&
descriptor.fields[5] == field5_expected; (descriptor.fields[2] & field2_reserved_mask) == 0 &&
descriptor.fields[3] == field3_expected &&
descriptor.fields[4] == field4_expected &&
descriptor.fields[5] == field5_expected;
if (!common || (descriptor.fields[6] == 0 && descriptor.fields[7] != 0)) { if (!common || (descriptor.fields[6] == 0 && descriptor.fields[7] != 0)) {
return false; return false;
} }
@@ -251,8 +297,9 @@ bool IsSupportedDepthTextureEncoding(const ShaderTextureResource& descriptor, co
return true; return true;
} }
constexpr uint32_t htile_control = 0x00280000u; constexpr uint32_t htile_control = 0x00280000u;
const auto metadata_addr = descriptor.MetaAddr() << 8u; const uint32_t expected_control = htile_control | (descriptor.MsaaDepth() ? (1u << 10u) : 0u);
return (descriptor.fields[6] & 0x00ffffffu) == htile_control && metadata_addr != 0 && const auto metadata_addr = descriptor.MetaAddr() << 8u;
return (descriptor.fields[6] & 0x00ffffffu) == expected_control && metadata_addr != 0 &&
descriptor.TileMode() == Prospero::GpuEnumValue(Prospero::TileMode::kDepth) && descriptor.TileMode() == Prospero::GpuEnumValue(Prospero::TileMode::kDepth) &&
image.info.tile_mode == Prospero::GpuEnumValue(Prospero::TileMode::kDepth) && image.info.tile_mode == Prospero::GpuEnumValue(Prospero::TileMode::kDepth) &&
image.info.metadata.kind == ImageMetadataKind::Htile && image.info.metadata.kind == ImageMetadataKind::Htile &&
@@ -518,6 +565,7 @@ static ImageViewInfo TextureViewInfo(const ShaderRecompiler::IR::ImageResource&
view.layer_count = 1; view.layer_count = 1;
break; break;
case ShaderRecompiler::Decoder::ImageDimension::Dim2DArray: case ShaderRecompiler::Decoder::ImageDimension::Dim2DArray:
case ShaderRecompiler::Decoder::ImageDimension::Dim2DMsaaArray:
view.type = vk::ImageViewType::e2DArray; view.type = vk::ImageViewType::e2DArray;
view.base_layer = descriptor.BaseArray5(); view.base_layer = descriptor.BaseArray5();
if (view.base_layer >= image_layers) { if (view.base_layer >= image_layers) {
@@ -526,6 +574,7 @@ static ImageViewInfo TextureViewInfo(const ShaderRecompiler::IR::ImageResource&
view.layer_count = image_layers - view.base_layer; view.layer_count = image_layers - view.base_layer;
break; break;
case ShaderRecompiler::Decoder::ImageDimension::Dim2D: case ShaderRecompiler::Decoder::ImageDimension::Dim2D:
case ShaderRecompiler::Decoder::ImageDimension::Dim2DMsaa:
view.type = vk::ImageViewType::e2D; view.type = vk::ImageViewType::e2D;
view.base_layer = descriptor.BaseArray5(); view.base_layer = descriptor.BaseArray5();
if (view.base_layer >= image_layers) { if (view.base_layer >= image_layers) {
@@ -556,22 +605,23 @@ RenderExecutor::ResolveTexture(const ShaderRecompiler::IR::ImageResource& reso
return {id, nullptr, std::move(desc)}; return {id, nullptr, std::move(desc)};
} }
const auto address = descriptor.Base40(); const auto address = descriptor.Base40();
const auto width = static_cast<uint32_t>(descriptor.Width5()) + 1u; const auto width = static_cast<uint32_t>(descriptor.Width5()) + 1u;
const auto height = static_cast<uint32_t>(descriptor.Height5()) + 1u; const auto height = static_cast<uint32_t>(descriptor.Height5()) + 1u;
const auto base_level = descriptor.BaseLevel(); const auto base_level = descriptor.BaseLevel();
const auto last_level = descriptor.LastLevel(); const auto last_level = descriptor.LastLevel();
const auto type = TextureType(descriptor); const auto type = TextureType(descriptor);
const bool multisampled = const bool multisampled = IsMultisampledTexture(type);
type == Prospero::ImageType::kColor2DMsaa || type == Prospero::ImageType::kColor2DMsaaArray; const auto levels = multisampled ? 1u : static_cast<uint32_t>(descriptor.MaxMip()) + 1u;
const auto levels = multisampled ? 1u : static_cast<uint32_t>(descriptor.MaxMip()) + 1u; const auto tile = descriptor.TileMode();
const auto tile = descriptor.TileMode(); const bool msaa_tile =
const bool msaa_tile = tile == Prospero::GpuEnumValue(Prospero::TileMode::kRenderTarget); tile == Prospero::GpuEnumValue(descriptor.MsaaDepth() ? Prospero::TileMode::kDepth
: Prospero::TileMode::kRenderTarget);
const bool msaa_array = type == Prospero::ImageType::kColor2DMsaaArray; const bool msaa_array = type == Prospero::ImageType::kColor2DMsaaArray;
if ((!multisampled && (base_level > last_level || last_level >= levels)) || if ((!multisampled && (base_level > last_level || last_level >= levels)) ||
(multisampled && (multisampled &&
(base_level != 0 || last_level == 0 || last_level > 3 || (base_level != 0 || last_level == 0 || last_level > 3 ||
descriptor.MaxMip() != last_level || !msaa_tile || descriptor.MsaaDepth() || descriptor.MaxMip() != last_level || !msaa_tile ||
(!msaa_array && (descriptor.Depth() != 0 || descriptor.BaseArray5() != 0))))) { (!msaa_array && (descriptor.Depth() != 0 || descriptor.BaseArray5() != 0))))) {
EXIT("unsupported texture mip view: base=%u last=%u levels=%u\n", base_level, last_level, EXIT("unsupported texture mip view: base=%u last=%u levels=%u\n", base_level, last_level,
levels); levels);
@@ -36,7 +36,7 @@ ResolveTargetTextureView(const ShaderRecompiler::IR::ImageResource& resource,
[[nodiscard]] bool IsSupportedDepthTargetDescriptor(const ShaderTextureResource& descriptor, [[nodiscard]] bool IsSupportedDepthTargetDescriptor(const ShaderTextureResource& descriptor,
const Image& image); const Image& image);
[[nodiscard]] bool IsSupportedDepthTextureEncoding(const ShaderTextureResource& descriptor, [[nodiscard]] bool IsSupportedDepthTextureEncoding(const ShaderTextureResource& descriptor,
const Image& image); const Image& image);
[[nodiscard]] bool [[nodiscard]] bool
IsSupportedSampledVideoOutView(const ShaderRecompiler::IR::ImageResource& resource, IsSupportedSampledVideoOutView(const ShaderRecompiler::IR::ImageResource& resource,
const ShaderTextureResource& descriptor, const Image& image); const ShaderTextureResource& descriptor, const Image& image);
@@ -88,12 +88,12 @@ PipelineCache::GraphicsPipeline& PipelineCache::CreateGraphicsPipeline(
PipelineStaticParameters static_params {}; PipelineStaticParameters static_params {};
GraphicsPipeline p {}; GraphicsPipeline p {};
p.ps_shader_id = ps_id; p.ps_shader_id = ps_id;
p.vs_shader_id = vs_id; p.vs_shader_id = vs_id;
static_params.color_count = color_count; static_params.color_count = color_count;
PipelineRenderingState rendering {}; PipelineRenderingState rendering {};
rendering.color_count = color_count; rendering.color_count = color_count;
uint32_t attachment_samples = 0; uint32_t attachment_samples = 0;
for (uint32_t i = 0; i < color_count; i++) { for (uint32_t i = 0; i < color_count; i++) {
EXIT_IF(!colors[i].image_id || colors[i].format == vk::Format::eUndefined); EXIT_IF(!colors[i].image_id || colors[i].format == vk::Format::eUndefined);
@@ -116,8 +116,8 @@ PipelineCache::GraphicsPipeline& PipelineCache::CreateGraphicsPipeline(
if (attachment_samples == 0) { if (attachment_samples == 0) {
attachment_samples = depth.samples; attachment_samples = depth.samples;
} else if (attachment_samples != depth.samples) { } else if (attachment_samples != depth.samples) {
EXIT("mixed color/depth sample counts are unsupported: %u and %u\n", EXIT("mixed color/depth sample counts are unsupported: %u and %u\n", attachment_samples,
attachment_samples, depth.samples); depth.samples);
} }
} }
EXIT_IF(attachment_samples == 0 || EXIT_IF(attachment_samples == 0 ||
@@ -179,10 +179,10 @@ PipelineCache::GraphicsPipeline& PipelineCache::CreateGraphicsPipeline(
NormalizeStaticParamsForDynamicState(static_params); NormalizeStaticParamsForDynamicState(static_params);
GraphicsPipelineKey key {}; GraphicsPipelineKey key {};
key.rendering = rendering; key.rendering = rendering;
key.vs_shader_id = p.vs_shader_id; key.vs_shader_id = p.vs_shader_id;
key.ps_shader_id = p.ps_shader_id; key.ps_shader_id = p.ps_shader_id;
key.static_params = static_params; key.static_params = static_params;
if (auto iter = m_graphics_pipelines.find(key); iter != m_graphics_pipelines.end()) { if (auto iter = m_graphics_pipelines.find(key); iter != m_graphics_pipelines.end()) {
return *iter->second; return *iter->second;
@@ -203,9 +203,8 @@ PipelineCache::GraphicsPipeline& PipelineCache::CreateGraphicsPipeline(
LogPipelineTrace("CreatePipelineInternal begin", vs_id.hash0, vs_id.crc32, ps_id.hash0, LogPipelineTrace("CreatePipelineInternal begin", vs_id.hash0, vs_id.crc32, ps_id.hash0,
ps_id.crc32); ps_id.crc32);
CreatePipelineInternal(m_graphics, m_descriptor_cache, *cached, rendering, vs_input_info, CreatePipelineInternal(m_graphics, m_descriptor_cache, *cached, rendering, vs_input_info,
vs_spirv, ps_input_info, vs_spirv, ps_input_info, ps_spirv, static_params, vs_id.hash0,
ps_spirv, static_params, vs_id.hash0, vs_id.crc32, ps_id.hash0, vs_id.crc32, ps_id.hash0, ps_id.crc32, ps_active);
ps_id.crc32, ps_active);
LogPipelineTrace("CreatePipelineInternal done", vs_id.hash0, vs_id.crc32, ps_id.hash0, LogPipelineTrace("CreatePipelineInternal done", vs_id.hash0, vs_id.crc32, ps_id.hash0,
ps_id.crc32); ps_id.crc32);
@@ -88,9 +88,9 @@ static_assert(sizeof(PipelineStaticParameters) ==
struct PipelineRenderingState { struct PipelineRenderingState {
std::array<vk::Format, RENDER_COLOR_ATTACHMENTS_MAX> color_formats {}; std::array<vk::Format, RENDER_COLOR_ATTACHMENTS_MAX> color_formats {};
vk::Format depth_format = vk::Format::eUndefined; vk::Format depth_format = vk::Format::eUndefined;
vk::Format stencil_format = vk::Format::eUndefined; vk::Format stencil_format = vk::Format::eUndefined;
uint32_t color_count = 0; uint32_t color_count = 0;
bool operator==(const PipelineRenderingState&) const = default; bool operator==(const PipelineRenderingState&) const = default;
}; };
@@ -118,11 +118,12 @@ public:
ShaderId cs_shader_id; ShaderId cs_shader_id;
}; };
GraphicsPipeline& CreateGraphicsPipeline( GraphicsPipeline&
RenderColorInfo* colors, uint32_t color_count, RenderDepthInfo& depth, CreateGraphicsPipeline(RenderColorInfo* colors, uint32_t color_count, RenderDepthInfo& depth,
ShaderVertexInputInfo& vs_input_info, RenderCommandBuffer& command, ShaderVertexInputInfo& vs_input_info, RenderCommandBuffer& command,
ShaderPixelInputInfo* ps_input_info, vk::PrimitiveTopology topology, bool ps_active, ShaderPixelInputInfo* ps_input_info, vk::PrimitiveTopology topology,
std::span<const uint32_t> vs_spirv, std::span<const uint32_t> ps_spirv); bool ps_active, std::span<const uint32_t> vs_spirv,
std::span<const uint32_t> ps_spirv);
ComputePipeline& CreateComputePipeline(ShaderComputeInputInfo& input_info, ComputePipeline& CreateComputePipeline(ShaderComputeInputInfo& input_info,
const HW::ComputeShaderInfo& cs_regs, const HW::ComputeShaderInfo& cs_regs,
std::span<const uint32_t> cs_spirv); std::span<const uint32_t> cs_spirv);
@@ -199,7 +200,7 @@ private:
} }
}; };
GraphicContext& m_graphics; GraphicContext& m_graphics;
DescriptorCache& m_descriptor_cache; DescriptorCache& m_descriptor_cache;
std::unordered_map<GraphicsPipelineKey, std::unique_ptr<GraphicsPipeline>, std::unordered_map<GraphicsPipelineKey, std::unique_ptr<GraphicsPipeline>,
GraphicsPipelineKeyHash> GraphicsPipelineKeyHash>
@@ -211,16 +212,13 @@ private:
void LogPipelineTrace(const char* phase, uint32_t vs_hash0, uint32_t vs_crc32, uint32_t ps_hash0, void LogPipelineTrace(const char* phase, uint32_t vs_hash0, uint32_t vs_crc32, uint32_t ps_hash0,
uint32_t ps_crc32); uint32_t ps_crc32);
void CreatePipelineInternal(GraphicContext& graphics, DescriptorCache& descriptor_cache, void CreatePipelineInternal(
PipelineCache::GraphicsPipeline& pipeline, GraphicContext& graphics, DescriptorCache& descriptor_cache,
const PipelineRenderingState& rendering, PipelineCache::GraphicsPipeline& pipeline, const PipelineRenderingState& rendering,
const ShaderVertexInputInfo& vs_input_info, const ShaderVertexInputInfo& vs_input_info, std::span<const uint32_t> vs_shader,
std::span<const uint32_t> vs_shader, const ShaderPixelInputInfo* ps_input_info, std::span<const uint32_t> ps_shader,
const ShaderPixelInputInfo* ps_input_info, const PipelineStaticParameters& static_params, uint32_t vs_hash0, uint32_t vs_crc32,
std::span<const uint32_t> ps_shader, uint32_t ps_hash0, uint32_t ps_crc32, bool ps_active);
const PipelineStaticParameters& static_params, uint32_t vs_hash0,
uint32_t vs_crc32, uint32_t ps_hash0, uint32_t ps_crc32,
bool ps_active);
void CreatePipelineInternal(GraphicContext& graphics, DescriptorCache& descriptor_cache, void CreatePipelineInternal(GraphicContext& graphics, DescriptorCache& descriptor_cache,
PipelineCache::ComputePipeline& pipeline, PipelineCache::ComputePipeline& pipeline,
const ShaderComputeInputInfo& input_info, const ShaderComputeInputInfo& input_info,
@@ -8,10 +8,10 @@
#include "graphics/host_gpu/renderer/debug.h" #include "graphics/host_gpu/renderer/debug.h"
#include "graphics/host_gpu/renderer/pipeline/descriptorCache.h" #include "graphics/host_gpu/renderer/pipeline/descriptorCache.h"
#include "graphics/host_gpu/renderer/pipeline/pipelineCache.h" #include "graphics/host_gpu/renderer/pipeline/pipelineCache.h"
#include "graphics/host_gpu/renderer/pipeline/shaderSubgroup.h"
#include "graphics/host_gpu/renderer/render.h" #include "graphics/host_gpu/renderer/render.h"
#include "graphics/host_gpu/renderer/renderContext.h" #include "graphics/host_gpu/renderer/renderContext.h"
#include "graphics/host_gpu/renderer/renderTarget.h" #include "graphics/host_gpu/renderer/renderTarget.h"
#include "graphics/host_gpu/renderer/pipeline/shaderSubgroup.h"
#include "graphics/host_gpu/vulkanCommon.h" #include "graphics/host_gpu/vulkanCommon.h"
#include "graphics/shader/recompiler/ir/ShaderIR.h" #include "graphics/shader/recompiler/ir/ShaderIR.h"
#include "graphics/shader/shader.h" #include "graphics/shader/shader.h"
@@ -385,9 +385,8 @@ static vk::BlendOp GetBlendOp(uint32_t op) {
return vk::BlendOp::eAdd; return vk::BlendOp::eAdd;
} }
static void CreateLayout(DescriptorCache& descriptor_cache, static void CreateLayout(DescriptorCache& descriptor_cache,
std::span<vk::DescriptorSetLayout> set_layouts, std::span<vk::DescriptorSetLayout> set_layouts, uint32_t& set_layouts_num,
uint32_t& set_layouts_num,
std::span<vk::PushConstantRange> push_constant_info, std::span<vk::PushConstantRange> push_constant_info,
uint32_t& push_constant_info_num, uint32_t& push_constant_info_num,
const ShaderRecompiler::IR::Program& program, const ShaderRecompiler::IR::Program& program,
@@ -412,12 +411,11 @@ static void CreateLayout(DescriptorCache& descriptor_cache,
} }
} }
static void ConfigureSubgroupSize(const GraphicContext& graphics, static void ConfigureSubgroupSize(const GraphicContext& graphics, vk::ShaderStageFlagBits vk_stage,
vk::ShaderStageFlagBits vk_stage,
const ShaderRecompiler::IR::Program& program, const ShaderRecompiler::IR::Program& program,
vk::PipelineShaderStageRequiredSubgroupSizeCreateInfo& required, vk::PipelineShaderStageRequiredSubgroupSizeCreateInfo& required,
vk::PipelineShaderStageCreateInfo& stage) { vk::PipelineShaderStageCreateInfo& stage) {
const auto config = const auto config =
ConfigureShaderSubgroup(ShaderSubgroupCapabilities {graphics}, vk_stage, program); ConfigureShaderSubgroup(ShaderSubgroupCapabilities {graphics}, vk_stage, program);
switch (config.mode) { switch (config.mode) {
case ShaderSubgroupMode::Natural: return; case ShaderSubgroupMode::Natural: return;
@@ -456,16 +454,13 @@ static void ConfigureSubgroupSize(const GraphicContext&
} }
// NOLINTNEXTLINE(readability-function-cognitive-complexity) // NOLINTNEXTLINE(readability-function-cognitive-complexity)
void CreatePipelineInternal(GraphicContext& graphics, DescriptorCache& descriptor_cache, void CreatePipelineInternal(
PipelineCache::GraphicsPipeline& pipeline, GraphicContext& graphics, DescriptorCache& descriptor_cache,
const PipelineRenderingState& rendering, PipelineCache::GraphicsPipeline& pipeline, const PipelineRenderingState& rendering,
const ShaderVertexInputInfo& vs_input_info, const ShaderVertexInputInfo& vs_input_info, std::span<const uint32_t> vs_shader,
std::span<const uint32_t> vs_shader, const ShaderPixelInputInfo* ps_input_info, std::span<const uint32_t> ps_shader,
const ShaderPixelInputInfo* ps_input_info, const PipelineStaticParameters& static_params, uint32_t vs_hash0, uint32_t vs_crc32,
std::span<const uint32_t> ps_shader, uint32_t ps_hash0, uint32_t ps_crc32, bool ps_active) {
const PipelineStaticParameters& static_params, uint32_t vs_hash0,
uint32_t vs_crc32, uint32_t ps_hash0, uint32_t ps_crc32,
bool ps_active) {
EXIT_IF(ps_active && ps_input_info == nullptr); EXIT_IF(ps_active && ps_input_info == nullptr);
vk::ShaderModule vert_shader_module = nullptr; vk::ShaderModule vert_shader_module = nullptr;
@@ -511,8 +506,7 @@ void CreatePipelineInternal(GraphicContext& graphics, DescriptorCache& descripto
vert_shader_stage_info.pName = "main"; vert_shader_stage_info.pName = "main";
vert_shader_stage_info.pSpecializationInfo = nullptr; vert_shader_stage_info.pSpecializationInfo = nullptr;
EXIT_IF(!vs_input_info.stage); EXIT_IF(!vs_input_info.stage);
ConfigureSubgroupSize(graphics, vk::ShaderStageFlagBits::eVertex, ConfigureSubgroupSize(graphics, vk::ShaderStageFlagBits::eVertex, *vs_input_info.stage.program,
*vs_input_info.stage.program,
vert_subgroup_size, vert_shader_stage_info); vert_subgroup_size, vert_shader_stage_info);
vk::PipelineShaderStageCreateInfo frag_shader_stage_info {}; vk::PipelineShaderStageCreateInfo frag_shader_stage_info {};
@@ -527,8 +521,8 @@ void CreatePipelineInternal(GraphicContext& graphics, DescriptorCache& descripto
if (ps_active) { if (ps_active) {
EXIT_IF(!ps_input_info->stage); EXIT_IF(!ps_input_info->stage);
ConfigureSubgroupSize(graphics, vk::ShaderStageFlagBits::eFragment, ConfigureSubgroupSize(graphics, vk::ShaderStageFlagBits::eFragment,
*ps_input_info->stage.program, *ps_input_info->stage.program, frag_subgroup_size,
frag_subgroup_size, frag_shader_stage_info); frag_shader_stage_info);
} }
vk::PipelineShaderStageCreateInfo shader_stages[] = {vert_shader_stage_info, vk::PipelineShaderStageCreateInfo shader_stages[] = {vert_shader_stage_info,
@@ -728,13 +722,13 @@ void CreatePipelineInternal(GraphicContext& graphics, DescriptorCache& descripto
clip_ext.depthClipEnable = static_params.depth_clip_enable ? VK_TRUE : VK_FALSE; clip_ext.depthClipEnable = static_params.depth_clip_enable ? VK_TRUE : VK_FALSE;
vk::PipelineRasterizationStateCreateInfo rasterizer {}; vk::PipelineRasterizationStateCreateInfo rasterizer {};
rasterizer.sType = vk::StructureType::ePipelineRasterizationStateCreateInfo; rasterizer.sType = vk::StructureType::ePipelineRasterizationStateCreateInfo;
// MoltenVK lacks VK_EXT_depth_clip_enable; omit the depth-clip struct on macOS and accept // MoltenVK lacks VK_EXT_depth_clip_enable; omit the depth-clip struct on macOS and accept
// Vulkan's default depth clipping (enabled) instead of the PS5's clamp behavior. // Vulkan's default depth clipping (enabled) instead of the PS5's clamp behavior.
#if defined(__APPLE__) #if defined(__APPLE__)
rasterizer.pNext = nullptr; rasterizer.pNext = nullptr;
#else #else
rasterizer.pNext = &clip_ext; rasterizer.pNext = &clip_ext;
#endif #endif
rasterizer.flags = {}; rasterizer.flags = {};
rasterizer.depthClampEnable = VK_FALSE; rasterizer.depthClampEnable = VK_FALSE;
@@ -812,13 +806,13 @@ void CreatePipelineInternal(GraphicContext& graphics, DescriptorCache& descripto
color_write.pColorWriteEnables = color_write_enable; color_write.pColorWriteEnables = color_write_enable;
vk::PipelineColorBlendStateCreateInfo color_blending {}; vk::PipelineColorBlendStateCreateInfo color_blending {};
color_blending.sType = vk::StructureType::ePipelineColorBlendStateCreateInfo; color_blending.sType = vk::StructureType::ePipelineColorBlendStateCreateInfo;
// MoltenVK lacks VK_EXT_color_write_enable; drop the dynamic color-write struct on macOS // MoltenVK lacks VK_EXT_color_write_enable; drop the dynamic color-write struct on macOS
// and rely on each attachment's static colorWriteMask (all channels enabled by default). // and rely on each attachment's static colorWriteMask (all channels enabled by default).
#if defined(__APPLE__) #if defined(__APPLE__)
color_blending.pNext = nullptr; color_blending.pNext = nullptr;
#else #else
color_blending.pNext = &color_write; color_blending.pNext = &color_write;
#endif #endif
color_blending.flags = {}; color_blending.flags = {};
color_blending.logicOpEnable = VK_FALSE; color_blending.logicOpEnable = VK_FALSE;
@@ -838,15 +832,13 @@ void CreatePipelineInternal(GraphicContext& graphics, DescriptorCache& descripto
EXIT_IF(!vs_input_info.stage); EXIT_IF(!vs_input_info.stage);
CreateLayout(descriptor_cache, set_layouts, set_layouts_num, push_constant_info, CreateLayout(descriptor_cache, set_layouts, set_layouts_num, push_constant_info,
push_constant_info_num, push_constant_info_num, *vs_input_info.stage.program,
*vs_input_info.stage.program, vk::ShaderStageFlagBits::eVertex, vk::ShaderStageFlagBits::eVertex, DescriptorCache::Stage::Vertex);
DescriptorCache::Stage::Vertex);
if (ps_active) { if (ps_active) {
EXIT_IF(!ps_input_info->stage); EXIT_IF(!ps_input_info->stage);
CreateLayout(descriptor_cache, set_layouts, set_layouts_num, push_constant_info, CreateLayout(descriptor_cache, set_layouts, set_layouts_num, push_constant_info,
push_constant_info_num, push_constant_info_num, *ps_input_info->stage.program,
*ps_input_info->stage.program, vk::ShaderStageFlagBits::eFragment, vk::ShaderStageFlagBits::eFragment, DescriptorCache::Stage::Pixel);
DescriptorCache::Stage::Pixel);
} }
vk::PipelineLayoutCreateInfo pipeline_layout_info {}; vk::PipelineLayoutCreateInfo pipeline_layout_info {};
@@ -923,32 +915,32 @@ void CreatePipelineInternal(GraphicContext& graphics, DescriptorCache& descripto
dynamic_state.dynamicStateCount = dynamic_states_count; dynamic_state.dynamicStateCount = dynamic_states_count;
dynamic_state.pDynamicStates = dynamic_states; dynamic_state.pDynamicStates = dynamic_states;
vk::GraphicsPipelineCreateInfo pipeline_info {}; vk::GraphicsPipelineCreateInfo pipeline_info {};
vk::PipelineRenderingCreateInfo rendering_info {}; vk::PipelineRenderingCreateInfo rendering_info {};
rendering_info.sType = vk::StructureType::ePipelineRenderingCreateInfo; rendering_info.sType = vk::StructureType::ePipelineRenderingCreateInfo;
rendering_info.colorAttachmentCount = rendering.color_count; rendering_info.colorAttachmentCount = rendering.color_count;
rendering_info.pColorAttachmentFormats = rendering.color_formats.data(); rendering_info.pColorAttachmentFormats = rendering.color_formats.data();
rendering_info.depthAttachmentFormat = rendering.depth_format; rendering_info.depthAttachmentFormat = rendering.depth_format;
rendering_info.stencilAttachmentFormat = rendering.stencil_format; rendering_info.stencilAttachmentFormat = rendering.stencil_format;
pipeline_info.sType = vk::StructureType::eGraphicsPipelineCreateInfo; pipeline_info.sType = vk::StructureType::eGraphicsPipelineCreateInfo;
pipeline_info.pNext = &rendering_info; pipeline_info.pNext = &rendering_info;
pipeline_info.flags = {}; pipeline_info.flags = {};
pipeline_info.stageCount = shader_stage_count; pipeline_info.stageCount = shader_stage_count;
pipeline_info.pStages = shader_stages; pipeline_info.pStages = shader_stages;
pipeline_info.pVertexInputState = &vertex_input_info; pipeline_info.pVertexInputState = &vertex_input_info;
pipeline_info.pInputAssemblyState = &input_assembly; pipeline_info.pInputAssemblyState = &input_assembly;
pipeline_info.pTessellationState = nullptr; pipeline_info.pTessellationState = nullptr;
pipeline_info.pViewportState = &viewport_state; pipeline_info.pViewportState = &viewport_state;
pipeline_info.pRasterizationState = &rasterizer; pipeline_info.pRasterizationState = &rasterizer;
pipeline_info.pMultisampleState = &multisampling; pipeline_info.pMultisampleState = &multisampling;
pipeline_info.pDepthStencilState = (static_params.with_depth ? &depth_stencil_info : nullptr); pipeline_info.pDepthStencilState = (static_params.with_depth ? &depth_stencil_info : nullptr);
pipeline_info.pColorBlendState = &color_blending; pipeline_info.pColorBlendState = &color_blending;
pipeline_info.pDynamicState = &dynamic_state; pipeline_info.pDynamicState = &dynamic_state;
pipeline_info.layout = pipeline.pipeline_layout; pipeline_info.layout = pipeline.pipeline_layout;
pipeline_info.renderPass = nullptr; pipeline_info.renderPass = nullptr;
pipeline_info.subpass = 0; pipeline_info.subpass = 0;
pipeline_info.basePipelineHandle = nullptr; pipeline_info.basePipelineHandle = nullptr;
pipeline_info.basePipelineIndex = -1; pipeline_info.basePipelineIndex = -1;
EXIT_IF(pipeline.pipeline != nullptr); EXIT_IF(pipeline.pipeline != nullptr);
@@ -1012,8 +1004,7 @@ void CreatePipelineInternal(GraphicContext& graphics, DescriptorCache& descripto
comp_shader_stage_info.pName = "main"; comp_shader_stage_info.pName = "main";
comp_shader_stage_info.pSpecializationInfo = nullptr; comp_shader_stage_info.pSpecializationInfo = nullptr;
EXIT_IF(!input_info.stage); EXIT_IF(!input_info.stage);
ConfigureSubgroupSize(graphics, vk::ShaderStageFlagBits::eCompute, ConfigureSubgroupSize(graphics, vk::ShaderStageFlagBits::eCompute, *input_info.stage.program,
*input_info.stage.program,
comp_subgroup_size, comp_shader_stage_info); comp_subgroup_size, comp_shader_stage_info);
vk::DescriptorSetLayout set_layouts[1] = {}; vk::DescriptorSetLayout set_layouts[1] = {};
@@ -1024,9 +1015,8 @@ void CreatePipelineInternal(GraphicContext& graphics, DescriptorCache& descripto
EXIT_IF(!input_info.stage); EXIT_IF(!input_info.stage);
CreateLayout(descriptor_cache, set_layouts, set_layouts_num, push_constant_info, CreateLayout(descriptor_cache, set_layouts, set_layouts_num, push_constant_info,
push_constant_info_num, push_constant_info_num, *input_info.stage.program,
*input_info.stage.program, vk::ShaderStageFlagBits::eCompute, vk::ShaderStageFlagBits::eCompute, DescriptorCache::Stage::Compute);
DescriptorCache::Stage::Compute);
vk::PipelineLayoutCreateInfo pipeline_layout_info {}; vk::PipelineLayoutCreateInfo pipeline_layout_info {};
pipeline_layout_info.sType = vk::StructureType::ePipelineLayoutCreateInfo; pipeline_layout_info.sType = vk::StructureType::ePipelineLayoutCreateInfo;
@@ -10,14 +10,14 @@
#include "graphics/guest_gpu/graphicsRun.h" #include "graphics/guest_gpu/graphicsRun.h"
#include "graphics/guest_gpu/hardwareContext.h" #include "graphics/guest_gpu/hardwareContext.h"
#include "graphics/host_gpu/graphicContext.h" #include "graphics/host_gpu/graphicContext.h"
#include "graphics/host_gpu/renderer/image/imageInfo.h"
#include "graphics/host_gpu/renderer/pipeline/descriptorCache.h" #include "graphics/host_gpu/renderer/pipeline/descriptorCache.h"
#include "graphics/host_gpu/renderer/pipeline/descriptors.h" #include "graphics/host_gpu/renderer/pipeline/descriptors.h"
#include "graphics/host_gpu/renderer/image/imageInfo.h"
#include "graphics/host_gpu/renderer/pipeline/pipelineCache.h" #include "graphics/host_gpu/renderer/pipeline/pipelineCache.h"
#include "graphics/host_gpu/renderer/render.h"
#include "graphics/host_gpu/renderer/renderContext.h"
#include "graphics/host_gpu/renderer/pipeline/shaderResourceBarrier.h" #include "graphics/host_gpu/renderer/pipeline/shaderResourceBarrier.h"
#include "graphics/host_gpu/renderer/pipeline/shaderSubgroup.h" #include "graphics/host_gpu/renderer/pipeline/shaderSubgroup.h"
#include "graphics/host_gpu/renderer/render.h"
#include "graphics/host_gpu/renderer/renderContext.h"
#include "graphics/host_gpu/vulkanCommon.h" #include "graphics/host_gpu/vulkanCommon.h"
#include "graphics/shader/recompiler/ir/ResourceMaterialization.h" #include "graphics/shader/recompiler/ir/ResourceMaterialization.h"
#include "graphics/shader/recompiler/ir/ShaderIR.h" #include "graphics/shader/recompiler/ir/ShaderIR.h"
@@ -14,8 +14,7 @@ namespace Libs::Graphics {
RenderContext::RenderContext(GraphicContext& graphics) RenderContext::RenderContext(GraphicContext& graphics)
: m_graphics(graphics), m_render_executor(*this), m_command_scheduler(*this, graphics), : m_graphics(graphics), m_render_executor(*this), m_command_scheduler(*this, graphics),
m_descriptor_cache(graphics), m_pipeline_cache(graphics, m_descriptor_cache), m_descriptor_cache(graphics), m_pipeline_cache(graphics, m_descriptor_cache),
m_sampler_cache(graphics), m_sampler_cache(graphics), m_gpu_resources(graphics, m_command_scheduler) {
m_gpu_resources(graphics, m_command_scheduler) {
EXIT_NOT_IMPLEMENTED(!Common::Thread::IsMainThread()); EXIT_NOT_IMPLEMENTED(!Common::Thread::IsMainThread());
} }
@@ -27,7 +26,7 @@ RenderContext::~RenderContext() {
void RenderContext::InitializeGpu(VideoOut::VideoOutDriver* video_out) { void RenderContext::InitializeGpu(VideoOut::VideoOutDriver* video_out) {
EXIT_IF(m_gpu != nullptr); EXIT_IF(m_gpu != nullptr);
m_video_out = video_out; m_video_out = video_out;
m_gpu = std::make_unique<Gpu>(*this); m_gpu = std::make_unique<Gpu>(*this);
m_gpu_resources.SetGpu(m_gpu.get()); m_gpu_resources.SetGpu(m_gpu.get());
} }
@@ -99,8 +98,7 @@ void RenderContext::TriggerEopEvent(uint32_t context_id) {
registration.eq, static_cast<uintptr_t>(registration.id), registration.eq, static_cast<uintptr_t>(registration.id),
LibKernel::EventQueue::KERNEL_EVFILT_GRAPHICS, LibKernel::EventQueue::KERNEL_EVFILT_GRAPHICS,
reinterpret_cast<void*>(static_cast<uintptr_t>(context_id))); reinterpret_cast<void*>(static_cast<uintptr_t>(context_id)));
if (result == LibKernel::KERNEL_ERROR_EBADF || if (result == LibKernel::KERNEL_ERROR_EBADF || result == LibKernel::KERNEL_ERROR_ENOENT) {
result == LibKernel::KERNEL_ERROR_ENOENT) {
DeleteEopEq(registration.eq, registration.id); DeleteEopEq(registration.eq, registration.id);
continue; continue;
} }
+17 -17
View File
@@ -6,12 +6,12 @@
#include "common/common.h" #include "common/common.h"
#include "common/threads.h" #include "common/threads.h"
#include "graphics/host_gpu/renderer/cache/bufferCache.h" #include "graphics/host_gpu/renderer/cache/bufferCache.h"
#include "graphics/host_gpu/renderer/commandScheduler.h"
#include "graphics/host_gpu/renderer/pipeline/descriptorCache.h"
#include "graphics/host_gpu/renderer/cache/gpuResourceManager.h" #include "graphics/host_gpu/renderer/cache/gpuResourceManager.h"
#include "graphics/host_gpu/renderer/pipeline/pipelineCache.h"
#include "graphics/host_gpu/renderer/cache/samplerCache.h" #include "graphics/host_gpu/renderer/cache/samplerCache.h"
#include "graphics/host_gpu/renderer/cache/textureCache.h" #include "graphics/host_gpu/renderer/cache/textureCache.h"
#include "graphics/host_gpu/renderer/commandScheduler.h"
#include "graphics/host_gpu/renderer/pipeline/descriptorCache.h"
#include "graphics/host_gpu/renderer/pipeline/pipelineCache.h"
#include "kernel/eventQueue.h" #include "kernel/eventQueue.h"
#include <memory> #include <memory>
@@ -32,10 +32,10 @@ public:
~RenderContext(); ~RenderContext();
KYTY_CLASS_NO_COPY(RenderContext); KYTY_CLASS_NO_COPY(RenderContext);
[[nodiscard]] GraphicContext& GetGraphics() const noexcept { return m_graphics; } [[nodiscard]] GraphicContext& GetGraphics() const noexcept { return m_graphics; }
void InitializeGpu(VideoOut::VideoOutDriver* video_out); void InitializeGpu(VideoOut::VideoOutDriver* video_out);
void ShutdownGpu(); void ShutdownGpu();
[[nodiscard]] Gpu& GetGpu() const; [[nodiscard]] Gpu& GetGpu() const;
[[nodiscard]] VideoOut::VideoOutDriver& GetVideoOut() const; [[nodiscard]] VideoOut::VideoOutDriver& GetVideoOut() const;
Common::Mutex& GetMutex() { return m_mutex; } Common::Mutex& GetMutex() { return m_mutex; }
@@ -56,18 +56,18 @@ private:
struct EopEqRegistration { struct EopEqRegistration {
LibKernel::EventQueue::KernelEqueue eq = LibKernel::EventQueue::KERNEL_EQUEUE_INVALID; LibKernel::EventQueue::KernelEqueue eq = LibKernel::EventQueue::KERNEL_EQUEUE_INVALID;
LibKernel::EventQueue::KernelEqueueRef queue; LibKernel::EventQueue::KernelEqueueRef queue;
int id = 0; int id = 0;
}; };
GraphicContext& m_graphics; GraphicContext& m_graphics;
Common::Mutex m_mutex; Common::Mutex m_mutex;
RenderExecutor m_render_executor; RenderExecutor m_render_executor;
CommandScheduler m_command_scheduler; CommandScheduler m_command_scheduler;
DescriptorCache m_descriptor_cache; DescriptorCache m_descriptor_cache;
PipelineCache m_pipeline_cache; PipelineCache m_pipeline_cache;
SamplerCache m_sampler_cache; SamplerCache m_sampler_cache;
GpuResourceManager m_gpu_resources; GpuResourceManager m_gpu_resources;
std::unique_ptr<Gpu> m_gpu; std::unique_ptr<Gpu> m_gpu;
VideoOut::VideoOutDriver* m_video_out = nullptr; VideoOut::VideoOutDriver* m_video_out = nullptr;
Common::Mutex m_eop_mutex; Common::Mutex m_eop_mutex;
@@ -12,13 +12,13 @@ namespace Libs::Graphics {
static constexpr uint32_t RENDER_COLOR_ATTACHMENTS_MAX = 8; static constexpr uint32_t RENDER_COLOR_ATTACHMENTS_MAX = 8;
struct RenderAttachment { struct RenderAttachment {
vk::ImageView image_view = nullptr; vk::ImageView image_view = nullptr;
vk::ImageLayout image_layout = vk::ImageLayout::eUndefined; vk::ImageLayout image_layout = vk::ImageLayout::eUndefined;
std::array<uint32_t, 4> clear_value = {}; std::array<uint32_t, 4> clear_value = {};
bool is_clear = false; bool is_clear = false;
bool has_depth = false; bool has_depth = false;
bool depth_clear = false; bool depth_clear = false;
bool has_stencil = false; bool has_stencil = false;
bool stencil_clear = false; bool stencil_clear = false;
bool operator==(const RenderAttachment&) const = default; bool operator==(const RenderAttachment&) const = default;
+3 -3
View File
@@ -251,9 +251,9 @@ uint64_t PrepareVideoOutFlip(CommandBuffer& buffer, int handle, int index, int f
int64_t flip_arg) { int64_t flip_arg) {
for (;;) { for (;;) {
uint64_t request_id = 0; uint64_t request_id = 0;
auto& video_out = buffer.GetContext().GetVideoOut(); auto& video_out = buffer.GetContext().GetVideoOut();
const auto result = video_out.SubmitFlipFromGpu( const auto result =
buffer, handle, index, flip_mode, flip_arg, request_id); video_out.SubmitFlipFromGpu(buffer, handle, index, flip_mode, flip_arg, request_id);
if (result == OK) { if (result == OK) {
EXIT_IF(request_id == 0); EXIT_IF(request_id == 0);
return request_id; return request_id;
+6 -7
View File
@@ -122,9 +122,9 @@ uint64_t GraphicContext::GetDeviceMemoryUsage() const {
physical_device_properties.deviceType == vk::PhysicalDeviceType::eDiscreteGpu; physical_device_properties.deviceType == vk::PhysicalDeviceType::eDiscreteGpu;
uint64_t usage = 0; uint64_t usage = 0;
for (uint32_t heap = 0; heap < physical_device_memory_properties.memoryHeapCount; heap++) { for (uint32_t heap = 0; heap < physical_device_memory_properties.memoryHeapCount; heap++) {
const bool device_local = static_cast<bool>( const bool device_local =
physical_device_memory_properties.memoryHeaps[heap].flags & static_cast<bool>(physical_device_memory_properties.memoryHeaps[heap].flags &
vk::MemoryHeapFlagBits::eDeviceLocal); vk::MemoryHeapFlagBits::eDeviceLocal);
if (!discrete || device_local) { if (!discrete || device_local) {
usage += budgets[heap].usage; usage += budgets[heap].usage;
} }
@@ -144,7 +144,7 @@ uint64_t GraphicContext::GetTotalMemoryBudget() const {
uint64_t local = 0; uint64_t local = 0;
uint64_t usage = 0; uint64_t usage = 0;
for (uint32_t heap = 0; heap < physical_device_memory_properties.memoryHeapCount; heap++) { for (uint32_t heap = 0; heap < physical_device_memory_properties.memoryHeapCount; heap++) {
const auto& properties = physical_device_memory_properties.memoryHeaps[heap]; const auto& properties = physical_device_memory_properties.memoryHeaps[heap];
const bool device_local = const bool device_local =
static_cast<bool>(properties.flags & vk::MemoryHeapFlagBits::eDeviceLocal); static_cast<bool>(properties.flags & vk::MemoryHeapFlagBits::eDeviceLocal);
if (device_local) { if (device_local) {
@@ -159,9 +159,8 @@ uint64_t GraphicContext::GetTotalMemoryBudget() const {
return budget - std::min<uint64_t>(budget / 8, 1024ull * 1024 * 1024); return budget - std::min<uint64_t>(budget / 8, 1024ull * 1024 * 1024);
} }
constexpr uint64_t system_reserve = 8ull * 1024 * 1024 * 1024; constexpr uint64_t system_reserve = 8ull * 1024 * 1024 * 1024;
const auto available = budget > usage ? budget - usage : uint64_t {0}; const auto available = budget > usage ? budget - usage : uint64_t {0};
return std::max(local, return std::max(local, available > system_reserve ? available - system_reserve : uint64_t {0});
available > system_reserve ? available - system_reserve : uint64_t {0});
} }
void GraphicContext::CreateBuffer(uint64_t size, VulkanBuffer& buffer) { void GraphicContext::CreateBuffer(uint64_t size, VulkanBuffer& buffer) {
+4
View File
@@ -55,6 +55,10 @@ constexpr FormatMapping kFormatMappings[] = {
{Prospero::BufferFormat::k32_32_32_32UInt, vk::Format::eR32G32B32A32Uint}, {Prospero::BufferFormat::k32_32_32_32UInt, vk::Format::eR32G32B32A32Uint},
{Prospero::BufferFormat::k32_32_32_32SInt, vk::Format::eR32G32B32A32Sint}, {Prospero::BufferFormat::k32_32_32_32SInt, vk::Format::eR32G32B32A32Sint},
{Prospero::BufferFormat::k32_32_32_32Float, vk::Format::eR32G32B32A32Sfloat}, {Prospero::BufferFormat::k32_32_32_32Float, vk::Format::eR32G32B32A32Sfloat},
// Narrow-channel sRGB formats are optional in Vulkan. Keep a same-width fallback until
// sampler-aware sRGB emulation is available.
{Prospero::BufferFormat::k8Srgb, vk::Format::eR8Unorm},
{Prospero::BufferFormat::k8_8Srgb, vk::Format::eR8G8Unorm},
{Prospero::BufferFormat::k8_8_8_8Srgb, vk::Format::eR8G8B8A8Srgb}, {Prospero::BufferFormat::k8_8_8_8Srgb, vk::Format::eR8G8B8A8Srgb},
{Prospero::BufferFormat::k9_9_9_5Float, vk::Format::eE5B9G9R9UfloatPack32}, {Prospero::BufferFormat::k9_9_9_5Float, vk::Format::eE5B9G9R9UfloatPack32},
{Prospero::BufferFormat::k5_6_5UNorm, vk::Format::eB5G6R5UnormPack16}, {Prospero::BufferFormat::k5_6_5UNorm, vk::Format::eB5G6R5UnormPack16},
+7 -7
View File
@@ -20,14 +20,14 @@ public:
~Presenter(); ~Presenter();
KYTY_CLASS_NO_COPY(Presenter); KYTY_CLASS_NO_COPY(Presenter);
[[nodiscard]] Frame& PrepareFrame(CommandBuffer& command, const ImageInfo& info); [[nodiscard]] Frame& PrepareFrame(CommandBuffer& command, const ImageInfo& info);
[[nodiscard]] Frame& PrepareBlankFrame(uint32_t width, uint32_t height, bool opaque, [[nodiscard]] Frame& PrepareBlankFrame(uint32_t width, uint32_t height, bool opaque,
CommandBuffer* producer = nullptr); CommandBuffer* producer = nullptr);
[[nodiscard]] Frame* PrepareLastFrame(); [[nodiscard]] Frame* PrepareLastFrame();
[[nodiscard]] bool IsGuestPaused() const noexcept; [[nodiscard]] bool IsGuestPaused() const noexcept;
[[nodiscard]] RenderContext& Renderer() const noexcept; [[nodiscard]] RenderContext& Renderer() const noexcept;
void Present(Frame& frame, bool reuse = false); void Present(Frame& frame, bool reuse = false);
void Discard(Frame& frame); void Discard(Frame& frame);
private: private:
struct Impl; struct Impl;
+37 -41
View File
@@ -69,8 +69,8 @@ enum class FlipRequestSource { Cpu, GpuEop };
struct VideoOutEventState; struct VideoOutEventState;
struct VideoOutEventRegistration { struct VideoOutEventRegistration {
EventQueue::KernelEqueue handle = EventQueue::KERNEL_EQUEUE_INVALID; EventQueue::KernelEqueue handle = EventQueue::KERNEL_EQUEUE_INVALID;
std::shared_ptr<VideoOutEventState> state; std::shared_ptr<VideoOutEventState> state;
uint64_t generation = 0; uint64_t generation = 0;
VideoOutEventKind kind = VideoOutEventKind::Flip; VideoOutEventKind kind = VideoOutEventKind::Flip;
}; };
@@ -170,13 +170,13 @@ struct BufferAttributeGroup {
struct VideoOutConfig { struct VideoOutConfig {
Common::Mutex mutex; Common::Mutex mutex;
Common::CondVar vblank_cond; Common::CondVar vblank_cond;
std::shared_ptr<VideoOutEventState> events = std::make_shared<VideoOutEventState>(); std::shared_ptr<VideoOutEventState> events = std::make_shared<VideoOutEventState>();
uint32_t width = 0; uint32_t width = 0;
uint32_t height = 0; uint32_t height = 0;
uint64_t generation = 0; uint64_t generation = 0;
bool opened = false; bool opened = false;
bool closing = false; bool closing = false;
int flip_rate = 0; int flip_rate = 0;
uint64_t output_mode = VIDEO_OUT_OUTPUT_MODE_DEFAULT; uint64_t output_mode = VIDEO_OUT_OUTPUT_MODE_DEFAULT;
float gamma = 1.0f; float gamma = 1.0f;
VideoOutFlipStatus flip_status; VideoOutFlipStatus flip_status;
@@ -250,8 +250,8 @@ public:
VideoOutConfig* Get(int handle, uint64_t& generation); VideoOutConfig* Get(int handle, uint64_t& generation);
bool IsOpened(int handle); bool IsOpened(int handle);
void Init(uint32_t width, uint32_t height); void Init(uint32_t width, uint32_t height);
FlipQueue& GetFlipQueue() { return m_flip_queue; } FlipQueue& GetFlipQueue() { return m_flip_queue; }
Graphics::RenderContext& Renderer() const noexcept { return m_renderer; } Graphics::RenderContext& Renderer() const noexcept { return m_renderer; }
void VblankBegin(); void VblankBegin();
@@ -259,12 +259,12 @@ public:
void PresentThread(std::stop_token token); void PresentThread(std::stop_token token);
private: private:
Common::Mutex m_mutex; Common::Mutex m_mutex;
VideoOutConfig m_video_out_ctx[VIDEO_OUT_NUM_MAX]; VideoOutConfig m_video_out_ctx[VIDEO_OUT_NUM_MAX];
Graphics::RenderContext& m_renderer; Graphics::RenderContext& m_renderer;
Graphics::Presenter& m_presenter; Graphics::Presenter& m_presenter;
FlipQueue m_flip_queue; FlipQueue m_flip_queue;
std::jthread m_present_thread; std::jthread m_present_thread;
}; };
static std::unique_ptr<VideoOutDriver> g_video_out_driver; static std::unique_ptr<VideoOutDriver> g_video_out_driver;
@@ -279,7 +279,7 @@ static uintptr_t VideoOutEventId(VideoOutEventKind kind) {
} }
static VideoOutEventQueues& VideoOutEventQueuesFor(VideoOutEventState& state, static VideoOutEventQueues& VideoOutEventQueuesFor(VideoOutEventState& state,
VideoOutEventKind kind) { VideoOutEventKind kind) {
switch (kind) { switch (kind) {
case VideoOutEventKind::Flip: return state.flip; case VideoOutEventKind::Flip: return state.flip;
case VideoOutEventKind::Vblank: return state.vblank; case VideoOutEventKind::Vblank: return state.vblank;
@@ -359,9 +359,9 @@ static void TriggerVideoOutEvents(VideoOutConfig& video_out, VideoOutEventKind k
if (!registration || registration->generation != video_out.generation) { if (!registration || registration->generation != video_out.generation) {
continue; continue;
} }
const auto result = EventQueue::KernelTriggerEvent( const auto result =
registration->handle, VideoOutEventId(kind), EventQueue::KERNEL_EVFILT_VIDEO_OUT, EventQueue::KernelTriggerEvent(registration->handle, VideoOutEventId(kind),
trigger_data); EventQueue::KERNEL_EVFILT_VIDEO_OUT, trigger_data);
EXIT_NOT_IMPLEMENTED(result != OK && result != LibKernel::KERNEL_ERROR_EBADF && EXIT_NOT_IMPLEMENTED(result != OK && result != LibKernel::KERNEL_ERROR_EBADF &&
result != LibKernel::KERNEL_ERROR_ENOENT); result != LibKernel::KERNEL_ERROR_ENOENT);
} }
@@ -372,9 +372,8 @@ static void DeleteVideoOutEvents(const VideoOutEventQueues& queues, VideoOutEven
if (!registration) { if (!registration) {
continue; continue;
} }
const auto result = const auto result = EventQueue::KernelDeleteEvent(
EventQueue::KernelDeleteEvent(registration->handle, VideoOutEventId(kind), registration->handle, VideoOutEventId(kind), EventQueue::KERNEL_EVFILT_VIDEO_OUT);
EventQueue::KERNEL_EVFILT_VIDEO_OUT);
EXIT_NOT_IMPLEMENTED(result != OK && result != LibKernel::KERNEL_ERROR_EBADF && EXIT_NOT_IMPLEMENTED(result != OK && result != LibKernel::KERNEL_ERROR_EBADF &&
result != LibKernel::KERNEL_ERROR_ENOENT); result != LibKernel::KERNEL_ERROR_ENOENT);
} }
@@ -383,7 +382,7 @@ static void DeleteVideoOutEvents(const VideoOutEventQueues& queues, VideoOutEven
static int RegisterVideoOutEvent(int handle, EventQueue::KernelEqueue eq, VideoOutEventKind kind, static int RegisterVideoOutEvent(int handle, EventQueue::KernelEqueue eq, VideoOutEventKind kind,
void* udata) { void* udata) {
uint64_t generation = 0; uint64_t generation = 0;
auto* video_out = DriverState().Get(handle, generation); auto* video_out = DriverState().Get(handle, generation);
if (video_out == nullptr) { if (video_out == nullptr) {
return VIDEO_OUT_ERROR_INVALID_HANDLE; return VIDEO_OUT_ERROR_INVALID_HANDLE;
} }
@@ -425,27 +424,25 @@ static int RegisterVideoOutEvent(int handle, EventQueue::KernelEqueue eq, VideoO
bool add_queue = false; bool add_queue = false;
{ {
Common::LockGuard event_lock(event_state->mutex); Common::LockGuard event_lock(event_state->mutex);
const auto existing = std::find_if(queues.begin(), queues.end(), [&](const auto& candidate) { const auto existing =
return candidate->handle == eq && candidate->generation == generation; std::find_if(queues.begin(), queues.end(), [&](const auto& candidate) {
}); return candidate->handle == eq && candidate->generation == generation;
});
if (existing != queues.end()) { if (existing != queues.end()) {
registration = *existing; registration = *existing;
} else { } else {
registration = std::make_shared<VideoOutEventRegistration>( registration = std::make_shared<VideoOutEventRegistration>(VideoOutEventRegistration {
VideoOutEventRegistration {.handle = eq, .handle = eq, .state = event_state, .generation = generation, .kind = kind});
.state = event_state,
.generation = generation,
.kind = kind});
queues.push_back(registration); queues.push_back(registration);
add_queue = true; add_queue = true;
} }
} }
event.filter.data = registration.get(); event.filter.data = registration.get();
event.filter.owner = registration; event.filter.owner = registration;
const int result = EventQueue::KernelAddEvent(eq, event); const int result = EventQueue::KernelAddEvent(eq, event);
if (result != OK && add_queue) { if (result != OK && add_queue) {
Common::LockGuard event_lock(event_state->mutex); Common::LockGuard event_lock(event_state->mutex);
const auto added = std::find(queues.begin(), queues.end(), registration); const auto added = std::find(queues.begin(), queues.end(), registration);
if (added != queues.end()) { if (added != queues.end()) {
queues.erase(added); queues.erase(added);
} }
@@ -455,7 +452,7 @@ static int RegisterVideoOutEvent(int handle, EventQueue::KernelEqueue eq, VideoO
static int DeleteVideoOutEvent(int handle, EventQueue::KernelEqueue eq, VideoOutEventKind kind) { static int DeleteVideoOutEvent(int handle, EventQueue::KernelEqueue eq, VideoOutEventKind kind) {
uint64_t generation = 0; uint64_t generation = 0;
auto* video_out = DriverState().Get(handle, generation); auto* video_out = DriverState().Get(handle, generation);
if (video_out == nullptr) { if (video_out == nullptr) {
return VIDEO_OUT_ERROR_INVALID_HANDLE; return VIDEO_OUT_ERROR_INVALID_HANDLE;
} }
@@ -814,8 +811,8 @@ void VideoOutDriver::Impl::PresentThread(std::stop_token token) {
m_presenter.Present(*frame, true); m_presenter.Present(*frame, true);
} }
const auto frame_end = Common::Timer::QueryPerformanceCounter(); const auto frame_end = Common::Timer::QueryPerformanceCounter();
total_wait += static_cast<int64_t>(period) - total_wait +=
static_cast<int64_t>(frame_end - frame_begin); static_cast<int64_t>(period) - static_cast<int64_t>(frame_end - frame_begin);
continue; continue;
} }
@@ -841,8 +838,7 @@ void VideoOutDriver::Impl::PresentThread(std::stop_token token) {
VblankEnd(); VblankEnd();
const auto frame_end = Common::Timer::QueryPerformanceCounter(); const auto frame_end = Common::Timer::QueryPerformanceCounter();
total_wait += static_cast<int64_t>(period) - total_wait += static_cast<int64_t>(period) - static_cast<int64_t>(frame_end - frame_begin);
static_cast<int64_t>(frame_end - frame_begin);
} }
} }
@@ -1000,8 +996,8 @@ void FlipQueue::Prepare(uint64_t request_id, Graphics::CommandBuffer& buffer) {
} }
Graphics::Presenter::Frame* frame = nullptr; Graphics::Presenter::Frame* frame = nullptr;
if (special) { if (special) {
frame = &m_presenter.PrepareBlankFrame(width, height, frame = &m_presenter.PrepareBlankFrame(width, height, index == VIDEO_OUT_BUFFER_INDEX_BLACK,
index == VIDEO_OUT_BUFFER_INDEX_BLACK, &buffer); &buffer);
} else { } else {
frame = &m_presenter.PrepareFrame(buffer, source_info); frame = &m_presenter.PrepareFrame(buffer, source_info);
} }
+7 -7
View File
@@ -32,13 +32,13 @@ public:
~VideoOutDriver(); ~VideoOutDriver();
KYTY_CLASS_NO_COPY(VideoOutDriver); KYTY_CLASS_NO_COPY(VideoOutDriver);
int SubmitFlipFromGpu(Graphics::CommandBuffer& buffer, int handle, int index, int flip_mode, int SubmitFlipFromGpu(Graphics::CommandBuffer& buffer, int handle, int index, int flip_mode,
int64_t flip_arg, uint64_t& request_id); int64_t flip_arg, uint64_t& request_id);
void PrepareFlip(uint64_t request_id, Graphics::CommandBuffer& buffer); void PrepareFlip(uint64_t request_id, Graphics::CommandBuffer& buffer);
void CompleteFlip(uint64_t request_id); void CompleteFlip(uint64_t request_id);
void SubmitFlipPreparation(uint64_t request_id); void SubmitFlipPreparation(uint64_t request_id);
void WaitForSubmitSlot(); void WaitForSubmitSlot();
void WaitFlipDone(int handle, int index); void WaitFlipDone(int handle, int index);
[[nodiscard]] Impl& State() noexcept; [[nodiscard]] Impl& State() noexcept;
+61 -83
View File
@@ -61,7 +61,7 @@ namespace Libs::Graphics {
struct Presenter::Frame { struct Presenter::Frame {
VulkanImage image; VulkanImage image;
std::unique_ptr<CommandBuffer> present_commands; std::unique_ptr<CommandBuffer> present_commands;
bool busy = false; bool busy = false;
bool reusing_last = false; bool reusing_last = false;
void Configure(GraphicContext& graphics, vk::Extent2D extent, vk::Format format); void Configure(GraphicContext& graphics, vk::Extent2D extent, vk::Format format);
@@ -155,7 +155,7 @@ public:
EXIT("last submitted frame is not available for reuse\n"); EXIT("last submitted frame is not available for reuse\n");
} }
m_free.erase(free); m_free.erase(free);
m_last_frame = nullptr; m_last_frame = nullptr;
frame->busy = true; frame->busy = true;
frame->reusing_last = true; frame->reusing_last = true;
m_mutex.Unlock(); m_mutex.Unlock();
@@ -197,30 +197,27 @@ private:
} }
} }
WindowContext& m_window; WindowContext& m_window;
Common::Mutex m_mutex; Common::Mutex m_mutex;
Common::CondVar m_available; Common::CondVar m_available;
std::vector<std::unique_ptr<Presenter::Frame>> m_frames; std::vector<std::unique_ptr<Presenter::Frame>> m_frames;
std::deque<Presenter::Frame*> m_free; std::deque<Presenter::Frame*> m_free;
Presenter::Frame* m_last_frame = nullptr; Presenter::Frame* m_last_frame = nullptr;
vk::Format m_format = vk::Format::eUndefined; vk::Format m_format = vk::Format::eUndefined;
}; };
void Presenter::Frame::Configure(GraphicContext& graphics, vk::Extent2D extent, void Presenter::Frame::Configure(GraphicContext& graphics, vk::Extent2D extent, vk::Format format) {
vk::Format format) {
if (extent.width == 0 || extent.height == 0 || format == vk::Format::eUndefined) { if (extent.width == 0 || extent.height == 0 || format == vk::Format::eUndefined) {
EXIT("unsupported prepared frame, extent=%ux%u format=%d\n", extent.width, extent.height, EXIT("unsupported prepared frame, extent=%ux%u format=%d\n", extent.width, extent.height,
static_cast<int>(format)); static_cast<int>(format));
} }
const auto features = graphics.GetFormatProperties(format).optimalTilingFeatures; const auto features = graphics.GetFormatProperties(format).optimalTilingFeatures;
const auto required = vk::FormatFeatureFlagBits::eBlitSrc | const auto required =
vk::FormatFeatureFlagBits::eSampledImageFilterLinear | vk::FormatFeatureFlagBits::eBlitSrc | vk::FormatFeatureFlagBits::eSampledImageFilterLinear |
vk::FormatFeatureFlagBits::eTransferSrc | vk::FormatFeatureFlagBits::eTransferSrc | vk::FormatFeatureFlagBits::eTransferDst;
vk::FormatFeatureFlagBits::eTransferDst;
if ((features & required) != required) { if ((features & required) != required) {
EXIT("prepared presentation format lacks optimal blit support: format=%d features=0x%x\n", EXIT("prepared presentation format lacks optimal blit support: format=%d features=0x%x\n",
static_cast<int>(format), static_cast<int>(format), static_cast<vk::FormatFeatureFlags::MaskType>(features));
static_cast<vk::FormatFeatureFlags::MaskType>(features));
} }
auto& dst = image; auto& dst = image;
@@ -234,11 +231,11 @@ void Presenter::Frame::Configure(GraphicContext& graphics, vk::Extent2D extent,
dst.memory = {}; dst.memory = {};
} }
dst.extent = {extent.width, extent.height, 1}; dst.extent = {extent.width, extent.height, 1};
dst.format = format; dst.format = format;
dst.layers = 1; dst.layers = 1;
dst.mip_levels = 1; dst.mip_levels = 1;
dst.state = {}; dst.state = {};
dst.subresource_states.clear(); dst.subresource_states.clear();
dst.memory.property = vk::MemoryPropertyFlagBits::eDeviceLocal; dst.memory.property = vk::MemoryPropertyFlagBits::eDeviceLocal;
@@ -262,13 +259,12 @@ void Presenter::Frame::Configure(GraphicContext& graphics, vk::Extent2D extent,
void Presenter::Frame::Transit(vk::CommandBuffer command, vk::ImageLayout layout, void Presenter::Frame::Transit(vk::CommandBuffer command, vk::ImageLayout layout,
vk::AccessFlags2 access) { vk::AccessFlags2 access) {
const auto stage = access == vk::AccessFlagBits2::eTransferRead || const auto stage = access == vk::AccessFlagBits2::eTransferRead ||
access == vk::AccessFlagBits2::eTransferWrite access == vk::AccessFlagBits2::eTransferWrite
? vk::PipelineStageFlagBits2::eTransfer ? vk::PipelineStageFlagBits2::eTransfer
: vk::PipelineStageFlagBits2::eAllCommands; : vk::PipelineStageFlagBits2::eAllCommands;
constexpr auto writes = vk::AccessFlagBits2::eTransferWrite | constexpr auto writes = vk::AccessFlagBits2::eTransferWrite |
vk::AccessFlagBits2::eShaderWrite | vk::AccessFlagBits2::eShaderWrite | vk::AccessFlagBits2::eMemoryWrite;
vk::AccessFlagBits2::eMemoryWrite;
if (image.state.layout == layout && image.state.access_mask == access && if (image.state.layout == layout && image.state.access_mask == access &&
!static_cast<bool>(image.state.access_mask & writes)) { !static_cast<bool>(image.state.access_mask & writes)) {
return; return;
@@ -299,35 +295,27 @@ void Presenter::Frame::Transit(vk::CommandBuffer command, vk::ImageLayout layout
void Presenter::Frame::CopyFrom(CommandBuffer& command_buffer, Image& source) { void Presenter::Frame::CopyFrom(CommandBuffer& command_buffer, Image& source) {
command_buffer.EndRendering(); command_buffer.EndRendering();
auto command = command_buffer.Handle(); auto command = command_buffer.Handle();
source.Transit(vk::ImageLayout::eTransferSrcOptimal, source.Transit(vk::ImageLayout::eTransferSrcOptimal, vk::AccessFlagBits2::eTransferRead, {},
vk::AccessFlagBits2::eTransferRead, {}, command); command);
Transit(command, vk::ImageLayout::eTransferDstOptimal, Transit(command, vk::ImageLayout::eTransferDstOptimal, vk::AccessFlagBits2::eTransferWrite);
vk::AccessFlagBits2::eTransferWrite);
vk::ImageCopy copy {}; vk::ImageCopy copy {};
copy.srcSubresource = {vk::ImageAspectFlagBits::eColor, 0, 0, copy.srcSubresource = {vk::ImageAspectFlagBits::eColor, 0, 0, source.backing.layers};
source.backing.layers};
copy.dstSubresource = {vk::ImageAspectFlagBits::eColor, 0, 0, image.layers}; copy.dstSubresource = {vk::ImageAspectFlagBits::eColor, 0, 0, image.layers};
copy.extent = {std::min(source.backing.extent.width, image.extent.width), copy.extent = {std::min(source.backing.extent.width, image.extent.width),
std::min(source.backing.extent.height, image.extent.height), 1}; std::min(source.backing.extent.height, image.extent.height), 1};
EXIT_IF(copy.srcSubresource.layerCount != copy.dstSubresource.layerCount); EXIT_IF(copy.srcSubresource.layerCount != copy.dstSubresource.layerCount);
command.copyImage(source.backing.image, vk::ImageLayout::eTransferSrcOptimal, command.copyImage(source.backing.image, vk::ImageLayout::eTransferSrcOptimal, image.image,
image.image, vk::ImageLayout::eTransferDstOptimal, copy); vk::ImageLayout::eTransferDstOptimal, copy);
Transit(command, vk::ImageLayout::eTransferSrcOptimal, Transit(command, vk::ImageLayout::eTransferSrcOptimal, vk::AccessFlagBits2::eTransferRead);
vk::AccessFlagBits2::eTransferRead);
} }
void Presenter::Frame::Clear(CommandBuffer& command_buffer, void Presenter::Frame::Clear(CommandBuffer& command_buffer, const vk::ClearColorValue& color) {
const vk::ClearColorValue& color) {
command_buffer.EndRendering(); command_buffer.EndRendering();
auto command = command_buffer.Handle(); auto command = command_buffer.Handle();
Transit(command, vk::ImageLayout::eTransferDstOptimal, Transit(command, vk::ImageLayout::eTransferDstOptimal, vk::AccessFlagBits2::eTransferWrite);
vk::AccessFlagBits2::eTransferWrite); const vk::ImageSubresourceRange range {vk::ImageAspectFlagBits::eColor, 0, 1, 0, 1};
const vk::ImageSubresourceRange range { command.clearColorImage(image.image, vk::ImageLayout::eTransferDstOptimal, &color, 1, &range);
vk::ImageAspectFlagBits::eColor, 0, 1, 0, 1}; Transit(command, vk::ImageLayout::eTransferSrcOptimal, vk::AccessFlagBits2::eTransferRead);
command.clearColorImage(image.image, vk::ImageLayout::eTransferDstOptimal, &color, 1,
&range);
Transit(command, vk::ImageLayout::eTransferSrcOptimal,
vk::AccessFlagBits2::eTransferRead);
} }
class Swapchain final { class Swapchain final {
@@ -338,8 +326,8 @@ public:
~Swapchain(); ~Swapchain();
KYTY_CLASS_NO_COPY(Swapchain); KYTY_CLASS_NO_COPY(Swapchain);
void Create(); void Create();
void Recreate(bool surface_lost = false); void Recreate(bool surface_lost = false);
[[nodiscard]] Status AcquireNextImage(); [[nodiscard]] Status AcquireNextImage();
void RecordPresentCommands(CommandBuffer& command, VulkanImage& source); void RecordPresentCommands(CommandBuffer& command, VulkanImage& source);
void Submit(CommandBuffer& command); void Submit(CommandBuffer& command);
@@ -395,17 +383,17 @@ struct Presenter::Impl {
desc.view_info.usage = vk::ImageUsageFlagBits::eTransferSrc; desc.view_info.usage = vk::ImageUsageFlagBits::eTransferSrc;
desc.type = TextureCache::BindingType::VideoOut; desc.type = TextureCache::BindingType::VideoOut;
auto& cache = renderer.GetTextureCache(); auto& cache = renderer.GetTextureCache();
auto& image = cache.GetImage(cache.FindImage(desc)); auto& image = cache.GetImage(cache.FindImage(desc));
image.usage.video_out = true; image.usage.video_out = true;
return image; return image;
} }
RenderContext& renderer; RenderContext& renderer;
WindowContext& window; WindowContext& window;
Swapchain swapchain; Swapchain swapchain;
CommandScheduler present_scheduler; CommandScheduler present_scheduler;
FramePool frames; FramePool frames;
}; };
void Swapchain::Create() { void Swapchain::Create() {
@@ -441,25 +429,20 @@ void Swapchain::Create() {
? vk::CompositeAlphaFlagBitsKHR::eOpaque ? vk::CompositeAlphaFlagBitsKHR::eOpaque
: vk::CompositeAlphaFlagBitsKHR::eInherit; : vk::CompositeAlphaFlagBitsKHR::eInherit;
vk::SurfaceFormatKHR format {vk::Format::eR8G8B8A8Unorm, vk::SurfaceFormatKHR format {vk::Format::eR8G8B8A8Unorm, vk::ColorSpaceKHR::eSrgbNonlinear};
vk::ColorSpaceKHR::eSrgbNonlinear}; if (surface.formats.size() != 1 || surface.formats.front().format != vk::Format::eUndefined) {
if (surface.formats.size() != 1 ||
surface.formats.front().format != vk::Format::eUndefined) {
const auto it = std::find_if(surface.formats.begin(), surface.formats.end(), const auto it = std::find_if(surface.formats.begin(), surface.formats.end(),
[](const vk::SurfaceFormatKHR& candidate) { [](const vk::SurfaceFormatKHR& candidate) {
return candidate.format == return candidate.format == vk::Format::eB8G8R8A8Unorm ||
vk::Format::eB8G8R8A8Unorm || candidate.format == vk::Format::eR8G8B8A8Unorm;
candidate.format ==
vk::Format::eR8G8B8A8Unorm;
}); });
if (it == surface.formats.end()) { if (it == surface.formats.end()) {
EXIT("no supported UNORM swapchain format\n"); EXIT("no supported UNORM swapchain format\n");
} }
format = *it; format = *it;
} }
m_format = format.format; m_format = format.format;
const auto swapchain_features = const auto swapchain_features = graphics.GetFormatProperties(m_format).optimalTilingFeatures;
graphics.GetFormatProperties(m_format).optimalTilingFeatures;
if (!static_cast<bool>(swapchain_features & vk::FormatFeatureFlagBits::eBlitDst)) { if (!static_cast<bool>(swapchain_features & vk::FormatFeatureFlagBits::eBlitDst)) {
EXIT("swapchain format cannot be a blit destination: format=%d\n", EXIT("swapchain format cannot be a blit destination: format=%d\n",
static_cast<int>(m_format)); static_cast<int>(m_format));
@@ -503,9 +486,8 @@ void Swapchain::Create() {
view.subresourceRange.baseMipLevel = 0; view.subresourceRange.baseMipLevel = 0;
view.subresourceRange.layerCount = 1; view.subresourceRange.layerCount = 1;
view.subresourceRange.levelCount = 1; view.subresourceRange.levelCount = 1;
RequireVulkanSuccess( RequireVulkanSuccess(graphics.device.createImageView(&view, nullptr, &m_image_views[i]),
graphics.device.createImageView(&view, nullptr, &m_image_views[i]), "vkCreateImageView");
"vkCreateImageView");
EXIT_IF(m_image_views[i] == nullptr); EXIT_IF(m_image_views[i] == nullptr);
} }
@@ -600,7 +582,7 @@ void Swapchain::Recreate(bool surface_lost) {
Swapchain::Status Swapchain::AcquireNextImage() { Swapchain::Status Swapchain::AcquireNextImage() {
EXIT_IF(m_handle == nullptr || m_frame_index >= m_image_acquired.size()); EXIT_IF(m_handle == nullptr || m_frame_index >= m_image_acquired.size());
m_image_index = static_cast<uint32_t>(-1); m_image_index = static_cast<uint32_t>(-1);
const auto result = m_window.graphic_ctx.device.acquireNextImageKHR( const auto result = m_window.graphic_ctx.device.acquireNextImageKHR(
m_handle, std::numeric_limits<uint64_t>::max(), m_image_acquired[m_frame_index], nullptr, m_handle, std::numeric_limits<uint64_t>::max(), m_image_acquired[m_frame_index], nullptr,
&m_image_index); &m_image_index);
@@ -683,10 +665,9 @@ void Swapchain::RecordPresentCommands(CommandBuffer& command, VulkanImage& sourc
to_present.subresourceRange.levelCount = 1; to_present.subresourceRange.levelCount = 1;
to_present.subresourceRange.baseArrayLayer = 0; to_present.subresourceRange.baseArrayLayer = 0;
to_present.subresourceRange.layerCount = 1; to_present.subresourceRange.layerCount = 1;
vk_command.pipelineBarrier(vk::PipelineStageFlagBits::eAllCommands, vk_command.pipelineBarrier(
vk::PipelineStageFlagBits::eAllCommands, vk::PipelineStageFlagBits::eAllCommands, vk::PipelineStageFlagBits::eAllCommands,
vk::DependencyFlagBits::eByRegion, 0, vk::DependencyFlagBits::eByRegion, 0, nullptr, 0, nullptr, 1, &to_present);
nullptr, 0, nullptr, 1, &to_present);
command.End(); command.End();
} }
@@ -700,7 +681,7 @@ void Swapchain::Submit(CommandBuffer& command) {
Swapchain::Status Swapchain::Present() { Swapchain::Status Swapchain::Present() {
EXIT_IF(m_image_index >= m_render_complete.size()); EXIT_IF(m_image_index >= m_render_complete.size());
const auto ready = m_render_complete[m_image_index]; const auto ready = m_render_complete[m_image_index];
vk::PresentInfoKHR present {}; vk::PresentInfoKHR present {};
present.sType = vk::StructureType::ePresentInfoKHR; present.sType = vk::StructureType::ePresentInfoKHR;
present.swapchainCount = 1; present.swapchainCount = 1;
@@ -738,7 +719,7 @@ Presenter::~Presenter() = default;
Presenter::Frame& Presenter::PrepareFrame(CommandBuffer& buffer, const ImageInfo& info) { Presenter::Frame& Presenter::PrepareFrame(CommandBuffer& buffer, const ImageInfo& info) {
KYTY_PROFILER_FUNCTION(); KYTY_PROFILER_FUNCTION();
EXIT_IF(buffer.IsInvalid()); EXIT_IF(buffer.IsInvalid());
auto* frame = m_impl->frames.Acquire(); auto* frame = m_impl->frames.Acquire();
Common::LockGuard render_lock(m_impl->renderer.GetMutex()); Common::LockGuard render_lock(m_impl->renderer.GetMutex());
auto& image = m_impl->ResolveSurface(info); auto& image = m_impl->ResolveSurface(info);
if (image.backing.format == vk::Format::eUndefined) { if (image.backing.format == vk::Format::eUndefined) {
@@ -752,14 +733,13 @@ Presenter::Frame& Presenter::PrepareFrame(CommandBuffer& buffer, const ImageInfo
default: break; default: break;
} }
frame->Configure(m_impl->window.graphic_ctx, frame->Configure(m_impl->window.graphic_ctx,
{image.backing.extent.width, image.backing.extent.height}, {image.backing.extent.width, image.backing.extent.height}, frame_format);
frame_format);
frame->CopyFrom(buffer, image); frame->CopyFrom(buffer, image);
return *frame; return *frame;
} }
Presenter::Frame& Presenter::PrepareBlankFrame(uint32_t width, uint32_t height, bool opaque, Presenter::Frame& Presenter::PrepareBlankFrame(uint32_t width, uint32_t height, bool opaque,
CommandBuffer* producer) { CommandBuffer* producer) {
KYTY_PROFILER_FUNCTION(); KYTY_PROFILER_FUNCTION();
auto format = m_impl->frames.GetFormat(); auto format = m_impl->frames.GetFormat();
auto* frame = m_impl->frames.Acquire(); auto* frame = m_impl->frames.Acquire();
@@ -772,8 +752,7 @@ Presenter::Frame& Presenter::PrepareBlankFrame(uint32_t width, uint32_t height,
frame->Clear(*producer, clear); frame->Clear(*producer, clear);
} else { } else {
if (frame->present_commands == nullptr) { if (frame->present_commands == nullptr) {
frame->present_commands = frame->present_commands = std::make_unique<CommandBuffer>(m_impl->present_scheduler);
std::make_unique<CommandBuffer>(m_impl->present_scheduler);
} }
auto& command = *frame->present_commands; auto& command = *frame->present_commands;
command.WaitForFenceAndReset(); command.WaitForFenceAndReset();
@@ -830,8 +809,7 @@ void Presenter::Present(Frame& frame, bool reuse) {
continue; continue;
} }
if (frame.present_commands == nullptr) { if (frame.present_commands == nullptr) {
frame.present_commands = frame.present_commands = std::make_unique<CommandBuffer>(m_impl->present_scheduler);
std::make_unique<CommandBuffer>(m_impl->present_scheduler);
} }
{ {
Common::LockGuard render_lock(m_impl->renderer.GetMutex()); Common::LockGuard render_lock(m_impl->renderer.GetMutex());
@@ -32,11 +32,11 @@
#include "graphics/host_gpu/vma.h" #include "graphics/host_gpu/vma.h"
#include "graphics/host_gpu/vulkanCommon.h" #include "graphics/host_gpu/vulkanCommon.h"
#include "graphics/presentation/presenter.h" #include "graphics/presentation/presenter.h"
#include "kernel/memory.h"
#include "graphics/presentation/renderDoc.h" #include "graphics/presentation/renderDoc.h"
#include "graphics/presentation/videoOut.h" #include "graphics/presentation/videoOut.h"
#include "graphics/presentation/window.h" #include "graphics/presentation/window.h"
#include "graphics/presentation/window/windowInternal.h" #include "graphics/presentation/window/windowInternal.h"
#include "kernel/memory.h"
#include "libs/controller.h" #include "libs/controller.h"
#include "loader/systemContent.h" #include "loader/systemContent.h"
@@ -475,9 +475,9 @@ static void VulkanInitSubgroupSizeControl(vk::PhysicalDevice physical_device,
} }
static vk::Device VulkanCreateDevice(vk::PhysicalDevice physical_device, const VulkanExtensions& r, static vk::Device VulkanCreateDevice(vk::PhysicalDevice physical_device, const VulkanExtensions& r,
uint32_t queue_family, uint32_t queue_family,
const std::vector<const char*>& device_extensions, const std::vector<const char*>& device_extensions,
GraphicContext& graphics) { GraphicContext& graphics) {
EXIT_IF(physical_device == nullptr); EXIT_IF(physical_device == nullptr);
EXIT_IF(queue_family == static_cast<uint32_t>(-1)); EXIT_IF(queue_family == static_cast<uint32_t>(-1));
@@ -551,19 +551,19 @@ static vk::Device VulkanCreateDevice(vk::PhysicalDevice physical_device, const V
features12.timelineSemaphore = VK_TRUE; features12.timelineSemaphore = VK_TRUE;
vk::PhysicalDeviceFeatures device_features {}; vk::PhysicalDeviceFeatures device_features {};
device_features.fragmentStoresAndAtomics = VK_TRUE; device_features.fragmentStoresAndAtomics = VK_TRUE;
device_features.samplerAnisotropy = VK_TRUE; device_features.samplerAnisotropy = VK_TRUE;
device_features.robustBufferAccess = VK_TRUE; device_features.robustBufferAccess = VK_TRUE;
#if !defined(__APPLE__) #if !defined(__APPLE__)
device_features.depthBounds = VK_TRUE; // unsupported by MoltenVK device_features.depthBounds = VK_TRUE; // unsupported by MoltenVK
#endif #endif
device_features.shaderStorageImageWriteWithoutFormat = VK_TRUE; device_features.shaderStorageImageWriteWithoutFormat = VK_TRUE;
device_features.shaderStorageImageReadWithoutFormat = VK_TRUE; device_features.shaderStorageImageReadWithoutFormat = VK_TRUE;
device_features.shaderImageGatherExtended = VK_TRUE; device_features.shaderImageGatherExtended = VK_TRUE;
device_features.independentBlend = VK_TRUE; device_features.independentBlend = VK_TRUE;
device_features.tessellationShader = VK_TRUE; device_features.tessellationShader = VK_TRUE;
device_features.sampleRateShading = VK_TRUE; device_features.sampleRateShading = VK_TRUE;
graphics.sample_rate_shading_enabled = true; graphics.sample_rate_shading_enabled = true;
device_features.vertexPipelineStoresAndAtomics = device_features.vertexPipelineStoresAndAtomics =
supported_features2.features.vertexPipelineStoresAndAtomics; supported_features2.features.vertexPipelineStoresAndAtomics;
@@ -909,10 +909,9 @@ void WindowContext::CreateVulkan() {
} }
surface = native_surface; surface = native_surface;
std::vector<const char*> device_extensions = {VK_KHR_SWAPCHAIN_EXTENSION_NAME, std::vector<const char*> device_extensions = {
VK_EXT_DEPTH_CLIP_CONTROL_EXTENSION_NAME, VK_KHR_SWAPCHAIN_EXTENSION_NAME, VK_EXT_DEPTH_CLIP_CONTROL_EXTENSION_NAME,
VK_KHR_PUSH_DESCRIPTOR_EXTENSION_NAME, VK_KHR_PUSH_DESCRIPTOR_EXTENSION_NAME, "VK_KHR_maintenance1"};
"VK_KHR_maintenance1"};
#if defined(__APPLE__) #if defined(__APPLE__)
// MoltenVK lacks VK_EXT_depth_clip_enable and VK_EXT_color_write_enable; the renderer // MoltenVK lacks VK_EXT_depth_clip_enable and VK_EXT_color_write_enable; the renderer
@@ -932,8 +931,8 @@ void WindowContext::CreateVulkan() {
uint32_t queue_family = static_cast<uint32_t>(-1); uint32_t queue_family = static_cast<uint32_t>(-1);
VulkanFindPhysicalDevice(graphic_ctx.instance, surface, device_extensions, VulkanFindPhysicalDevice(graphic_ctx.instance, surface, device_extensions, surface_capabilities,
surface_capabilities, graphic_ctx.physical_device, queue_family); graphic_ctx.physical_device, queue_family);
if (graphic_ctx.physical_device == nullptr) { if (graphic_ctx.physical_device == nullptr) {
EXIT("Could not find suitable device"); EXIT("Could not find suitable device");
@@ -949,9 +948,8 @@ void WindowContext::CreateVulkan() {
auto available_extensions = EnumerateVulkan<vk::ExtensionProperties>( auto available_extensions = EnumerateVulkan<vk::ExtensionProperties>(
"vkEnumerateDeviceExtensionProperties", "vkEnumerateDeviceExtensionProperties",
[&](uint32_t* count, vk::ExtensionProperties* values) { [&](uint32_t* count, vk::ExtensionProperties* values) {
return graphic_ctx.physical_device.enumerateDeviceExtensionProperties(nullptr, return graphic_ctx.physical_device.enumerateDeviceExtensionProperties(
count, nullptr, count, values);
values);
}); });
if (HasExtension(available_extensions, VK_EXT_MEMORY_BUDGET_EXTENSION_NAME)) { if (HasExtension(available_extensions, VK_EXT_MEMORY_BUDGET_EXTENSION_NAME)) {
@@ -985,7 +983,7 @@ void WindowContext::CreateVulkan() {
render_context = std::make_unique<RenderContext>(graphic_ctx); render_context = std::make_unique<RenderContext>(graphic_ctx);
LibKernel::Memory::InstallGpuResources(&render_context->GetGpuResources()); LibKernel::Memory::InstallGpuResources(&render_context->GetGpuResources());
presenter = std::make_unique<Presenter>(*this); presenter = std::make_unique<Presenter>(*this);
RenderDocSetActiveWindow(graphic_ctx.instance, window); RenderDocSetActiveWindow(graphic_ctx.instance, window);
} }
+23 -25
View File
@@ -1,7 +1,5 @@
#include "graphics/presentation/window.h" #include "graphics/presentation/window.h"
#include <cstdlib>
#include "SDL.h" #include "SDL.h"
#include "SDL_error.h" #include "SDL_error.h"
#include "SDL_events.h" #include "SDL_events.h"
@@ -40,6 +38,7 @@
#include <algorithm> #include <algorithm>
#include <cstdio> #include <cstdio>
#include <cstdlib>
#include <cstring> #include <cstring>
#include <memory> #include <memory>
#include <string> #include <string>
@@ -59,7 +58,7 @@
namespace Libs::Graphics { namespace Libs::Graphics {
constexpr int KEYBOARD_CONTROLLER_ID = -1000; constexpr int KEYBOARD_CONTROLLER_ID = -1000;
struct EventKeyboard { struct EventKeyboard {
bool down; bool down;
@@ -251,9 +250,7 @@ static void GameEventKeyboard(WindowLoopState& game, const EventKeyboard& key) {
if (key.down) { if (key.down) {
switch (key.key_code) { switch (key.key_code) {
case SDLK_ESCAPE: game.need_exit = true; break; case SDLK_ESCAPE: game.need_exit = true; break;
case SDLK_SPACE: case SDLK_SPACE: SetPause(game, !game.paused.load(std::memory_order_acquire)); break;
SetPause(game, !game.paused.load(std::memory_order_acquire));
break;
case SDLK_F1: case SDLK_F1:
if (!key.repeat) { if (!key.repeat) {
RenderDocRequestCapture(); RenderDocRequestCapture();
@@ -390,7 +387,9 @@ void WindowContext::Resize(uint32_t new_width, uint32_t new_height) {
void WindowContext::ProcessWindowEvent(const SDL_WindowEvent& event) { void WindowContext::ProcessWindowEvent(const SDL_WindowEvent& event) {
const auto& window_event = event; const auto& window_event = event;
switch (window_event.event) { switch (window_event.event) {
case SDL_WINDOWEVENT_SHOWN: LOGF("Window %" PRIu32 " shown\n", window_event.windowID); break; case SDL_WINDOWEVENT_SHOWN:
LOGF("Window %" PRIu32 " shown\n", window_event.windowID);
break;
case SDL_WINDOWEVENT_HIDDEN: case SDL_WINDOWEVENT_HIDDEN:
LOGF("Window %" PRIu32 " hidden\n", window_event.windowID); LOGF("Window %" PRIu32 " hidden\n", window_event.windowID);
@@ -401,13 +400,13 @@ void WindowContext::ProcessWindowEvent(const SDL_WindowEvent& event) {
break; break;
case SDL_WINDOWEVENT_MOVED: case SDL_WINDOWEVENT_MOVED:
LOGF("Window %" PRIu32 " moved to %" PRId32 ",%" PRId32 "\n", LOGF("Window %" PRIu32 " moved to %" PRId32 ",%" PRId32 "\n", window_event.windowID,
window_event.windowID, window_event.data1, window_event.data2); window_event.data1, window_event.data2);
break; break;
case SDL_WINDOWEVENT_RESIZED: case SDL_WINDOWEVENT_RESIZED:
LOGF("Window %" PRIu32 " resized to %" PRId32 "x%" PRId32 "\n", LOGF("Window %" PRIu32 " resized to %" PRId32 "x%" PRId32 "\n", window_event.windowID,
window_event.windowID, window_event.data1, window_event.data2); window_event.data1, window_event.data2);
LOGF("m: %d\n", static_cast<int>(SDL_ThreadID())); LOGF("m: %d\n", static_cast<int>(SDL_ThreadID()));
Resize(window_event.data1, window_event.data2); Resize(window_event.data1, window_event.data2);
@@ -807,9 +806,8 @@ static void WindowCreate(WindowContext& context) {
window_flags |= static_cast<uint32_t>(SDL_WINDOW_BORDERLESS); window_flags |= static_cast<uint32_t>(SDL_WINDOW_BORDERLESS);
} }
#endif #endif
context.window = context.window = SDL_CreateWindow(KYTY_SDL_WINDOW_CAPTION, KYTY_SDL_WINDOWPOS_CENTERED,
SDL_CreateWindow(KYTY_SDL_WINDOW_CAPTION, KYTY_SDL_WINDOWPOS_CENTERED, KYTY_SDL_WINDOWPOS_CENTERED, width, height, window_flags);
KYTY_SDL_WINDOWPOS_CENTERED, width, height, window_flags);
context.window_hidden = true; context.window_hidden = true;
@@ -832,7 +830,7 @@ Presenter& WindowInit(uint32_t width, uint32_t height) {
WindowCreate(*window); WindowCreate(*window);
window->CreateVulkan(); window->CreateVulkan();
auto& presenter = *window->presenter; auto& presenter = *window->presenter;
g_window = std::move(window); g_window = std::move(window);
return presenter; return presenter;
} }
@@ -934,9 +932,9 @@ void WindowContext::UpdateTitle() {
Loader::SystemContentParamSfoGetString("TITLE_ID", title_id, sizeof(title_id)); Loader::SystemContentParamSfoGetString("TITLE_ID", title_id, sizeof(title_id));
static bool has_app_ver = static bool has_app_ver =
Loader::SystemContentParamSfoGetString("APP_VER", app_ver, sizeof(app_ver)); Loader::SystemContentParamSfoGetString("APP_VER", app_ver, sizeof(app_ver));
static uint64_t fps_start = Common::Timer::QueryPerformanceCounter(); static uint64_t fps_start = Common::Timer::QueryPerformanceCounter();
static uint64_t frame_num = 0; static uint64_t frame_num = 0;
static uint64_t fps_frames = 0; static uint64_t fps_frames = 0;
static double current_fps = 0.0; static double current_fps = 0.0;
const auto now = Common::Timer::QueryPerformanceCounter(); const auto now = Common::Timer::QueryPerformanceCounter();
@@ -946,15 +944,15 @@ void WindowContext::UpdateTitle() {
if (now - fps_start >= frequency) { if (now - fps_start >= frequency) {
current_fps = static_cast<double>(fps_frames) * static_cast<double>(frequency) / current_fps = static_cast<double>(fps_frames) * static_cast<double>(frequency) /
static_cast<double>(now - fps_start); static_cast<double>(now - fps_start);
fps_start = now; fps_start = now;
fps_frames = 0; fps_frames = 0;
} }
auto fps = fmt::format("{}{}{}{}{}{}[{}] [{}], frame: {}, fps: {:f}", (has_title ? title : ""), auto fps =
(has_title ? ", " : ""), (has_title_id ? title_id : ""), fmt::format("{}{}{}{}{}{}[{}] [{}], frame: {}, fps: {:f}", (has_title ? title : ""),
(has_title_id ? ", " : ""), (has_app_ver ? app_ver : ""), (has_title ? ", " : ""), (has_title_id ? title_id : ""),
(has_app_ver ? " " : ""), device_name, processor_name, (has_title_id ? ", " : ""), (has_app_ver ? app_ver : ""),
frame_num, current_fps); (has_app_ver ? " " : ""), device_name, processor_name, frame_num, current_fps);
#if defined(__APPLE__) #if defined(__APPLE__)
// AppKit traps on title changes off the main thread; fire-and-forget keeps present pacing. // AppKit traps on title changes off the main thread; fire-and-forget keeps present pacing.
@@ -28,8 +28,8 @@ struct SurfaceCapabilities {
}; };
struct WindowLoopState { struct WindowLoopState {
SDL_Event event {}; SDL_Event event {};
bool need_exit = false; bool need_exit = false;
std::atomic_bool paused = false; std::atomic_bool paused = false;
}; };
@@ -38,14 +38,13 @@ struct WindowContext {
~WindowContext(); ~WindowContext();
KYTY_CLASS_NO_COPY(WindowContext); KYTY_CLASS_NO_COPY(WindowContext);
[[nodiscard]] static vk::PhysicalDeviceVulkan13Features [[nodiscard]] static vk::PhysicalDeviceVulkan13Features RequiredVulkan13Features() noexcept;
RequiredVulkan13Features() noexcept; void CreateVulkan();
void CreateVulkan(); void RecreateSurface();
void RecreateSurface(); void RefreshSurfaceCapabilities();
void RefreshSurfaceCapabilities(); void UpdateIcon();
void UpdateIcon(); void UpdateTitle();
void UpdateTitle(); void Resize(uint32_t width, uint32_t height);
void Resize(uint32_t width, uint32_t height);
void ProcessWindowEvent(const SDL_WindowEvent& event); void ProcessWindowEvent(const SDL_WindowEvent& event);
void ProcessDisplayEvent(const SDL_DisplayEvent& event); void ProcessDisplayEvent(const SDL_DisplayEvent& event);
void ProcessEvent(double time_seconds); void ProcessEvent(double time_seconds);
@@ -59,14 +58,14 @@ struct WindowContext {
void DrainMainThreadTasks(); void DrainMainThreadTasks();
#endif #endif
GraphicContext graphic_ctx; GraphicContext graphic_ctx;
SDL_Window* window = nullptr; SDL_Window* window = nullptr;
bool window_hidden = true; bool window_hidden = true;
vk::SurfaceKHR surface = nullptr; vk::SurfaceKHR surface = nullptr;
SurfaceCapabilities surface_capabilities; SurfaceCapabilities surface_capabilities;
std::unique_ptr<RenderContext> render_context; std::unique_ptr<RenderContext> render_context;
std::unique_ptr<Presenter> presenter; std::unique_ptr<Presenter> presenter;
WindowLoopState loop; WindowLoopState loop;
char device_name[VK_MAX_PHYSICAL_DEVICE_NAME_SIZE] = {0}; char device_name[VK_MAX_PHYSICAL_DEVICE_NAME_SIZE] = {0};
char processor_name[64] = {0}; char processor_name[64] = {0};
@@ -76,7 +75,7 @@ struct WindowContext {
#if defined(__APPLE__) #if defined(__APPLE__)
Common::Mutex main_task_mutex; Common::Mutex main_task_mutex;
Common::CondVar main_task_done; Common::CondVar main_task_done;
std::vector<std::function<void()>> main_tasks; // guarded by main_task_mutex std::vector<std::function<void()>> main_tasks; // guarded by main_task_mutex
uint64_t main_tasks_queued = 0; // guarded by main_task_mutex uint64_t main_tasks_queued = 0; // guarded by main_task_mutex
uint64_t main_tasks_run = 0; // guarded by main_task_mutex uint64_t main_tasks_run = 0; // guarded by main_task_mutex
#endif #endif
@@ -2,15 +2,16 @@
#include "common/assert.h" #include "common/assert.h"
#include "common/logging/log.h" #include "common/logging/log.h"
#include "graphics/shader/recompiler/cfg/ShaderCFG.h"
#include "graphics/shader/recompiler/decompiler/ShaderDecoder.h"
#include "graphics/shader/recompiler/emitter/SpirvEmitter.h"
#include "graphics/shader/recompiler/ir/BindingLayout.h" #include "graphics/shader/recompiler/ir/BindingLayout.h"
#include "graphics/shader/recompiler/ir/ReadLaneElimination.h"
#include "graphics/shader/recompiler/ir/ResourceMaterialization.h" #include "graphics/shader/recompiler/ir/ResourceMaterialization.h"
#include "graphics/shader/recompiler/ir/ResourceTracking.h" #include "graphics/shader/recompiler/ir/ResourceTracking.h"
#include "graphics/shader/recompiler/ir/ScalarProvenance.h" #include "graphics/shader/recompiler/ir/ScalarProvenance.h"
#include "graphics/shader/recompiler/cfg/ShaderCFG.h"
#include "graphics/shader/recompiler/decompiler/ShaderDecoder.h"
#include "graphics/shader/recompiler/ir/ShaderIR.h" #include "graphics/shader/recompiler/ir/ShaderIR.h"
#include "graphics/shader/recompiler/ir/ShaderInfoCollection.h" #include "graphics/shader/recompiler/ir/ShaderInfoCollection.h"
#include "graphics/shader/recompiler/emitter/SpirvEmitter.h"
#include "graphics/shader/recompiler/ir/SrtPatcher.h" #include "graphics/shader/recompiler/ir/SrtPatcher.h"
#include "graphics/shader/recompiler/ir/SrtWalker.h" #include "graphics/shader/recompiler/ir/SrtWalker.h"
@@ -838,6 +839,11 @@ bool TryRecompile(std::span<const uint32_t> code, const CompileOptions& options,
if (!IR::AllocateBindings(ir, layout_options, error)) { if (!IR::AllocateBindings(ir, layout_options, error)) {
return false; return false;
} }
const auto read_lane_stats = IR::EliminateReadLane(ir);
if (read_lane_stats.rewritten_reads != 0) {
LOGF("%s read-lane elimination: reads=%" PRIu32 " shadow_writes=%" PRIu32 "\n",
GetDumpLabel(options), read_lane_stats.rewritten_reads, read_lane_stats.shadow_writes);
}
std::string ir_dump; std::string ir_dump;
if (options.dump_ir) { if (options.dump_ir) {
ir_dump = MakeIrDump(cfg, ir); ir_dump = MakeIrDump(cfg, ir);
@@ -44,7 +44,7 @@ struct CompileResult {
}; };
bool TryRecompile(std::span<const uint32_t> code, const CompileOptions& options, bool TryRecompile(std::span<const uint32_t> code, const CompileOptions& options,
CompileResult& result, std::string* error); CompileResult& result, std::string* error);
} // namespace Libs::Graphics::ShaderRecompiler } // namespace Libs::Graphics::ShaderRecompiler
@@ -35,9 +35,9 @@ constexpr ImageDimension DecodeImageDimension(uint32_t dim) {
case 2u: return ImageDimension::Dim3D; case 2u: return ImageDimension::Dim3D;
case 3u: return ImageDimension::Dim2DArray; case 3u: return ImageDimension::Dim2DArray;
case 4u: return ImageDimension::Dim1DArray; case 4u: return ImageDimension::Dim1DArray;
case 5u: case 5u: return ImageDimension::Dim2DArray;
case 7u: return ImageDimension::Dim2DArray; case 6u: return ImageDimension::Dim2DMsaa;
case 6u: return ImageDimension::Dim2D; case 7u: return ImageDimension::Dim2DMsaaArray;
default: return ImageDimension::Unknown; default: return ImageDimension::Unknown;
} }
} }
@@ -46,8 +46,10 @@ constexpr uint32_t ImageCoordComponents(ImageDimension dimension) {
switch (dimension) { switch (dimension) {
case ImageDimension::Dim1D: return 1u; case ImageDimension::Dim1D: return 1u;
case ImageDimension::Dim1DArray: return 2u; case ImageDimension::Dim1DArray: return 2u;
case ImageDimension::Dim2DMsaa:
case ImageDimension::Dim3D: case ImageDimension::Dim3D:
case ImageDimension::Dim2DArray: return 3u; case ImageDimension::Dim2DArray: return 3u;
case ImageDimension::Dim2DMsaaArray: return 4u;
default: return 2u; default: return 2u;
} }
} }
@@ -196,10 +196,10 @@ bool DecodeSopk(uint32_t pc, std::span<const uint32_t> code, uint32_t word_index
case Opcode::SMovkI32: return DecodeScalarDestination(sdst, pc, inst.dst, error); case Opcode::SMovkI32: return DecodeScalarDestination(sdst, pc, inst.dst, error);
case Opcode::SWaitcnt: { case Opcode::SWaitcnt: {
const uint32_t waitcnt = word & 0xffffu; const uint32_t waitcnt = word & 0xffffu;
inst.dst.kind = OperandKind::Null; inst.dst.kind = OperandKind::Null;
inst.src0.signed_val = static_cast<int32_t>(waitcnt); inst.src0.signed_val = static_cast<int32_t>(waitcnt);
inst.src0.value = waitcnt; inst.src0.value = waitcnt;
inst.src_count = 1; inst.src_count = 1;
return true; return true;
} }
case Opcode::SSetregB32: case Opcode::SSetregB32:
@@ -266,10 +266,10 @@ bool DecodeSopp(uint32_t pc, std::span<const uint32_t> code, uint32_t word_index
inst.src0.value = simm; inst.src0.value = simm;
inst.src0.signed_val = static_cast<int16_t>(simm); inst.src0.signed_val = static_cast<int16_t>(simm);
inst.src_count = (inst.opcode == Opcode::SNop || inst.opcode == Opcode::SWaitcnt || inst.src_count = (inst.opcode == Opcode::SNop || inst.opcode == Opcode::SWaitcnt ||
inst.opcode == Opcode::SSleep || inst.opcode == Opcode::SSendmsg || inst.opcode == Opcode::SSleep || inst.opcode == Opcode::SSendmsg ||
inst.opcode == Opcode::STtraceData || inst.opcode == Opcode::SInstPrefetch) inst.opcode == Opcode::STtraceData || inst.opcode == Opcode::SInstPrefetch)
? 1 ? 1
: 0; : 0;
inst.branch_offset = static_cast<int32_t>(static_cast<int16_t>(simm)) * 4; inst.branch_offset = static_cast<int32_t>(static_cast<int16_t>(simm)) * 4;
inst.branch_target = pc + 4u + static_cast<uint32_t>(inst.branch_offset); inst.branch_target = pc + 4u + static_cast<uint32_t>(inst.branch_offset);
SetRawWords(inst, code, word_index, 1); SetRawWords(inst, code, word_index, 1);
@@ -194,6 +194,8 @@ const char* ImageDimensionToString(ImageDimension dimension) {
case ImageDimension::Dim2D: return "2d"; case ImageDimension::Dim2D: return "2d";
case ImageDimension::Dim3D: return "3d"; case ImageDimension::Dim3D: return "3d";
case ImageDimension::Dim2DArray: return "2d_array"; case ImageDimension::Dim2DArray: return "2d_array";
case ImageDimension::Dim2DMsaa: return "2d_msaa";
case ImageDimension::Dim2DMsaaArray: return "2d_msaa_array";
default: return "unknown"; default: return "unknown";
} }
} }
@@ -220,9 +222,9 @@ bool DecodeScalarSource(uint32_t code, uint32_t pc, Operand& operand, std::strin
} }
if (code >= 240u && code <= 247u) { if (code >= 240u && code <= 247u) {
constexpr float values[] = {0.5f, -0.5f, 1.0f, -1.0f, 2.0f, -2.0f, 4.0f, -4.0f}; constexpr float values[] = {0.5f, -0.5f, 1.0f, -1.0f, 2.0f, -2.0f, 4.0f, -4.0f};
operand.kind = OperandKind::FloatInlineConstant; operand.kind = OperandKind::FloatInlineConstant;
operand.float_val = values[code - 240u]; operand.float_val = values[code - 240u];
operand.value = FloatBits(operand.float_val); operand.value = FloatBits(operand.float_val);
return true; return true;
} }
if (code >= 256u && code <= 511u) { if (code >= 256u && code <= 511u) {
@@ -285,7 +287,7 @@ bool DecodeVectorGpr(uint32_t reg, Operand& operand, std::string* error) {
SetError(error, "VGPR index is out of range"); SetError(error, "VGPR index is out of range");
return false; return false;
} }
operand = {}; operand = {};
operand.kind = OperandKind::Vgpr; operand.kind = OperandKind::Vgpr;
operand.reg = reg; operand.reg = reg;
return true; return true;
@@ -575,6 +575,8 @@ enum class ImageDimension : uint32_t {
Dim2D, Dim2D,
Dim3D, Dim3D,
Dim2DArray, Dim2DArray,
Dim2DMsaa,
Dim2DMsaaArray,
}; };
constexpr uint32_t MaxInstructionRawWords = 5u; constexpr uint32_t MaxInstructionRawWords = 5u;
@@ -30,10 +30,16 @@ bool ImageBinding(const IR::ImageResource& image, IR::DescriptorBindingKind& kin
kind = integer ? Kind::SampledUint1DArray : Kind::Sampled1DArray; kind = integer ? Kind::SampledUint1DArray : Kind::Sampled1DArray;
return true; return true;
case Dim::Dim2D: kind = integer ? Kind::SampledUint2D : Kind::Sampled2D; return true; case Dim::Dim2D: kind = integer ? Kind::SampledUint2D : Kind::Sampled2D; return true;
case Dim::Dim2DMsaa:
kind = integer ? Kind::SampledUint2DMsaa : Kind::Sampled2DMsaa;
return true;
case Dim::Dim3D: kind = integer ? Kind::SampledUint3D : Kind::Sampled3D; return true; case Dim::Dim3D: kind = integer ? Kind::SampledUint3D : Kind::Sampled3D; return true;
case Dim::Dim2DArray: case Dim::Dim2DArray:
kind = integer ? Kind::SampledUint2DArray : Kind::Sampled2DArray; kind = integer ? Kind::SampledUint2DArray : Kind::Sampled2DArray;
return true; return true;
case Dim::Dim2DMsaaArray:
kind = integer ? Kind::SampledUint2DMsaaArray : Kind::Sampled2DMsaaArray;
return true;
case Dim::Unknown: return false; case Dim::Unknown: return false;
} }
} }
@@ -51,6 +57,8 @@ bool ImageBinding(const IR::ImageResource& image, IR::DescriptorBindingKind& kin
case Dim::Dim2DArray: case Dim::Dim2DArray:
kind = uint_image ? Kind::StorageUint2DArray : Kind::Storage2DArray; kind = uint_image ? Kind::StorageUint2DArray : Kind::Storage2DArray;
return true; return true;
case Dim::Dim2DMsaa:
case Dim::Dim2DMsaaArray: return false;
case Dim::Unknown: return false; case Dim::Unknown: return false;
} }
return false; return false;
@@ -12,9 +12,9 @@ namespace Libs::Graphics::ShaderRecompiler::Spirv {
bool ProgramRequiresExactSubgroupSize(const IR::Program& program); bool ProgramRequiresExactSubgroupSize(const IR::Program& program);
bool EmitProgram(const IR::Program& program, const IR::ResourceSnapshot& resources, bool EmitProgram(const IR::Program& program, const IR::ResourceSnapshot& resources,
const ShaderVertexInputInfo* vertex_input_info, const ShaderVertexInputInfo* vertex_input_info,
const ShaderPixelInputInfo* pixel_input_info, const ShaderPixelInputInfo* pixel_input_info,
const ShaderComputeInputInfo* compute_input_info, std::vector<uint32_t>& spirv, const ShaderComputeInputInfo* compute_input_info, std::vector<uint32_t>& spirv,
std::string* error); std::string* error);
} // namespace Libs::Graphics::ShaderRecompiler::Spirv } // namespace Libs::Graphics::ShaderRecompiler::Spirv
@@ -129,7 +129,7 @@ uint32_t MaxCollectedVectorRegisterEnd(const std::vector<RegisterBinding>& regis
} }
void CollectMoveRelSourceRegisters(const IR::Program& program, void CollectMoveRelSourceRegisters(const IR::Program& program,
std::vector<RegisterBinding>& registers) { std::vector<RegisterBinding>& registers) {
const auto max_vector_end = MaxCollectedVectorRegisterEnd(registers); const auto max_vector_end = MaxCollectedVectorRegisterEnd(registers);
for (const auto& block: program.blocks) { for (const auto& block: program.blocks) {
for (const auto& inst: block.instructions) { for (const auto& inst: block.instructions) {
@@ -276,8 +276,7 @@ void CopyProgramInputsAndOutputs(EmitterState& state, const IR::Program& program
if (HasOutput(state.outputs, output.kind, output.index)) { if (HasOutput(state.outputs, output.kind, output.index)) {
continue; continue;
} }
state.outputs.push_back( state.outputs.push_back({output.kind, output.index, output.location, 0, output.debug_name});
{output.kind, output.index, output.location, 0, output.debug_name});
} }
} }
@@ -576,6 +575,8 @@ ImageViewKind ImageViewKindFromDimension(Decoder::ImageDimension dimension) {
case Decoder::ImageDimension::Dim1DArray: return ImageViewKind::Dim1DArray; case Decoder::ImageDimension::Dim1DArray: return ImageViewKind::Dim1DArray;
case Decoder::ImageDimension::Dim2DArray: return ImageViewKind::Dim2DArray; case Decoder::ImageDimension::Dim2DArray: return ImageViewKind::Dim2DArray;
case Decoder::ImageDimension::Dim3D: return ImageViewKind::Dim3D; case Decoder::ImageDimension::Dim3D: return ImageViewKind::Dim3D;
case Decoder::ImageDimension::Dim2DMsaa: return ImageViewKind::Dim2DMsaa;
case Decoder::ImageDimension::Dim2DMsaaArray: return ImageViewKind::Dim2DMsaaArray;
default: return ImageViewKind::Dim2D; default: return ImageViewKind::Dim2D;
} }
} }
@@ -601,7 +602,9 @@ uint32_t ImageViewCoordinateComponents(ImageViewKind view) {
case ImageViewKind::Dim1DArray: case ImageViewKind::Dim1DArray:
case ImageViewKind::Dim2D: return 2u; case ImageViewKind::Dim2D: return 2u;
case ImageViewKind::Dim2DArray: case ImageViewKind::Dim2DArray:
case ImageViewKind::Dim2DMsaaArray:
case ImageViewKind::Dim3D: return 3u; case ImageViewKind::Dim3D: return 3u;
case ImageViewKind::Dim2DMsaa: return 2u;
default: return 0u; default: return 0u;
} }
} }
@@ -611,7 +614,9 @@ uint32_t ImageViewSpatialComponents(ImageViewKind view) {
case ImageViewKind::Dim1D: case ImageViewKind::Dim1D:
case ImageViewKind::Dim1DArray: return 1u; case ImageViewKind::Dim1DArray: return 1u;
case ImageViewKind::Dim2D: case ImageViewKind::Dim2D:
case ImageViewKind::Dim2DArray: return 2u; case ImageViewKind::Dim2DArray:
case ImageViewKind::Dim2DMsaa:
case ImageViewKind::Dim2DMsaaArray: return 2u;
case ImageViewKind::Dim3D: return 3u; case ImageViewKind::Dim3D: return 3u;
default: return 0u; default: return 0u;
} }
@@ -663,8 +668,7 @@ uint32_t LoadSampledImageDescriptor(EmitterState& state, const IR::MemoryInfo& m
uint32_t LoadSamplerDescriptor(EmitterState& state, uint32_t sampler, uint32_t use_pc) { uint32_t LoadSamplerDescriptor(EmitterState& state, uint32_t sampler, uint32_t use_pc) {
(void)use_pc; (void)use_pc;
const auto binding = const auto binding = ResourceForDescriptor(state, IR::DescriptorBindingKind::Samplers, sampler);
ResourceForDescriptor(state, IR::DescriptorBindingKind::Samplers, sampler);
const auto pointer = DescriptorElementPointer( const auto pointer = DescriptorElementPointer(
state, state.ptr_uniform_sampler, state.sampler_variable, binding.array_index, state, state.ptr_uniform_sampler, state.sampler_variable, binding.array_index,
IR::DescriptorBindingKind::Samplers, sampler, "sampler descriptor array was not emitted"); IR::DescriptorBindingKind::Samplers, sampler, "sampler descriptor array was not emitted");
@@ -24,7 +24,7 @@ uint32_t EmitExportVec4F32(EmitterState& state, const IR::Instruction& inst) {
const auto raw = EmitValueLoad(state, inst.src[pair_index]); const auto raw = EmitValueLoad(state, inst.src[pair_index]);
const auto unpacked = state.builder.AllocateId(); const auto unpacked = state.builder.AllocateId();
state.builder.AddFunction({OpExtInst, state.vec2_float_type, unpacked, state.builder.AddFunction({OpExtInst, state.vec2_float_type, unpacked,
state.glsl_std450, GlslUnpackHalf2x16, raw}); state.glsl_std450, GlslUnpackHalf2x16, raw});
for (uint32_t lane = 0; lane < 2u; lane++) { for (uint32_t lane = 0; lane < 2u; lane++) {
const auto component = pair_index * 2u + lane; const auto component = pair_index * 2u + lane;
if (((inst.export_info.en >> component) & 1u) == 0) { if (((inst.export_info.en >> component) & 1u) == 0) {
@@ -36,8 +36,8 @@ uint32_t EmitExportVec4F32(EmitterState& state, const IR::Instruction& inst) {
} }
} }
const auto vec = state.builder.AllocateId(); const auto vec = state.builder.AllocateId();
state.builder.AddFunction({OpCompositeConstruct, state.vec4_float_type, vec, state.builder.AddFunction({OpCompositeConstruct, state.vec4_float_type, vec, components[0],
components[0], components[1], components[2], components[3]}); components[1], components[2], components[3]});
return vec; return vec;
} }
@@ -50,7 +50,59 @@ uint32_t EmitExportVec4F32(EmitterState& state, const IR::Instruction& inst) {
return vec; return vec;
} }
uint32_t ApplyMrtExportMapping(EmitterState& state, const IR::Instruction& inst, uint32_t value) { uint32_t EmitExportComponentU32(EmitterState& state, const IR::Instruction& inst,
uint32_t component) {
const bool enabled = ((inst.export_info.en >> component) & 1u) != 0;
if (!enabled || component >= inst.src_count || component >= 4u) {
return ConstantU32(state, component == 3u ? 1u : 0u);
}
return EmitValueLoad(state, inst.src[component]);
}
uint32_t EmitExportVec4U32(EmitterState& state, const IR::Instruction& inst) {
uint32_t components[4] = {
ConstantU32(state, 0u),
ConstantU32(state, 0u),
ConstantU32(state, 0u),
ConstantU32(state, 1u),
};
if (inst.export_info.compr) {
for (uint32_t pair_index = 0; pair_index < 2u && pair_index < inst.src_count;
pair_index++) {
const auto raw = EmitValueLoad(state, inst.src[pair_index]);
for (uint32_t lane = 0; lane < 2u; lane++) {
const auto component = pair_index * 2u + lane;
if (((inst.export_info.en >> component) & 1u) == 0) {
continue;
}
components[component] = state.builder.AllocateId();
state.builder.AddFunction(
{OpBitFieldUExtract, state.uint_type, components[component], raw,
ConstantU32(state, lane * 16u), ConstantU32(state, 16u)});
}
}
} else {
for (uint32_t component = 0; component < 4u; component++) {
components[component] = EmitExportComponentU32(state, inst, component);
}
}
const auto vec = state.builder.AllocateId();
state.builder.AddFunction({OpCompositeConstruct, state.vec4_uint_type, vec, components[0],
components[1], components[2], components[3]});
return vec;
}
static bool MrtUsesUintOutput(const EmitterState& state, const IR::Instruction& inst) {
return inst.export_info.kind == IR::ExportTargetKind::Mrt &&
state.pixel_input_info != nullptr &&
inst.export_info.index < std::size(state.pixel_input_info->target_output_mode) &&
state.pixel_input_info->target_output_mode[inst.export_info.index] == 7u;
}
uint32_t ApplyMrtExportMapping(EmitterState& state, const IR::Instruction& inst, uint32_t value,
uint32_t vector_type) {
if (inst.export_info.kind != IR::ExportTargetKind::Mrt || state.pixel_input_info == nullptr || if (inst.export_info.kind != IR::ExportTargetKind::Mrt || state.pixel_input_info == nullptr ||
inst.export_info.index >= state.pixel_input_info->target_export_mapping.size()) { inst.export_info.index >= state.pixel_input_info->target_export_mapping.size()) {
return value; return value;
@@ -62,8 +114,8 @@ uint32_t ApplyMrtExportMapping(EmitterState& state, const IR::Instruction& inst,
} }
const auto mapped = state.builder.AllocateId(); const auto mapped = state.builder.AllocateId();
state.builder.AddFunction({OpVectorShuffle, state.vec4_float_type, mapped, value, value, state.builder.AddFunction({OpVectorShuffle, vector_type, mapped, value, value, mapping.Map(0),
mapping.Map(0), mapping.Map(1), mapping.Map(2), mapping.Map(3)}); mapping.Map(1), mapping.Map(2), mapping.Map(3)});
return mapped; return mapped;
} }
@@ -89,7 +141,7 @@ void EmitMrtZExport(EmitterState& state, const IR::Instruction& inst) {
const auto ptr = state.builder.AllocateId(); const auto ptr = state.builder.AllocateId();
state.builder.AddFunction({OpBitcast, state.int_type, mask, raw}); state.builder.AddFunction({OpBitcast, state.int_type, mask, raw});
state.builder.AddFunction({OpAccessChain, state.ptr_output_int, ptr, state.builder.AddFunction({OpAccessChain, state.ptr_output_int, ptr,
state.sample_mask_variable, ConstantU32(state, 0)}); state.sample_mask_variable, ConstantU32(state, 0)});
state.builder.AddFunction({OpStore, ptr, mask}); state.builder.AddFunction({OpStore, ptr, mask});
} }
} }
@@ -114,11 +166,15 @@ void EmitExport(EmitterState& state, const IR::Instruction& inst) {
return; return;
} }
const auto value = ApplyMrtExportMapping(state, inst, EmitExportVec4F32(state, inst)); const auto uint_output = MrtUsesUintOutput(state, inst);
const auto vector_type = uint_output ? state.vec4_uint_type : state.vec4_float_type;
const auto value = ApplyMrtExportMapping(
state, inst, uint_output ? EmitExportVec4U32(state, inst) : EmitExportVec4F32(state, inst),
vector_type);
if (inst.export_info.kind == IR::ExportTargetKind::Position) { if (inst.export_info.kind == IR::ExportTargetKind::Position) {
const auto pointer = state.builder.AllocateId(); const auto pointer = state.builder.AllocateId();
state.builder.AddFunction({OpAccessChain, state.ptr_output_vec4_float, pointer, variable, state.builder.AddFunction(
ConstantU32(state, 0)}); {OpAccessChain, state.ptr_output_vec4_float, pointer, variable, ConstantU32(state, 0)});
state.builder.AddFunction({OpStore, pointer, value}); state.builder.AddFunction({OpStore, pointer, value});
return; return;
} }
@@ -102,7 +102,7 @@ uint32_t EmitWqmLaneU32(EmitterState& state, uint32_t src) {
state.builder.AddFunction( state.builder.AddFunction(
{OpINotEqual, state.bool_type, non_zero, masked, ConstantU32(state, 0)}); {OpINotEqual, state.bool_type, non_zero, masked, ConstantU32(state, 0)});
state.builder.AddFunction({OpSelect, state.uint_type, expanded, non_zero, state.builder.AddFunction({OpSelect, state.uint_type, expanded, non_zero,
ConstantU32(state, mask), ConstantU32(state, 0)}); ConstantU32(state, mask), ConstantU32(state, 0)});
state.builder.AddFunction({OpBitwiseOr, state.uint_type, combined, ret, expanded}); state.builder.AddFunction({OpBitwiseOr, state.uint_type, combined, ret, expanded});
ret = combined; ret = combined;
} }
@@ -122,8 +122,8 @@ void EmitWqmB64(EmitterState& state, const IR::Instruction& inst) {
} }
const auto ballot = state.builder.AllocateId(); const auto ballot = state.builder.AllocateId();
state.builder.AddFunction({OpGroupNonUniformBallot, state.vec4_uint_type, ballot, state.builder.AddFunction({OpGroupNonUniformBallot, state.vec4_uint_type, ballot,
ConstantU32(state, ScopeSubgroup), ConstantU32(state, ScopeSubgroup),
EmitLaneMaskOperandActiveBool(state, inst.src[0])}); EmitLaneMaskOperandActiveBool(state, inst.src[0])});
const auto low = state.builder.AllocateId(); const auto low = state.builder.AllocateId();
const auto high = state.builder.AllocateId(); const auto high = state.builder.AllocateId();
state.builder.AddFunction({OpCompositeExtract, state.uint_type, low, ballot, 0}); state.builder.AddFunction({OpCompositeExtract, state.uint_type, low, ballot, 0});
@@ -150,8 +150,8 @@ void EmitWqmB64(EmitterState& state, const IR::Instruction& inst) {
EmitPerInvocationMask(state, inst.dst, active); EmitPerInvocationMask(state, inst.dst, active);
} else { } else {
const auto result = state.builder.AllocateId(); const auto result = state.builder.AllocateId();
state.builder.AddFunction({OpSelect, state.uint_type, result, active, state.builder.AddFunction({OpSelect, state.uint_type, result, active, ConstantU32(state, 1),
ConstantU32(state, 1), ConstantU32(state, 0)}); ConstantU32(state, 0)});
EmitStoreU32(state, inst.dst, result); EmitStoreU32(state, inst.dst, result);
EmitStoreU32(state, OffsetRegisterOperand(inst.dst, 1), ConstantU32(state, 0)); EmitStoreU32(state, OffsetRegisterOperand(inst.dst, 1), ConstantU32(state, 0));
} }
@@ -205,8 +205,7 @@ void EmitSaveexecB32(EmitterState& state, const IR::Instruction& inst) {
const auto cond = state.builder.AllocateId(); const auto cond = state.builder.AllocateId();
const auto scc = state.builder.AllocateId(); const auto scc = state.builder.AllocateId();
state.builder.AddFunction( state.builder.AddFunction({OpINotEqual, state.bool_type, cond, new_low, ConstantU32(state, 0)});
{OpINotEqual, state.bool_type, cond, new_low, ConstantU32(state, 0)});
state.builder.AddFunction( state.builder.AddFunction(
{OpSelect, state.uint_type, scc, cond, ConstantU32(state, 1), ConstantU32(state, 0)}); {OpSelect, state.uint_type, scc, cond, ConstantU32(state, 1), ConstantU32(state, 0)});
EmitStoreU32(state, SccOperand(), scc); EmitStoreU32(state, SccOperand(), scc);
@@ -269,18 +268,19 @@ void EmitReadFirstLaneU32(EmitterState& state, const IR::Instruction& inst) {
const auto first_lane = state.builder.AllocateId(); const auto first_lane = state.builder.AllocateId();
const auto first_value = state.builder.AllocateId(); const auto first_value = state.builder.AllocateId();
state.builder.AddFunction({OpGroupNonUniformBallot, state.vec4_uint_type, ballot, state.builder.AddFunction({OpGroupNonUniformBallot, state.vec4_uint_type, ballot,
ConstantU32(state, ScopeSubgroup), active}); ConstantU32(state, ScopeSubgroup), active});
state.builder.AddFunction({OpGroupNonUniformBallotFindLSB, state.uint_type, first_lane, state.builder.AddFunction({OpGroupNonUniformBallotFindLSB, state.uint_type, first_lane,
ConstantU32(state, ScopeSubgroup), ballot}); ConstantU32(state, ScopeSubgroup), ballot});
state.builder.AddFunction({OpGroupNonUniformShuffle, state.uint_type, first_value, state.builder.AddFunction({OpGroupNonUniformShuffle, state.uint_type, first_value,
ConstantU32(state, ScopeSubgroup), src, first_lane}); ConstantU32(state, ScopeSubgroup), src, first_lane});
EmitStoreU32(state, inst.dst, first_value); EmitStoreU32(state, inst.dst, first_value);
} }
uint32_t EmitLaneIndex(EmitterState& state, const IR::Operand& operand) { uint32_t EmitLaneIndex(EmitterState& state, const IR::Operand& operand) {
const auto lane = state.builder.AllocateId(); const auto lane = state.builder.AllocateId();
const auto mask = state.wave_size == 32u ? 31u : 63u;
state.builder.AddFunction({OpBitwiseAnd, state.uint_type, lane, EmitValueLoad(state, operand), state.builder.AddFunction({OpBitwiseAnd, state.uint_type, lane, EmitValueLoad(state, operand),
ConstantU32(state, 63)}); ConstantU32(state, mask)});
return lane; return lane;
} }
@@ -289,7 +289,7 @@ void EmitReadLaneU32(EmitterState& state, const IR::Instruction& inst) {
const auto lane = EmitLaneIndex(state, inst.src[1]); const auto lane = EmitLaneIndex(state, inst.src[1]);
const auto value = state.builder.AllocateId(); const auto value = state.builder.AllocateId();
state.builder.AddFunction({OpGroupNonUniformShuffle, state.uint_type, value, state.builder.AddFunction({OpGroupNonUniformShuffle, state.uint_type, value,
ConstantU32(state, ScopeSubgroup), src, lane}); ConstantU32(state, ScopeSubgroup), src, lane});
EmitStoreU32(state, inst.dst, value); EmitStoreU32(state, inst.dst, value);
} }
@@ -336,10 +336,8 @@ void EmitPermlaneB32(EmitterState& state, const IR::Instruction& inst, bool x16)
state.builder.AddFunction( state.builder.AddFunction(
{OpBitwiseXor, state.uint_type, row_value, row, ConstantU32(state, 16)}); {OpBitwiseXor, state.uint_type, row_value, row, ConstantU32(state, 16)});
} }
state.builder.AddFunction( state.builder.AddFunction({OpBitwiseAnd, state.uint_type, lane, subid, ConstantU32(state, 15)});
{OpBitwiseAnd, state.uint_type, lane, subid, ConstantU32(state, 15)}); state.builder.AddFunction({OpBitwiseAnd, state.uint_type, lane8, lane, ConstantU32(state, 7)});
state.builder.AddFunction(
{OpBitwiseAnd, state.uint_type, lane8, lane, ConstantU32(state, 7)});
state.builder.AddFunction( state.builder.AddFunction(
{OpShiftLeftLogical, state.uint_type, shift, lane8, ConstantU32(state, 2)}); {OpShiftLeftLogical, state.uint_type, shift, lane8, ConstantU32(state, 2)});
state.builder.AddFunction( state.builder.AddFunction(
@@ -350,7 +348,7 @@ void EmitPermlaneB32(EmitterState& state, const IR::Instruction& inst, bool x16)
{OpBitwiseAnd, state.uint_type, index1, index0, ConstantU32(state, 15)}); {OpBitwiseAnd, state.uint_type, index1, index0, ConstantU32(state, 15)});
state.builder.AddFunction({OpBitwiseOr, state.uint_type, target, row_value, index1}); state.builder.AddFunction({OpBitwiseOr, state.uint_type, target, row_value, index1});
state.builder.AddFunction({OpGroupNonUniformShuffle, state.uint_type, shuffled, state.builder.AddFunction({OpGroupNonUniformShuffle, state.uint_type, shuffled,
ConstantU32(state, ScopeSubgroup), value, target}); ConstantU32(state, ScopeSubgroup), value, target});
uint32_t ret = shuffled; uint32_t ret = shuffled;
if (!inst.dst.op_sel) { if (!inst.dst.op_sel) {
const auto source_active = EmitLaneIndexActiveBool(state, target); const auto source_active = EmitLaneIndexActiveBool(state, target);
@@ -375,7 +373,7 @@ void EmitBarrier(EmitterState& state, const IR::Instruction& inst) {
(void)inst; (void)inst;
const auto semantics = MemorySemanticsAcquireRelease | MemorySemanticsWorkgroupMemory; const auto semantics = MemorySemanticsAcquireRelease | MemorySemanticsWorkgroupMemory;
state.builder.AddFunction({OpControlBarrier, ConstantU32(state, ScopeWorkgroup), state.builder.AddFunction({OpControlBarrier, ConstantU32(state, ScopeWorkgroup),
ConstantU32(state, ScopeWorkgroup), ConstantU32(state, semantics)}); ConstantU32(state, ScopeWorkgroup), ConstantU32(state, semantics)});
} }
} // namespace Libs::Graphics::ShaderRecompiler::Spirv::Emitter } // namespace Libs::Graphics::ShaderRecompiler::Spirv::Emitter
@@ -35,9 +35,9 @@ uint32_t ConstantImageGatherHorizontalOffsets(EmitterState& state, ImageViewKind
uint32_t LoadStorageImageDescriptorAtIndex(EmitterState& state, uint32_t resource, uint32_t LoadStorageImageDescriptorAtIndex(EmitterState& state, uint32_t resource,
uint32_t array_index, bool uint_image, uint32_t array_index, bool uint_image,
ImageViewKind view) { ImageViewKind view) {
const auto kind = StorageBindingKind(uint_image, view); const auto kind = StorageBindingKind(uint_image, view);
const auto& descriptors = state.storage_images[StorageImageIndex(uint_image, view)]; const auto& descriptors = state.storage_images[StorageImageIndex(uint_image, view)];
const auto pointer = const auto pointer =
DescriptorElementPointer(state, descriptors.pointer_type, descriptors.variable, array_index, DescriptorElementPointer(state, descriptors.pointer_type, descriptors.variable, array_index,
kind, resource, "storage image descriptor array was not emitted"); kind, resource, "storage image descriptor array was not emitted");
const auto image = state.builder.AllocateId(); const auto image = state.builder.AllocateId();
@@ -133,10 +133,18 @@ void EmitImageLoad(EmitterState& state, const IR::Instruction& inst) {
const bool integer = inst.memory.kind == IR::ResourceKind::ImageUint; const bool integer = inst.memory.kind == IR::ResourceKind::ImageUint;
const auto color = state.builder.AllocateId(); const auto color = state.builder.AllocateId();
state.builder.AddFunction({OpImageFetch, integer ? state.vec4_uint_type : state.vec4_float_type, const auto coord = EmitImageLoadCoordU32(state, inst, view);
color, image, EmitImageLoadCoordU32(state, inst, view), if (ImageSpirvMultisampled(view) != 0) {
ImageOperandsLodMask, const auto sample = EmitImageAddressValueLoad(state, inst, inst.src[0],
EmitImageMipLodU32(state, inst, inst.src[0], view)}); ImageViewCoordinateComponents(view));
state.builder.AddFunction({OpImageFetch,
integer ? state.vec4_uint_type : state.vec4_float_type, color,
image, coord, ImageOperandsSampleMask, sample});
} else {
state.builder.AddFunction(
{OpImageFetch, integer ? state.vec4_uint_type : state.vec4_float_type, color, image,
coord, ImageOperandsLodMask, EmitImageMipLodU32(state, inst, inst.src[0], view)});
}
const auto dmask = inst.memory.dmask != 0 ? inst.memory.dmask : 1u; const auto dmask = inst.memory.dmask != 0 ? inst.memory.dmask : 1u;
uint32_t dst_index = 0; uint32_t dst_index = 0;
@@ -158,8 +166,8 @@ void EmitImageLoad(EmitterState& state, const IR::Instruction& inst) {
void EmitImageStore(EmitterState& state, const IR::Instruction& inst) { void EmitImageStore(EmitterState& state, const IR::Instruction& inst) {
const auto uint_image = inst.memory.kind == IR::ResourceKind::StorageImageUint; const auto uint_image = inst.memory.kind == IR::ResourceKind::StorageImageUint;
const auto view = StorageImageViewKind(state, inst.memory, uint_image, inst.pc); const auto view = StorageImageViewKind(state, inst.memory, uint_image, inst.pc);
const auto binding = ResourceForDescriptor(state, StorageBindingKind(uint_image, view), const auto binding =
inst.memory.resource); ResourceForDescriptor(state, StorageBindingKind(uint_image, view), inst.memory.resource);
const auto image = LoadStorageImageDescriptorAtIndex(state, inst.memory.resource, const auto image = LoadStorageImageDescriptorAtIndex(state, inst.memory.resource,
binding.array_index, uint_image, view); binding.array_index, uint_image, view);
@@ -261,9 +269,9 @@ void EmitImageSample(EmitterState& state, const IR::Instruction& inst) {
} else if (integer) { } else if (integer) {
result_type = state.vec4_uint_type; result_type = state.vec4_uint_type;
} }
const auto explicit_lod = ImageSampleNeedsExplicitLod(state, inst); const auto explicit_lod = ImageSampleNeedsExplicitLod(state, inst);
const auto opcode = ImageSampleOpcode(state, inst); const auto opcode = ImageSampleOpcode(state, inst);
std::vector<uint32_t> words = {opcode, result_type, sample, sampled_image, base_coord}; std::vector<uint32_t> words = {opcode, result_type, sample, sampled_image, base_coord};
if (dref) { if (dref) {
words.push_back(EmitImageDrefF32(state, inst, layout)); words.push_back(EmitImageDrefF32(state, inst, layout));
} }
@@ -99,6 +99,7 @@ enum : uint32_t {
ImageOperandsGradMask = 0x00000004u, ImageOperandsGradMask = 0x00000004u,
ImageOperandsOffsetMask = 0x00000010u, ImageOperandsOffsetMask = 0x00000010u,
ImageOperandsConstOffsetsMask = 0x00000020u, ImageOperandsConstOffsetsMask = 0x00000020u,
ImageOperandsSampleMask = 0x00000040u,
}; };
enum : uint32_t { enum : uint32_t {
@@ -150,7 +151,6 @@ enum : uint32_t {
OpImageGather = 96, OpImageGather = 96,
OpImageDrefGather = 97, OpImageDrefGather = 97,
OpImageWrite = 99, OpImageWrite = 99,
OpImage = 100,
OpImageQuerySizeLod = 103, OpImageQuerySizeLod = 103,
OpImageQueryLod = 105, OpImageQueryLod = 105,
OpImageQueryLevels = 106, OpImageQueryLevels = 106,
@@ -382,7 +382,7 @@ struct EmitterState {
uint32_t ptr_workgroup_array = 0; uint32_t ptr_workgroup_array = 0;
uint32_t ptr_workgroup_uint = 0; uint32_t ptr_workgroup_uint = 0;
uint32_t lds_variable = 0; uint32_t lds_variable = 0;
std::array<SampledImageDescriptors, 10> sampled_images; std::array<SampledImageDescriptors, 14> sampled_images;
std::array<StorageImageDescriptors, 10> storage_images; std::array<StorageImageDescriptors, 10> storage_images;
uint32_t sampler_type = 0; uint32_t sampler_type = 0;
uint32_t sampler_array_type = 0; uint32_t sampler_array_type = 0;
@@ -453,17 +453,20 @@ enum class ImageViewKind {
Dim2D, Dim2D,
Dim2DArray, Dim2DArray,
Dim3D, Dim3D,
Dim2DMsaa,
Dim2DMsaaArray,
Count, Count,
}; };
constexpr uint32_t ImageViewKindCount = static_cast<uint32_t>(ImageViewKind::Count); constexpr uint32_t SampledImageViewKindCount = static_cast<uint32_t>(ImageViewKind::Count);
constexpr uint32_t StorageImageViewKindCount = static_cast<uint32_t>(ImageViewKind::Dim2DMsaa);
constexpr uint32_t SampledImageIndex(bool integer, ImageViewKind view) { constexpr uint32_t SampledImageIndex(bool integer, ImageViewKind view) {
return static_cast<uint32_t>(view) + (integer ? ImageViewKindCount : 0u); return static_cast<uint32_t>(view) + (integer ? SampledImageViewKindCount : 0u);
} }
constexpr uint32_t StorageImageIndex(bool integer, ImageViewKind view) { constexpr uint32_t StorageImageIndex(bool integer, ImageViewKind view) {
return static_cast<uint32_t>(view) + (integer ? ImageViewKindCount : 0u); return static_cast<uint32_t>(view) + (integer ? StorageImageViewKindCount : 0u);
} }
constexpr IR::DescriptorBindingKind SampledBindingKind(bool integer, ImageViewKind view) { constexpr IR::DescriptorBindingKind SampledBindingKind(bool integer, ImageViewKind view) {
@@ -474,6 +477,9 @@ constexpr IR::DescriptorBindingKind SampledBindingKind(bool integer, ImageViewKi
case ImageViewKind::Dim2D: return IR::DescriptorBindingKind::SampledUint2D; case ImageViewKind::Dim2D: return IR::DescriptorBindingKind::SampledUint2D;
case ImageViewKind::Dim2DArray: return IR::DescriptorBindingKind::SampledUint2DArray; case ImageViewKind::Dim2DArray: return IR::DescriptorBindingKind::SampledUint2DArray;
case ImageViewKind::Dim3D: return IR::DescriptorBindingKind::SampledUint3D; case ImageViewKind::Dim3D: return IR::DescriptorBindingKind::SampledUint3D;
case ImageViewKind::Dim2DMsaa: return IR::DescriptorBindingKind::SampledUint2DMsaa;
case ImageViewKind::Dim2DMsaaArray:
return IR::DescriptorBindingKind::SampledUint2DMsaaArray;
default: break; default: break;
} }
} }
@@ -483,6 +489,8 @@ constexpr IR::DescriptorBindingKind SampledBindingKind(bool integer, ImageViewKi
case ImageViewKind::Dim2D: return IR::DescriptorBindingKind::Sampled2D; case ImageViewKind::Dim2D: return IR::DescriptorBindingKind::Sampled2D;
case ImageViewKind::Dim2DArray: return IR::DescriptorBindingKind::Sampled2DArray; case ImageViewKind::Dim2DArray: return IR::DescriptorBindingKind::Sampled2DArray;
case ImageViewKind::Dim3D: return IR::DescriptorBindingKind::Sampled3D; case ImageViewKind::Dim3D: return IR::DescriptorBindingKind::Sampled3D;
case ImageViewKind::Dim2DMsaa: return IR::DescriptorBindingKind::Sampled2DMsaa;
case ImageViewKind::Dim2DMsaaArray: return IR::DescriptorBindingKind::Sampled2DMsaaArray;
default: break; default: break;
} }
return IR::DescriptorBindingKind::Count; return IR::DescriptorBindingKind::Count;
@@ -516,6 +524,8 @@ constexpr uint32_t ImageSpirvDimension(ImageViewKind view) {
case ImageViewKind::Dim1DArray: return Dim1D; case ImageViewKind::Dim1DArray: return Dim1D;
case ImageViewKind::Dim2D: case ImageViewKind::Dim2D:
case ImageViewKind::Dim2DArray: case ImageViewKind::Dim2DArray:
case ImageViewKind::Dim2DMsaa:
case ImageViewKind::Dim2DMsaaArray:
case ImageViewKind::Count: return Dim2D; case ImageViewKind::Count: return Dim2D;
case ImageViewKind::Dim3D: return Dim3D; case ImageViewKind::Dim3D: return Dim3D;
} }
@@ -523,7 +533,14 @@ constexpr uint32_t ImageSpirvDimension(ImageViewKind view) {
} }
constexpr uint32_t ImageSpirvArrayed(ImageViewKind view) { constexpr uint32_t ImageSpirvArrayed(ImageViewKind view) {
return view == ImageViewKind::Dim1DArray || view == ImageViewKind::Dim2DArray ? 1u : 0u; return view == ImageViewKind::Dim1DArray || view == ImageViewKind::Dim2DArray ||
view == ImageViewKind::Dim2DMsaaArray
? 1u
: 0u;
}
constexpr uint32_t ImageSpirvMultisampled(ImageViewKind view) {
return view == ImageViewKind::Dim2DMsaa || view == ImageViewKind::Dim2DMsaaArray ? 1u : 0u;
} }
struct AddCarryResult { struct AddCarryResult {
@@ -174,6 +174,12 @@ uint32_t VertexParameterInputPointerType(const EmitterState& state, VertexInputS
} }
} }
static bool MrtUsesUintOutput(const EmitterState& state, uint32_t index) {
return state.stage == ShaderType::Pixel && state.pixel_input_info != nullptr &&
index < std::size(state.pixel_input_info->target_output_mode) &&
state.pixel_input_info->target_output_mode[index] == 7u;
}
void AllocateInputVariables(EmitterState& state) { void AllocateInputVariables(EmitterState& state) {
for (auto& binding: state.inputs) { for (auto& binding: state.inputs) {
binding.variable_id = state.builder.AllocateId(); binding.variable_id = state.builder.AllocateId();
@@ -323,23 +329,39 @@ void AddDescriptorAnnotationsAndNames(EmitterState& state) {
Decorate(state.address_memory_variable, "address_memory", Decorate(state.address_memory_variable, "address_memory",
IR::DescriptorBindingKind::AddressMemory); IR::DescriptorBindingKind::AddressMemory);
} }
constexpr const char* SampledNames[] = { constexpr const char* SampledNames[] = {"sampled_1d",
"sampled_1d", "sampled_1d_array", "sampled_2d", "sampled_2d_array", "sampled_1d_array",
"sampled_3d", "sampled_uint_1d", "sampled_uint_1d_array", "sampled_2d",
"sampled_uint_2d", "sampled_uint_2d_array", "sampled_uint_3d"}; "sampled_2d_array",
"sampled_3d",
"sampled_2d_msaa",
"sampled_2d_msaa_array",
"sampled_uint_1d",
"sampled_uint_1d_array",
"sampled_uint_2d",
"sampled_uint_2d_array",
"sampled_uint_3d",
"sampled_uint_2d_msaa",
"sampled_uint_2d_msaa_array"};
for (uint32_t i = 0; i < state.sampled_images.size(); i++) { for (uint32_t i = 0; i < state.sampled_images.size(); i++) {
const auto view = static_cast<ImageViewKind>(i % ImageViewKindCount); const auto view = static_cast<ImageViewKind>(i % SampledImageViewKindCount);
Decorate(state.sampled_images[i].variable, SampledNames[i], Decorate(state.sampled_images[i].variable, SampledNames[i],
SampledBindingKind(i >= ImageViewKindCount, view)); SampledBindingKind(i >= SampledImageViewKindCount, view));
} }
constexpr const char* StorageNames[] = { constexpr const char* StorageNames[] = {"storage_1d",
"storage_1d", "storage_1d_array", "storage_2d", "storage_2d_array", "storage_1d_array",
"storage_3d", "storage_uint_1d", "storage_uint_1d_array", "storage_2d",
"storage_uint_2d", "storage_uint_2d_array", "storage_uint_3d"}; "storage_2d_array",
"storage_3d",
"storage_uint_1d",
"storage_uint_1d_array",
"storage_uint_2d",
"storage_uint_2d_array",
"storage_uint_3d"};
for (uint32_t i = 0; i < state.storage_images.size(); i++) { for (uint32_t i = 0; i < state.storage_images.size(); i++) {
const auto view = static_cast<ImageViewKind>(i % ImageViewKindCount); const auto view = static_cast<ImageViewKind>(i % StorageImageViewKindCount);
Decorate(state.storage_images[i].variable, StorageNames[i], Decorate(state.storage_images[i].variable, StorageNames[i],
StorageBindingKind(i >= ImageViewKindCount, view)); StorageBindingKind(i >= StorageImageViewKindCount, view));
} }
if (state.sampler_variable != 0) { if (state.sampler_variable != 0) {
Decorate(state.sampler_variable, "samplers", IR::DescriptorBindingKind::Samplers); Decorate(state.sampler_variable, "samplers", IR::DescriptorBindingKind::Samplers);
@@ -409,6 +431,7 @@ void EmitHeaderAndTypes(EmitterState& state) {
state.ptr_output_sample_mask_array = state.builder.AllocateId(); state.ptr_output_sample_mask_array = state.builder.AllocateId();
state.ptr_output_float = state.builder.AllocateId(); state.ptr_output_float = state.builder.AllocateId();
state.ptr_output_vec4_float = state.builder.AllocateId(); state.ptr_output_vec4_float = state.builder.AllocateId();
const auto ptr_output_vec4_uint = state.builder.AllocateId();
state.per_vertex_type = state.builder.AllocateId(); state.per_vertex_type = state.builder.AllocateId();
state.ptr_output_per_vertex = state.builder.AllocateId(); state.ptr_output_per_vertex = state.builder.AllocateId();
state.storage_runtime_array_type = state.builder.AllocateId(); state.storage_runtime_array_type = state.builder.AllocateId();
@@ -444,15 +467,15 @@ void EmitHeaderAndTypes(EmitterState& state) {
image.array_type = state.builder.AllocateId(); image.array_type = state.builder.AllocateId();
image.array_pointer_type = state.builder.AllocateId(); image.array_pointer_type = state.builder.AllocateId();
} }
state.sampler_type = state.builder.AllocateId(); state.sampler_type = state.builder.AllocateId();
state.sampler_array_type = state.builder.AllocateId(); state.sampler_array_type = state.builder.AllocateId();
state.ptr_uniform_sampler = state.builder.AllocateId(); state.ptr_uniform_sampler = state.builder.AllocateId();
state.ptr_uniform_sampler_array = state.builder.AllocateId(); state.ptr_uniform_sampler_array = state.builder.AllocateId();
state.ptr_image_uint = state.builder.AllocateId(); state.ptr_image_uint = state.builder.AllocateId();
state.func_type = state.builder.AllocateId(); state.func_type = state.builder.AllocateId();
state.main_func = state.builder.AllocateId(); state.main_func = state.builder.AllocateId();
state.entry_label = state.builder.AllocateId(); state.entry_label = state.builder.AllocateId();
state.glsl_std450 = state.builder.AllocateId(); state.glsl_std450 = state.builder.AllocateId();
state.builder.AddCapability({CapabilityShader}); state.builder.AddCapability({CapabilityShader});
state.builder.AddCapability({CapabilitySampled1D}); state.builder.AddCapability({CapabilitySampled1D});
@@ -462,7 +485,7 @@ void EmitHeaderAndTypes(EmitterState& state) {
state.builder.AddCapability({CapabilityImageGatherExtended}); state.builder.AddCapability({CapabilityImageGatherExtended});
} }
if (std::any_of(state.storage_images.begin(), if (std::any_of(state.storage_images.begin(),
state.storage_images.begin() + ImageViewKindCount, state.storage_images.begin() + StorageImageViewKindCount,
[](const auto& image) { return image.variable != 0; })) { [](const auto& image) { return image.variable != 0; })) {
state.builder.AddCapability({CapabilityStorageImageReadWithoutFormat}); state.builder.AddCapability({CapabilityStorageImageReadWithoutFormat});
state.builder.AddCapability({CapabilityStorageImageWriteWithoutFormat}); state.builder.AddCapability({CapabilityStorageImageWriteWithoutFormat});
@@ -605,6 +628,8 @@ void EmitHeaderAndTypes(EmitterState& state) {
{OpTypePointer, state.ptr_output_int, StorageClassOutput, state.int_type}); {OpTypePointer, state.ptr_output_int, StorageClassOutput, state.int_type});
state.builder.AddType( state.builder.AddType(
{OpTypePointer, state.ptr_output_vec4_float, StorageClassOutput, state.vec4_float_type}); {OpTypePointer, state.ptr_output_vec4_float, StorageClassOutput, state.vec4_float_type});
state.builder.AddType(
{OpTypePointer, ptr_output_vec4_uint, StorageClassOutput, state.vec4_uint_type});
if (state.per_vertex_variable != 0) { if (state.per_vertex_variable != 0) {
state.builder.AddType({OpTypeStruct, state.per_vertex_type, state.vec4_float_type}); state.builder.AddType({OpTypeStruct, state.per_vertex_type, state.vec4_float_type});
state.builder.AddType({OpTypePointer, state.ptr_output_per_vertex, StorageClassOutput, state.builder.AddType({OpTypePointer, state.ptr_output_per_vertex, StorageClassOutput,
@@ -615,8 +640,12 @@ void EmitHeaderAndTypes(EmitterState& state) {
for (const auto& binding: state.outputs) { for (const auto& binding: state.outputs) {
if (binding.kind == IR::StageOutputKind::Parameter || if (binding.kind == IR::StageOutputKind::Parameter ||
binding.kind == IR::StageOutputKind::Mrt) { binding.kind == IR::StageOutputKind::Mrt) {
const auto pointer_type =
binding.kind == IR::StageOutputKind::Mrt && MrtUsesUintOutput(state, binding.index)
? ptr_output_vec4_uint
: state.ptr_output_vec4_float;
state.builder.AddType( state.builder.AddType(
{OpVariable, state.ptr_output_vec4_float, binding.variable_id, StorageClassOutput}); {OpVariable, pointer_type, binding.variable_id, StorageClassOutput});
} }
} }
if (state.depth_variable != 0) { if (state.depth_variable != 0) {
@@ -700,11 +729,11 @@ void EmitHeaderAndTypes(EmitterState& state) {
} }
for (uint32_t i = 0; i < state.sampled_images.size(); i++) { for (uint32_t i = 0; i < state.sampled_images.size(); i++) {
auto& image = state.sampled_images[i]; auto& image = state.sampled_images[i];
const auto view = static_cast<ImageViewKind>(i % ImageViewKindCount); const auto view = static_cast<ImageViewKind>(i % SampledImageViewKindCount);
const bool integer = i >= ImageViewKindCount; const bool integer = i >= SampledImageViewKindCount;
const auto component = integer ? state.uint_type : state.float_type; const auto component = integer ? state.uint_type : state.float_type;
state.builder.AddType({OpTypeImage, image.image_type, component, state.builder.AddType({OpTypeImage, image.image_type, component, ImageSpirvDimension(view),
ImageSpirvDimension(view), 0, ImageSpirvArrayed(view), 0, 1, 0, ImageSpirvArrayed(view), ImageSpirvMultisampled(view), 1,
ImageFormatUnknown}); ImageFormatUnknown});
state.builder.AddType({OpTypeSampledImage, image.sampled_image_type, image.image_type}); state.builder.AddType({OpTypeSampledImage, image.sampled_image_type, image.image_type});
state.builder.AddType( state.builder.AddType(
@@ -733,13 +762,12 @@ void EmitHeaderAndTypes(EmitterState& state) {
} }
for (uint32_t i = 0; i < state.storage_images.size(); i++) { for (uint32_t i = 0; i < state.storage_images.size(); i++) {
auto& image = state.storage_images[i]; auto& image = state.storage_images[i];
const auto view = static_cast<ImageViewKind>(i % ImageViewKindCount); const auto view = static_cast<ImageViewKind>(i % StorageImageViewKindCount);
const bool integer = i >= ImageViewKindCount; const bool integer = i >= StorageImageViewKindCount;
const auto component = integer ? state.uint_type : state.float_type; const auto component = integer ? state.uint_type : state.float_type;
const auto format = integer ? ImageFormatR32ui : ImageFormatUnknown; const auto format = integer ? ImageFormatR32ui : ImageFormatUnknown;
state.builder.AddType({OpTypeImage, image.image_type, component, state.builder.AddType({OpTypeImage, image.image_type, component, ImageSpirvDimension(view),
ImageSpirvDimension(view), 0, ImageSpirvArrayed(view), 0, 2, 0, ImageSpirvArrayed(view), 0, 2, format});
format});
state.builder.AddType( state.builder.AddType(
{OpTypePointer, image.pointer_type, StorageClassUniformConstant, image.image_type}); {OpTypePointer, image.pointer_type, StorageClassUniformConstant, image.image_type});
if (image.variable != 0) { if (image.variable != 0) {
@@ -786,15 +814,15 @@ void AllocateDescriptorVariables(EmitterState& state) {
state.flattened_srt_variable = state.builder.AllocateId(); state.flattened_srt_variable = state.builder.AllocateId();
} }
for (uint32_t i = 0; i < state.sampled_images.size(); i++) { for (uint32_t i = 0; i < state.sampled_images.size(); i++) {
const auto view = static_cast<ImageViewKind>(i % ImageViewKindCount); const auto view = static_cast<ImageViewKind>(i % SampledImageViewKindCount);
if (DescriptorBinding(state, SampledBindingKind(i >= ImageViewKindCount, view)) != if (DescriptorBinding(state, SampledBindingKind(i >= SampledImageViewKindCount, view)) !=
nullptr) { nullptr) {
state.sampled_images[i].variable = state.builder.AllocateId(); state.sampled_images[i].variable = state.builder.AllocateId();
} }
} }
for (uint32_t i = 0; i < state.storage_images.size(); i++) { for (uint32_t i = 0; i < state.storage_images.size(); i++) {
const auto view = static_cast<ImageViewKind>(i % ImageViewKindCount); const auto view = static_cast<ImageViewKind>(i % StorageImageViewKindCount);
if (DescriptorBinding(state, StorageBindingKind(i >= ImageViewKindCount, view)) != if (DescriptorBinding(state, StorageBindingKind(i >= StorageImageViewKindCount, view)) !=
nullptr) { nullptr) {
state.storage_images[i].variable = state.builder.AllocateId(); state.storage_images[i].variable = state.builder.AllocateId();
} }
@@ -13,16 +13,30 @@ namespace {
constexpr uint32_t MaxPushConstantBytes = 128; constexpr uint32_t MaxPushConstantBytes = 128;
constexpr std::array ImageBindingKinds = { constexpr std::array ImageBindingKinds = {
DescriptorBindingKind::Sampled1D, DescriptorBindingKind::Sampled1DArray, DescriptorBindingKind::Sampled1D,
DescriptorBindingKind::Sampled2D, DescriptorBindingKind::Sampled2DArray, DescriptorBindingKind::Sampled1DArray,
DescriptorBindingKind::Sampled3D, DescriptorBindingKind::SampledUint1D, DescriptorBindingKind::Sampled2D,
DescriptorBindingKind::SampledUint1DArray, DescriptorBindingKind::SampledUint2D, DescriptorBindingKind::Sampled2DArray,
DescriptorBindingKind::SampledUint2DArray, DescriptorBindingKind::SampledUint3D, DescriptorBindingKind::Sampled2DMsaa,
DescriptorBindingKind::Storage1D, DescriptorBindingKind::Storage1DArray, DescriptorBindingKind::Sampled2DMsaaArray,
DescriptorBindingKind::Storage2D, DescriptorBindingKind::Storage2DArray, DescriptorBindingKind::Sampled3D,
DescriptorBindingKind::Storage3D, DescriptorBindingKind::StorageUint1D, DescriptorBindingKind::SampledUint1D,
DescriptorBindingKind::StorageUint1DArray, DescriptorBindingKind::StorageUint2D, DescriptorBindingKind::SampledUint1DArray,
DescriptorBindingKind::StorageUint2DArray, DescriptorBindingKind::StorageUint3D, DescriptorBindingKind::SampledUint2D,
DescriptorBindingKind::SampledUint2DArray,
DescriptorBindingKind::SampledUint2DMsaa,
DescriptorBindingKind::SampledUint2DMsaaArray,
DescriptorBindingKind::SampledUint3D,
DescriptorBindingKind::Storage1D,
DescriptorBindingKind::Storage1DArray,
DescriptorBindingKind::Storage2D,
DescriptorBindingKind::Storage2DArray,
DescriptorBindingKind::Storage3D,
DescriptorBindingKind::StorageUint1D,
DescriptorBindingKind::StorageUint1DArray,
DescriptorBindingKind::StorageUint2D,
DescriptorBindingKind::StorageUint2DArray,
DescriptorBindingKind::StorageUint3D,
}; };
bool ImageBinding(const ImageResource& image, DescriptorBindingKind& result) { bool ImageBinding(const ImageResource& image, DescriptorBindingKind& result) {
@@ -36,6 +50,8 @@ bool ImageBinding(const ImageResource& image, DescriptorBindingKind& result) {
case Dimension::Dim1DArray: result = Kind::Sampled1DArray; return true; case Dimension::Dim1DArray: result = Kind::Sampled1DArray; return true;
case Dimension::Dim2D: result = Kind::Sampled2D; return true; case Dimension::Dim2D: result = Kind::Sampled2D; return true;
case Dimension::Dim2DArray: result = Kind::Sampled2DArray; return true; case Dimension::Dim2DArray: result = Kind::Sampled2DArray; return true;
case Dimension::Dim2DMsaa: result = Kind::Sampled2DMsaa; return true;
case Dimension::Dim2DMsaaArray: result = Kind::Sampled2DMsaaArray; return true;
case Dimension::Dim3D: result = Kind::Sampled3D; return true; case Dimension::Dim3D: result = Kind::Sampled3D; return true;
default: return false; default: return false;
} }
@@ -45,6 +61,8 @@ bool ImageBinding(const ImageResource& image, DescriptorBindingKind& result) {
case Dimension::Dim1DArray: result = Kind::SampledUint1DArray; return true; case Dimension::Dim1DArray: result = Kind::SampledUint1DArray; return true;
case Dimension::Dim2D: result = Kind::SampledUint2D; return true; case Dimension::Dim2D: result = Kind::SampledUint2D; return true;
case Dimension::Dim2DArray: result = Kind::SampledUint2DArray; return true; case Dimension::Dim2DArray: result = Kind::SampledUint2DArray; return true;
case Dimension::Dim2DMsaa: result = Kind::SampledUint2DMsaa; return true;
case Dimension::Dim2DMsaaArray: result = Kind::SampledUint2DMsaaArray; return true;
case Dimension::Dim3D: result = Kind::SampledUint3D; return true; case Dimension::Dim3D: result = Kind::SampledUint3D; return true;
default: return false; default: return false;
} }
@@ -71,7 +89,7 @@ bool ImageBinding(const ImageResource& image, DescriptorBindingKind& result) {
} }
bool CollectValue(const ScalarProvenance& provenance, uint32_t id, std::vector<uint8_t>& visited, bool CollectValue(const ScalarProvenance& provenance, uint32_t id, std::vector<uint8_t>& visited,
std::set<uint32_t>& registers) { std::set<uint32_t>& registers) {
if (id <= ScalarProvenance::Unknown) { if (id <= ScalarProvenance::Unknown) {
return true; return true;
} }
@@ -104,7 +122,7 @@ bool CollectValue(const ScalarProvenance& provenance, uint32_t id, std::vector<u
} }
bool CollectSource(const Program& program, uint32_t source, bool allow_unknown, bool CollectSource(const Program& program, uint32_t source, bool allow_unknown,
std::vector<uint8_t>& visited, std::set<uint32_t>& registers) { std::vector<uint8_t>& visited, std::set<uint32_t>& registers) {
if (allow_unknown && source == ScalarProvenance::Unknown) { if (allow_unknown && source == ScalarProvenance::Unknown) {
return true; return true;
} }
@@ -165,8 +183,7 @@ bool CollectUserData(const Program& program, std::vector<uint32_t>& result) {
return false; return false;
} }
for (uint32_t i = 0; i < inst.src_count; i++) { for (uint32_t i = 0; i < inst.src_count; i++) {
if (!CollectValue(program.provenance, inst.scalar_sources[i], visited, if (!CollectValue(program.provenance, inst.scalar_sources[i], visited, registers)) {
registers)) {
return false; return false;
} }
} }
@@ -199,7 +216,7 @@ bool AllocateBindings(Program& program, const BindingLayoutOptions& options, std
if (!program.shader_info_complete || program.binding_layout_complete) { if (!program.shader_info_complete || program.binding_layout_complete) {
if (error != nullptr) { if (error != nullptr) {
*error = !program.shader_info_complete ? "shader info is not ready" *error = !program.shader_info_complete ? "shader info is not ready"
: "binding layout already allocated"; : "binding layout already allocated";
} }
return false; return false;
} }
@@ -0,0 +1,323 @@
#include "graphics/shader/recompiler/ir/ReadLaneElimination.h"
#include "graphics/shader/recompiler/ir/SrtWalker.h"
#include <algorithm>
#include <iterator>
#include <map>
#include <set>
#include <utility>
namespace Libs::Graphics::ShaderRecompiler::IR {
namespace {
constexpr uint32_t FirstTemporaryScalarRegister = 128;
struct LaneKey {
uint32_t reg = 0;
uint32_t lane = 0;
auto operator<=>(const LaneKey&) const = default;
};
using LaneSet = std::set<LaneKey>;
bool PairDwordOpcode(Opcode op) {
switch (op) {
case Opcode::MoveU64:
case Opcode::WqmB64:
case Opcode::SaveexecB64:
case Opcode::BitwiseAndU64:
case Opcode::BitwiseAndNotU64:
case Opcode::BitwiseOrU64:
case Opcode::BitwiseOrNotU64:
case Opcode::BitwiseXorU64:
case Opcode::BitwiseNandU64:
case Opcode::BitwiseNorU64:
case Opcode::BitwiseXnorU64:
case Opcode::BitwiseNotU64:
case Opcode::BitFieldMaskU64:
case Opcode::BitFieldExtractU64:
case Opcode::BitReplicateB64B32:
case Opcode::ShiftLeftLogicalU64:
case Opcode::ShiftRightLogicalU64:
case Opcode::SelectU64: return true;
default: return false;
}
}
bool ResolveLane(const Program& program, const Instruction& inst, uint32_t source_index,
uint32_t& lane) {
if (source_index >= inst.src_count || (program.wave_size != 32 && program.wave_size != 64)) {
return false;
}
const auto& selector = inst.src[source_index];
if (selector.kind == OperandKind::ImmediateU32) {
lane = selector.imm % program.wave_size;
return true;
}
uint32_t folded = 0;
if (!FoldScalarConstant(program.provenance, inst.scalar_sources[source_index], folded)) {
return false;
}
lane = folded % program.wave_size;
return true;
}
bool UniformWriteSource(const Instruction& inst) {
if (inst.src_count == 0) {
return false;
}
const auto& source = inst.src[0];
if (source.kind == OperandKind::ImmediateU32 || source.kind == OperandKind::PcRelativeU32) {
return true;
}
return source.kind == OperandKind::Register &&
(source.reg.file == RegisterFile::Scalar || source.reg.file == RegisterFile::Scc ||
source.reg.file == RegisterFile::M0);
}
bool WriteLaneKey(const Program& program, const Instruction& inst, LaneKey& key) {
if (inst.op != Opcode::WriteLaneU32 || inst.dst.kind != OperandKind::Register ||
inst.dst.reg.file != RegisterFile::Vector || !UniformWriteSource(inst)) {
return false;
}
uint32_t lane = 0;
if (!ResolveLane(program, inst, 1, lane)) {
return false;
}
key = {inst.dst.reg.index, lane};
return true;
}
bool ReadLaneKey(const Program& program, const Instruction& inst, LaneKey& key) {
if (inst.op != Opcode::ReadLaneU32 || inst.src_count < 2 ||
inst.src[0].kind != OperandKind::Register || inst.src[0].reg.file != RegisterFile::Vector) {
return false;
}
uint32_t lane = 0;
if (!ResolveLane(program, inst, 1, lane)) {
return false;
}
key = {inst.src[0].reg.index, lane};
return true;
}
void InvalidateRegister(LaneSet& valid, uint32_t reg) {
const auto first = valid.lower_bound({reg, 0});
const auto last = valid.lower_bound({reg + 1u, 0});
valid.erase(first, last);
}
void ApplyInstruction(const Program& program, const Instruction& inst, LaneSet& valid) {
if (inst.op == Opcode::WriteLaneU32 && inst.dst.kind == OperandKind::Register &&
inst.dst.reg.file == RegisterFile::Vector) {
LaneKey key;
if (WriteLaneKey(program, inst, key)) {
valid.insert(key);
return;
}
uint32_t lane = 0;
if (ResolveLane(program, inst, 1, lane)) {
valid.erase({inst.dst.reg.index, lane});
} else {
InvalidateRegister(valid, inst.dst.reg.index);
}
return;
}
if (inst.op == Opcode::MoveRelDestU32 && inst.dst.kind == OperandKind::Register &&
inst.dst.reg.file == RegisterFile::Vector) {
valid.clear();
return;
}
if (inst.dst.kind == OperandKind::Register && inst.dst.reg.file == RegisterFile::Vector) {
uint32_t dwords = std::max(inst.memory.data_dwords, 1u);
if (PairDwordOpcode(inst.op) || inst.op == Opcode::UMadU64U32) {
dwords = std::max(dwords, 2u);
}
for (uint32_t i = 0; i < dwords && inst.dst.reg.index <= UINT32_MAX - i; i++) {
InvalidateRegister(valid, inst.dst.reg.index + i);
}
}
if (inst.dst2.kind == OperandKind::Register && inst.dst2.reg.file == RegisterFile::Vector) {
InvalidateRegister(valid, inst.dst2.reg.index);
}
}
LaneSet TransferBlock(const Program& program, const BasicBlock& block, LaneSet state) {
for (const auto& inst: block.instructions) {
ApplyInstruction(program, inst, state);
}
return state;
}
LaneSet Intersect(const LaneSet& left, const LaneSet& right) {
LaneSet result;
std::set_intersection(left.begin(), left.end(), right.begin(), right.end(),
std::inserter(result, result.end()));
return result;
}
uint32_t NextTemporaryScalarRegister(const Program& program) {
uint32_t next = FirstTemporaryScalarRegister;
const auto consider = [&next](const Operand& operand) {
if (operand.kind == OperandKind::Register && operand.reg.file == RegisterFile::Scalar &&
operand.reg.index >= next && operand.reg.index != UINT32_MAX) {
next = operand.reg.index + 1u;
}
};
for (const auto& block: program.blocks) {
for (const auto& inst: block.instructions) {
consider(inst.dst);
consider(inst.dst2);
for (uint32_t i = 0; i < inst.src_count; i++) {
consider(inst.src[i]);
}
}
}
return next;
}
Operand ScalarRegisterOperand(uint32_t reg) {
Operand operand;
operand.kind = OperandKind::Register;
operand.reg.file = RegisterFile::Scalar;
operand.reg.index = reg;
return operand;
}
Instruction ShadowWrite(const Instruction& write, uint32_t temporary) {
Instruction shadow;
shadow.pc = write.pc;
shadow.op = Opcode::MoveU32;
shadow.dst = ScalarRegisterOperand(temporary);
shadow.src[0] = write.src[0];
shadow.src_count = 1;
return shadow;
}
Instruction ShadowRead(const Instruction& read, uint32_t temporary) {
Instruction rewritten;
rewritten.pc = read.pc;
rewritten.op = Opcode::MoveU32;
rewritten.dst = read.dst;
rewritten.src[0] = ScalarRegisterOperand(temporary);
rewritten.src_count = 1;
return rewritten;
}
} // namespace
ReadLaneEliminationStats EliminateReadLane(Program& program) {
ReadLaneEliminationStats stats;
if (program.blocks.empty() || (program.wave_size != 32 && program.wave_size != 64)) {
return stats;
}
LaneSet universe;
for (const auto& block: program.blocks) {
for (const auto& inst: block.instructions) {
LaneKey key;
if (WriteLaneKey(program, inst, key)) {
universe.insert(key);
}
}
}
if (universe.empty()) {
return stats;
}
const size_t block_count = program.blocks.size();
std::vector<LaneSet> entry(block_count, universe);
std::vector<LaneSet> exit(block_count, universe);
entry[0].clear();
for (size_t block = 0; block < block_count; block++) {
exit[block] = TransferBlock(program, program.blocks[block], entry[block]);
}
bool changed = true;
while (changed) {
changed = false;
for (size_t block_index = 0; block_index < block_count; block_index++) {
LaneSet next_entry;
const auto& block = program.blocks[block_index];
if (block_index != 0 && !block.predecessors.empty()) {
next_entry = universe;
for (const auto predecessor: block.predecessors) {
if (predecessor >= block_count) {
next_entry.clear();
break;
}
next_entry = Intersect(next_entry, exit[predecessor]);
}
}
auto next_exit = TransferBlock(program, block, next_entry);
if (next_entry != entry[block_index] || next_exit != exit[block_index]) {
entry[block_index] = std::move(next_entry);
exit[block_index] = std::move(next_exit);
changed = true;
}
}
}
LaneSet forwarded;
for (size_t block_index = 0; block_index < block_count; block_index++) {
auto state = entry[block_index];
for (const auto& inst: program.blocks[block_index].instructions) {
LaneKey key;
if (ReadLaneKey(program, inst, key) && state.contains(key)) {
forwarded.insert(key);
}
ApplyInstruction(program, inst, state);
}
}
if (forwarded.empty()) {
return stats;
}
std::map<LaneKey, uint32_t> temporaries;
auto next_temporary = NextTemporaryScalarRegister(program);
for (const auto& key: forwarded) {
if (next_temporary == UINT32_MAX) {
return {};
}
temporaries.emplace(key, next_temporary++);
}
for (size_t block_index = 0; block_index < block_count; block_index++) {
const auto original = std::move(program.blocks[block_index].instructions);
auto& rewritten = program.blocks[block_index].instructions;
rewritten.clear();
rewritten.reserve(original.size() + temporaries.size());
auto state = entry[block_index];
for (const auto& inst: original) {
LaneKey read_key;
if (ReadLaneKey(program, inst, read_key) && state.contains(read_key)) {
const auto temporary = temporaries.find(read_key);
if (temporary != temporaries.end()) {
rewritten.push_back(ShadowRead(inst, temporary->second));
stats.rewritten_reads++;
ApplyInstruction(program, inst, state);
continue;
}
}
rewritten.push_back(inst);
LaneKey write_key;
if (WriteLaneKey(program, inst, write_key)) {
const auto temporary = temporaries.find(write_key);
if (temporary != temporaries.end()) {
rewritten.push_back(ShadowWrite(inst, temporary->second));
stats.shadow_writes++;
}
}
ApplyInstruction(program, inst, state);
}
}
return stats;
}
} // namespace Libs::Graphics::ShaderRecompiler::IR
@@ -0,0 +1,20 @@
#ifndef EMULATOR_INCLUDE_EMULATOR_GRAPHICS_SHADER_RECOMPILER_READLANEELIMINATION_H_
#define EMULATOR_INCLUDE_EMULATOR_GRAPHICS_SHADER_RECOMPILER_READLANEELIMINATION_H_
#include "graphics/shader/recompiler/ir/ShaderIR.h"
namespace Libs::Graphics::ShaderRecompiler::IR {
struct ReadLaneEliminationStats {
uint32_t rewritten_reads = 0;
uint32_t shadow_writes = 0;
};
// Replaces fixed-lane ReadLane operations that are reached by a matching WriteLane on every
// control-flow path. A synthetic scalar register snapshots the value at WriteLane execution time,
// so the rewrite remains valid when the source SGPR is subsequently overwritten.
[[nodiscard]] ReadLaneEliminationStats EliminateReadLane(Program& program);
} // namespace Libs::Graphics::ShaderRecompiler::IR
#endif /* EMULATOR_INCLUDE_EMULATOR_GRAPHICS_SHADER_RECOMPILER_READLANEELIMINATION_H_ */
@@ -15,7 +15,8 @@ constexpr uint64_t AddressMask = 0x0000ffffffffffffull;
Decoder::ImageDimension DescriptorDimension(const DescriptorValue& descriptor, Decoder::ImageDimension DescriptorDimension(const DescriptorValue& descriptor,
Decoder::ImageDimension requested) { Decoder::ImageDimension requested) {
const bool is_array = requested == Decoder::ImageDimension::Dim1DArray || const bool is_array = requested == Decoder::ImageDimension::Dim1DArray ||
requested == Decoder::ImageDimension::Dim2DArray; requested == Decoder::ImageDimension::Dim2DArray ||
requested == Decoder::ImageDimension::Dim2DMsaaArray;
switch (static_cast<Prospero::ImageType>((descriptor.dwords[3] >> 28u) & 0xfu)) { switch (static_cast<Prospero::ImageType>((descriptor.dwords[3] >> 28u) & 0xfu)) {
case Prospero::ImageType::kColor1D: return Decoder::ImageDimension::Dim1D; case Prospero::ImageType::kColor1D: return Decoder::ImageDimension::Dim1D;
case Prospero::ImageType::kColor1DArray: case Prospero::ImageType::kColor1DArray:
@@ -26,13 +27,17 @@ Decoder::ImageDimension DescriptorDimension(const DescriptorValue& descriptor,
case Prospero::ImageType::kColor3D: return Decoder::ImageDimension::Dim3D; case Prospero::ImageType::kColor3D: return Decoder::ImageDimension::Dim3D;
case Prospero::ImageType::kCube: return Decoder::ImageDimension::Dim2DArray; case Prospero::ImageType::kCube: return Decoder::ImageDimension::Dim2DArray;
case Prospero::ImageType::kColor2DArray: case Prospero::ImageType::kColor2DArray:
case Prospero::ImageType::kColor2DMsaaArray:
if (is_array) { if (is_array) {
return Decoder::ImageDimension::Dim2DArray; return Decoder::ImageDimension::Dim2DArray;
} }
return Decoder::ImageDimension::Dim2D; return Decoder::ImageDimension::Dim2D;
case Prospero::ImageType::kColor2D: case Prospero::ImageType::kColor2DMsaaArray:
case Prospero::ImageType::kColor2DMsaa: return Decoder::ImageDimension::Dim2D; if (is_array) {
return Decoder::ImageDimension::Dim2DMsaaArray;
}
return Decoder::ImageDimension::Dim2DMsaa;
case Prospero::ImageType::kColor2D: return Decoder::ImageDimension::Dim2D;
case Prospero::ImageType::kColor2DMsaa: return Decoder::ImageDimension::Dim2DMsaa;
default: return Decoder::ImageDimension::Unknown; default: return Decoder::ImageDimension::Unknown;
} }
} }
@@ -31,7 +31,7 @@ bool ValidateResourceSpecialization(const Program& program, const ResourceSnapsh
// Resolves the immutable dense resource topology against one runtime user-data/SRT snapshot. // Resolves the immutable dense resource topology against one runtime user-data/SRT snapshot.
// On failure the destination is unchanged. // On failure the destination is unchanged.
bool MaterializeResources(const Program& program, const SrtRuntime& runtime, bool MaterializeResources(const Program& program, const SrtRuntime& runtime,
ResourceSnapshot& snapshot, std::string* error); ResourceSnapshot& snapshot, std::string* error);
// Applies runtime descriptor shape/format facts to a copied dense topology before layout and // Applies runtime descriptor shape/format facts to a copied dense topology before layout and
// emission. On failure the program is unchanged. // emission. On failure the program is unchanged.
@@ -84,7 +84,7 @@ uint32_t ByteExtent(const Instruction& inst) {
} }
bool ContainsUnknown(const ScalarProvenance& provenance, uint32_t id, std::vector<uint8_t>& visited, bool ContainsUnknown(const ScalarProvenance& provenance, uint32_t id, std::vector<uint8_t>& visited,
std::vector<uint32_t>& path) { std::vector<uint32_t>& path) {
path.push_back(id); path.push_back(id);
if (id <= ScalarProvenance::Unknown || id >= provenance.values.size()) { if (id <= ScalarProvenance::Unknown || id >= provenance.values.size()) {
return true; return true;
@@ -115,7 +115,7 @@ bool ContainsUnknown(const ScalarProvenance& provenance, uint32_t id, std::vecto
} }
bool IsLoopInvariantValue(const ScalarProvenance& provenance, uint32_t id, bool IsLoopInvariantValue(const ScalarProvenance& provenance, uint32_t id,
std::vector<uint8_t>& visiting) { std::vector<uint8_t>& visiting) {
if (id <= ScalarProvenance::Unknown || id >= provenance.values.size()) { if (id <= ScalarProvenance::Unknown || id >= provenance.values.size()) {
return false; return false;
} }
@@ -679,11 +679,15 @@ enum class DescriptorBindingKind {
Sampled1DArray, Sampled1DArray,
Sampled2D, Sampled2D,
Sampled2DArray, Sampled2DArray,
Sampled2DMsaa,
Sampled2DMsaaArray,
Sampled3D, Sampled3D,
SampledUint1D, SampledUint1D,
SampledUint1DArray, SampledUint1DArray,
SampledUint2D, SampledUint2D,
SampledUint2DArray, SampledUint2DArray,
SampledUint2DMsaa,
SampledUint2DMsaaArray,
SampledUint3D, SampledUint3D,
Storage1D, Storage1D,
Storage1DArray, Storage1DArray,
+14 -14
View File
@@ -530,8 +530,8 @@ bool BuildSrtPlan(Program& program, std::string* error) {
} }
bool EvaluateDescriptorSource(const Program& program, uint32_t source, uint32_t use_pc, bool EvaluateDescriptorSource(const Program& program, uint32_t source, uint32_t use_pc,
const SrtRuntime& runtime, DescriptorValue& result, const SrtRuntime& runtime, DescriptorValue& result,
std::string* error) { std::string* error) {
const DescriptorSourceRequest request {source, use_pc}; const DescriptorSourceRequest request {source, use_pc};
std::vector<DescriptorValue> results; std::vector<DescriptorValue> results;
if (!EvaluateDescriptorSources(program, std::span {&request, 1}, runtime, results, error)) { if (!EvaluateDescriptorSources(program, std::span {&request, 1}, runtime, results, error)) {
@@ -542,11 +542,11 @@ bool EvaluateDescriptorSource(const Program& program, uint32_t source, uint32_t
} }
static bool EvaluateRuntimeSourcesImpl(const Program& program, static bool EvaluateRuntimeSourcesImpl(const Program& program,
std::span<const DescriptorSourceRequest> requests, std::span<const DescriptorSourceRequest> requests,
const SrtRuntime& runtime, const SrtRuntime& runtime,
std::vector<DescriptorValue>& results, std::vector<DescriptorValue>& results,
std::vector<uint32_t>& flat, bool evaluate_flat, std::vector<uint32_t>& flat, bool evaluate_flat,
std::string* error) { std::string* error) {
if (!program.srt_plan_complete) { if (!program.srt_plan_complete) {
if (error != nullptr) { if (error != nullptr) {
*error = Diagnostic(program, 0, "SRT plan is not ready"); *error = Diagnostic(program, 0, "SRT plan is not ready");
@@ -602,22 +602,22 @@ static bool EvaluateRuntimeSourcesImpl(const Program&
} }
bool EvaluateDescriptorSources(const Program& program, bool EvaluateDescriptorSources(const Program& program,
std::span<const DescriptorSourceRequest> requests, std::span<const DescriptorSourceRequest> requests,
const SrtRuntime& runtime, std::vector<DescriptorValue>& results, const SrtRuntime& runtime, std::vector<DescriptorValue>& results,
std::string* error) { std::string* error) {
std::vector<uint32_t> ignored; std::vector<uint32_t> ignored;
return EvaluateRuntimeSourcesImpl(program, requests, runtime, results, ignored, false, error); return EvaluateRuntimeSourcesImpl(program, requests, runtime, results, ignored, false, error);
} }
bool EvaluateRuntimeSources(const Program& program, bool EvaluateRuntimeSources(const Program& program,
std::span<const DescriptorSourceRequest> requests, std::span<const DescriptorSourceRequest> requests,
const SrtRuntime& runtime, std::vector<DescriptorValue>& results, const SrtRuntime& runtime, std::vector<DescriptorValue>& results,
std::vector<uint32_t>& flat, std::string* error) { std::vector<uint32_t>& flat, std::string* error) {
return EvaluateRuntimeSourcesImpl(program, requests, runtime, results, flat, true, error); return EvaluateRuntimeSourcesImpl(program, requests, runtime, results, flat, true, error);
} }
bool WalkSrt(const Program& program, const SrtRuntime& runtime, std::vector<uint32_t>& flat, bool WalkSrt(const Program& program, const SrtRuntime& runtime, std::vector<uint32_t>& flat,
std::string* error) { std::string* error) {
std::vector<DescriptorValue> ignored; std::vector<DescriptorValue> ignored;
return EvaluateRuntimeSources(program, {}, runtime, ignored, flat, error); return EvaluateRuntimeSources(program, {}, runtime, ignored, flat, error);
} }
@@ -30,22 +30,22 @@ bool FoldScalarConstant(const ScalarProvenance& provenance, uint32_t value, uint
bool BuildSrtPlan(Program& program, std::string* error); bool BuildSrtPlan(Program& program, std::string* error);
bool EvaluateDescriptorSource(const Program& program, uint32_t source, uint32_t use_pc, bool EvaluateDescriptorSource(const Program& program, uint32_t source, uint32_t use_pc,
const SrtRuntime& runtime, DescriptorValue& result, const SrtRuntime& runtime, DescriptorValue& result,
std::string* error); std::string* error);
// Evaluates one runtime snapshot transactionally. Scalar values and ReadConst results shared by // Evaluates one runtime snapshot transactionally. Scalar values and ReadConst results shared by
// several descriptors are memoized once across the batch. // several descriptors are memoized once across the batch.
bool EvaluateDescriptorSources(const Program& program, bool EvaluateDescriptorSources(const Program& program,
std::span<const DescriptorSourceRequest> requests, std::span<const DescriptorSourceRequest> requests,
const SrtRuntime& runtime, std::vector<DescriptorValue>& results, const SrtRuntime& runtime, std::vector<DescriptorValue>& results,
std::string* error); std::string* error);
// Evaluates descriptor sources and the flattened immediate SRT with one memoized scalar walk. // Evaluates descriptor sources and the flattened immediate SRT with one memoized scalar walk.
// On failure neither destination is changed. // On failure neither destination is changed.
bool EvaluateRuntimeSources(const Program& program, bool EvaluateRuntimeSources(const Program& program,
std::span<const DescriptorSourceRequest> requests, std::span<const DescriptorSourceRequest> requests,
const SrtRuntime& runtime, std::vector<DescriptorValue>& results, const SrtRuntime& runtime, std::vector<DescriptorValue>& results,
std::vector<uint32_t>& flat, std::string* error); std::vector<uint32_t>& flat, std::string* error);
bool WalkSrt(const Program& program, const SrtRuntime& runtime, std::vector<uint32_t>& flat, bool WalkSrt(const Program& program, const SrtRuntime& runtime, std::vector<uint32_t>& flat,
std::string* error); std::string* error);
+9 -12
View File
@@ -13,8 +13,8 @@
#include "graphics/guest_gpu/graphicsRun.h" #include "graphics/guest_gpu/graphicsRun.h"
#include "graphics/guest_gpu/hardwareContext.h" #include "graphics/guest_gpu/hardwareContext.h"
#include "graphics/host_gpu/renderer/renderContext.h" #include "graphics/host_gpu/renderer/renderContext.h"
#include "graphics/shader/recompiler/decompiler/ShaderDecoder.h"
#include "graphics/shader/recompiler/ShaderRecompiler.h" #include "graphics/shader/recompiler/ShaderRecompiler.h"
#include "graphics/shader/recompiler/decompiler/ShaderDecoder.h"
#include "graphics/shader/shaderVertexMetadata.h" #include "graphics/shader/shaderVertexMetadata.h"
#include "libs/errno.h" #include "libs/errno.h"
#include "spirv-tools/libspirv.h" #include "spirv-tools/libspirv.h"
@@ -828,11 +828,10 @@ static void ShaderGetStaticInputInfoPS(
vs_info.stage.program != nullptr && !vs_info.stage.program->bindings.descriptors.empty() vs_info.stage.program != nullptr && !vs_info.stage.program->bindings.descriptors.empty()
? 1 ? 1
: 0; : 0;
ps_info.push_constant_offset = ps_info.push_constant_offset = vs_info.stage.program != nullptr
vs_info.stage.program != nullptr ? vs_info.stage.program->bindings.push_constant_offset +
? vs_info.stage.program->bindings.push_constant_offset + vs_info.stage.program->bindings.push_constant_size
vs_info.stage.program->bindings.push_constant_size : 0;
: 0;
for (int i = 0; i < 8; i++) { for (int i = 0; i < 8; i++) {
ps_info.target_output_mode[i] = sh.target_output_mode[i]; ps_info.target_output_mode[i] = sh.target_output_mode[i];
@@ -1294,9 +1293,8 @@ static void DumpShaderRecompilerSpirv(const char* type, uint64_t shader_hash,
static std::atomic_int id = 0; static std::atomic_int id = 0;
const auto base_name = const auto base_name = Config::GetShaderLogFolder() /
Config::GetShaderLogFolder() / fmt::format("{:04d}_new_shader_{}_{:016x}", id++, type, shader_hash);
fmt::format("{:04d}_new_shader_{}_{:016x}", id++, type, shader_hash);
Common::File::CreateDirectories(base_name.parent_path()); Common::File::CreateDirectories(base_name.parent_path());
Common::File spv_file; Common::File spv_file;
@@ -1345,9 +1343,8 @@ static void DumpShaderRecompilerOriginal(const char* type, uint64_t shader_hash,
static std::atomic_int id = 0; static std::atomic_int id = 0;
const auto base_name = const auto base_name = Config::GetShaderLogFolder() / "original" /
Config::GetShaderLogFolder() / "original" / fmt::format("{:04d}_new_shader_{}_{:016x}", id++, type, shader_hash);
fmt::format("{:04d}_new_shader_{}_{:016x}", id++, type, shader_hash);
Common::File::CreateDirectories(base_name.parent_path()); Common::File::CreateDirectories(base_name.parent_path());
Common::File bin_file; Common::File bin_file;
+2 -2
View File
@@ -42,8 +42,8 @@ struct ShaderStageRuntime {
// Resolves an immutable native shader plan against current user data. The prior stage is preserved // Resolves an immutable native shader plan against current user data. The prior stage is preserved
// if any ReadConst, snapshot, or specialization check fails. // if any ReadConst, snapshot, or specialization check fails.
bool ShaderMaterializeStageRuntime(std::shared_ptr<const ShaderRecompiler::IR::Program> program, bool ShaderMaterializeStageRuntime(std::shared_ptr<const ShaderRecompiler::IR::Program> program,
std::span<const uint32_t> user_data, uint64_t shader_base, std::span<const uint32_t> user_data, uint64_t shader_base,
ShaderStageRuntime& stage, std::string* error); ShaderStageRuntime& stage, std::string* error);
struct ShaderId { struct ShaderId {
uint32_t hash0 = 0; uint32_t hash0 = 0;
+2 -2
View File
@@ -6,8 +6,8 @@
namespace Libs::Graphics { namespace Libs::Graphics {
bool ShaderMaterializeStageRuntime(std::shared_ptr<const ShaderRecompiler::IR::Program> program, bool ShaderMaterializeStageRuntime(std::shared_ptr<const ShaderRecompiler::IR::Program> program,
std::span<const uint32_t> user_data, uint64_t shader_base, std::span<const uint32_t> user_data, uint64_t shader_base,
ShaderStageRuntime& stage, std::string* error) { ShaderStageRuntime& stage, std::string* error) {
if (program == nullptr) { if (program == nullptr) {
if (error != nullptr) { if (error != nullptr) {
*error = "missing native shader plan"; *error = "missing native shader plan";
+1 -1
View File
@@ -17,7 +17,7 @@ bool Fail(std::string* error, const char* message) {
} // namespace } // namespace
bool ShaderReadVertexMetadata(const ShaderMappedData& data, uint32_t max_user_sgprs, bool ShaderReadVertexMetadata(const ShaderMappedData& data, uint32_t max_user_sgprs,
ShaderVertexMetadata& metadata, std::string* error) { ShaderVertexMetadata& metadata, std::string* error) {
if (data.user_data == nullptr) { if (data.user_data == nullptr) {
return Fail(error, "missing AGC user-data header"); return Fail(error, "missing AGC user-data header");
} }
+1 -1
View File
@@ -17,7 +17,7 @@ struct ShaderVertexMetadata {
// Copies the small AGC metadata subset used by the vertex path after validating every guest range. // Copies the small AGC metadata subset used by the vertex path after validating every guest range.
bool ShaderReadVertexMetadata(const ShaderMappedData& data, uint32_t max_user_sgprs, bool ShaderReadVertexMetadata(const ShaderMappedData& data, uint32_t max_user_sgprs,
ShaderVertexMetadata& metadata, std::string* error); ShaderVertexMetadata& metadata, std::string* error);
} // namespace Libs::Graphics } // namespace Libs::Graphics
+8 -11
View File
@@ -33,8 +33,8 @@ static uint64_t MonotonicTimeNs() {
} }
static std::unordered_map<KernelEqueue, KernelEqueueRef> g_equeues; static std::unordered_map<KernelEqueue, KernelEqueueRef> g_equeues;
static Common::Mutex g_equeues_mutex; static Common::Mutex g_equeues_mutex;
static uint64_t g_next_equeue = 1; static uint64_t g_next_equeue = 1;
class KernelEqueuePrivate { class KernelEqueuePrivate {
public: public:
@@ -459,8 +459,7 @@ int KYTY_SYSV_ABI KernelAddUserEvent(KernelEqueue eq, int id) {
int KYTY_SYSV_ABI KernelAddUserEventEdge(KernelEqueue eq, int id) { int KYTY_SYSV_ABI KernelAddUserEventEdge(KernelEqueue eq, int id) {
PRINT_NAME(); PRINT_NAME();
LOGF("\t user event edge add: eq = 0x%016" PRIx64 ", id = %d\n", static_cast<uint64_t>(eq), LOGF("\t user event edge add: eq = 0x%016" PRIx64 ", id = %d\n", static_cast<uint64_t>(eq), id);
id);
KernelEqueueEvent event {}; KernelEqueueEvent event {};
event.event.ident = static_cast<uintptr_t>(id); event.event.ident = static_cast<uintptr_t>(id);
@@ -485,7 +484,7 @@ int KYTY_SYSV_ABI KernelTriggerUserEvent(KernelEqueue eq, int id, void* udata) {
} }
int KYTY_SYSV_ABI KernelTriggerUserEventForAll(int id, void* udata) { int KYTY_SYSV_ABI KernelTriggerUserEventForAll(int id, void* udata) {
int triggered = 0; int triggered = 0;
std::vector<KernelEqueueRef> queues; std::vector<KernelEqueueRef> queues;
{ {
@@ -507,8 +506,7 @@ int KYTY_SYSV_ABI KernelTriggerUserEventForAll(int id, void* udata) {
int KYTY_SYSV_ABI KernelDeleteUserEvent(KernelEqueue eq, int id) { int KYTY_SYSV_ABI KernelDeleteUserEvent(KernelEqueue eq, int id) {
PRINT_NAME(); PRINT_NAME();
LOGF("\t user event delete: eq = 0x%016" PRIx64 ", id = %d\n", static_cast<uint64_t>(eq), LOGF("\t user event delete: eq = 0x%016" PRIx64 ", id = %d\n", static_cast<uint64_t>(eq), id);
id);
return KernelDeleteEvent(eq, static_cast<uintptr_t>(id), KERNEL_EVFILT_USER); return KernelDeleteEvent(eq, static_cast<uintptr_t>(id), KERNEL_EVFILT_USER);
} }
@@ -577,8 +575,7 @@ int KYTY_SYSV_ABI KernelAddAmprSystemEvent(KernelEqueue eq, int id, void* udata)
int KYTY_SYSV_ABI KernelDeleteAmprEvent(KernelEqueue eq, int id) { int KYTY_SYSV_ABI KernelDeleteAmprEvent(KernelEqueue eq, int id) {
PRINT_NAME(); PRINT_NAME();
LOGF("\t AMPR event delete: eq = 0x%016" PRIx64 ", id = %d\n", static_cast<uint64_t>(eq), LOGF("\t AMPR event delete: eq = 0x%016" PRIx64 ", id = %d\n", static_cast<uint64_t>(eq), id);
id);
if (eq != KERNEL_EQUEUE_INVALID) { if (eq != KERNEL_EQUEUE_INVALID) {
(void)KernelDeleteEvent(eq, static_cast<uintptr_t>(id), KERNEL_EVFILT_USER); (void)KernelDeleteEvent(eq, static_cast<uintptr_t>(id), KERNEL_EVFILT_USER);
@@ -590,8 +587,8 @@ int KYTY_SYSV_ABI KernelDeleteAmprEvent(KernelEqueue eq, int id) {
int KYTY_SYSV_ABI KernelDeleteAmprSystemEvent(KernelEqueue eq, int id) { int KYTY_SYSV_ABI KernelDeleteAmprSystemEvent(KernelEqueue eq, int id) {
PRINT_NAME(); PRINT_NAME();
LOGF("\t AMPR system event delete: eq = 0x%016" PRIx64 ", id = %d\n", LOGF("\t AMPR system event delete: eq = 0x%016" PRIx64 ", id = %d\n", static_cast<uint64_t>(eq),
static_cast<uint64_t>(eq), id); id);
return KernelDeleteAmprEvent(eq, id); return KernelDeleteAmprEvent(eq, id);
} }
+1 -1
View File
@@ -41,7 +41,7 @@ struct KernelEvent {
}; };
struct KernelFilter { struct KernelFilter {
void* data = nullptr; void* data = nullptr;
std::shared_ptr<void> owner; std::shared_ptr<void> owner;
trigger_func_t trigger_func = nullptr; trigger_func_t trigger_func = nullptr;
reset_func_t reset_func = nullptr; reset_func_t reset_func = nullptr;
+2 -2
View File
@@ -206,8 +206,8 @@ bool ConfigurationItem::operator<(const QTreeWidgetItem& other) const {
GetStatusText(other_item->m_info->game_status); GetStatusText(other_item->m_info->game_status);
case GameVersionColumn: case GameVersionColumn:
case FirmwareVersionColumn: { case FirmwareVersionColumn: {
const auto& version = column == GameVersionColumn ? m_info->gameVersion const auto& version =
: m_info->firmwareVer; column == GameVersionColumn ? m_info->gameVersion : m_info->firmwareVer;
const auto& other_version = column == GameVersionColumn const auto& other_version = column == GameVersionColumn
? other_item->m_info->gameVersion ? other_item->m_info->gameVersion
: other_item->m_info->firmwareVer; : other_item->m_info->firmwareVer;
+1 -1
View File
@@ -1,6 +1,5 @@
#include "configurationListWidget.h" #include "configurationListWidget.h"
#include "patchesDialog.h"
#include "common.h" #include "common.h"
#include "compatibilityDatabase.h" #include "compatibilityDatabase.h"
#include "configuration.h" #include "configuration.h"
@@ -8,6 +7,7 @@
#include "configurationItem.h" #include "configurationItem.h"
#include "gameListTreeWidget.h" #include "gameListTreeWidget.h"
#include "mainDialog.h" #include "mainDialog.h"
#include "patchesDialog.h"
#include "trophyViewerDialog.h" #include "trophyViewerDialog.h"
#include <QAbstractItemModel> #include <QAbstractItemModel>
+16 -8
View File
@@ -272,10 +272,18 @@ static bool FindTerminal(QString* program, QStringList* prefix) {
}; };
static const TerminalSpec candidates[] = { static const TerminalSpec candidates[] = {
{"x-terminal-emulator", "-e"}, {"gnome-terminal", "--"}, {"konsole", "-e"}, {"x-terminal-emulator", "-e"},
{"xfce4-terminal", "-x"}, {"mate-terminal", "--"}, {"tilix", "-e"}, {"gnome-terminal", "--"},
{"alacritty", "-e"}, {"kitty", nullptr}, {"foot", nullptr}, {"konsole", "-e"},
{"wezterm", "-e"}, {"urxvt", "-e"}, {"xterm", "-e"}, {"xfce4-terminal", "-x"},
{"mate-terminal", "--"},
{"tilix", "-e"},
{"alacritty", "-e"},
{"kitty", nullptr},
{"foot", nullptr},
{"wezterm", "-e"},
{"urxvt", "-e"},
{"xterm", "-e"},
}; };
const auto try_candidate = [program, prefix](const QString& executable, const char* separator) { const auto try_candidate = [program, prefix](const QString& executable, const char* separator) {
@@ -293,7 +301,7 @@ static bool FindTerminal(QString* program, QStringList* prefix) {
if (const auto from_env = qEnvironmentVariable("TERMINAL"); !from_env.isEmpty()) { if (const auto from_env = qEnvironmentVariable("TERMINAL"); !from_env.isEmpty()) {
// Reuse the known separator for an explicit terminal. // Reuse the known separator for an explicit terminal.
const auto env_name = QFileInfo(from_env).fileName(); const auto env_name = QFileInfo(from_env).fileName();
const char* separator = "-e"; const char* separator = "-e";
for (const auto& candidate: candidates) { for (const auto& candidate: candidates) {
if (env_name == QLatin1String(candidate.executable)) { if (env_name == QLatin1String(candidate.executable)) {
@@ -379,9 +387,9 @@ void MainDialog::RunInterpreter(QProcess* process, const Configuration& info) {
#if !defined(_WIN32) #if !defined(_WIN32)
// Report immediate launch failures. // Report immediate launch failures.
if (!process->waitForStarted(5000)) { if (!process->waitForStarted(5000)) {
QMessageBox::critical(this, tr("Error"), QMessageBox::critical(
tr("Failed to start:\n%1\n\n%2") this, tr("Error"),
.arg(process->program(), process->errorString())); tr("Failed to start:\n%1\n\n%2").arg(process->program(), process->errorString()));
return; return;
} }
#endif #endif
+5 -7
View File
@@ -59,16 +59,14 @@ void PatchesDialog::Load() {
return; return;
} }
const auto patches = QJsonDocument::fromJson(file.readAll()) const auto patches =
.object() QJsonDocument::fromJson(file.readAll()).object().value(QStringLiteral("patches")).toArray();
.value(QStringLiteral("patches"))
.toArray();
for (const auto& value: patches) { for (const auto& value: patches) {
const auto patch = value.toObject(); const auto patch = value.toObject();
auto* item = new QListWidgetItem(patch.value(QStringLiteral("name")).toString(), m_patches); auto* item = new QListWidgetItem(patch.value(QStringLiteral("name")).toString(), m_patches);
item->setFlags(item->flags() | Qt::ItemIsUserCheckable); item->setFlags(item->flags() | Qt::ItemIsUserCheckable);
item->setCheckState(patch.value(QStringLiteral("enabled")).toBool(true) ? Qt::Checked item->setCheckState(patch.value(QStringLiteral("enabled")).toBool(true) ? Qt::Checked
: Qt::Unchecked); : Qt::Unchecked);
} }
m_apply->setEnabled(!patches.isEmpty()); m_apply->setEnabled(!patches.isEmpty());
@@ -84,8 +82,8 @@ void PatchesDialog::Save() {
auto document = QJsonDocument::fromJson(input.readAll()); auto document = QJsonDocument::fromJson(input.readAll());
input.close(); input.close();
auto root = document.object(); auto root = document.object();
auto patches = root.value(QStringLiteral("patches")).toArray(); auto patches = root.value(QStringLiteral("patches")).toArray();
for (int index = 0; index < patches.size(); index++) { for (int index = 0; index < patches.size(); index++) {
auto patch = patches[index].toObject(); auto patch = patches[index].toObject();
patch.insert(QStringLiteral("enabled"), patch.insert(QStringLiteral("enabled"),
+1 -1
View File
@@ -52,7 +52,7 @@ KYTY_SUBSYSTEM_INIT(Graphics) {
auto& presenter = WindowInit(width, height); auto& presenter = WindowInit(width, height);
auto& video_out = VideoOut::VideoOutInit(width, height, presenter); auto& video_out = VideoOut::VideoOutInit(width, height, presenter);
g_renderer = &presenter.Renderer(); g_renderer = &presenter.Renderer();
g_renderer->InitializeGpu(&video_out); g_renderer->InitializeGpu(&video_out);
ShaderInit(); ShaderInit();
} }
+2 -2
View File
@@ -1511,8 +1511,8 @@ static int ExecuteAprCommandBuffer(uint64_t command_buffer, int32_t* execution_r
} break; } break;
case CommandKind::KernelEvent: { case CommandKind::KernelEvent: {
const auto& command = state.kernel_event_commands[entry.index]; const auto& command = state.kernel_event_commands[entry.index];
const auto eq = static_cast<LibKernel::EventQueue::KernelEqueue>(command.eq); const auto eq = static_cast<LibKernel::EventQueue::KernelEqueue>(command.eq);
auto result = LibKernel::EventQueue::KernelTriggerUserEvent( auto result = LibKernel::EventQueue::KernelTriggerUserEvent(
eq, command.id, reinterpret_cast<void*>(command.data)); eq, command.id, reinterpret_cast<void*>(command.data));
if (result != OK) { if (result != OK) {
LOGF("\tAPR submit event failed: eq=0x%016" PRIx64 ", id=%" PRId32 LOGF("\tAPR submit event failed: eq=0x%016" PRIx64 ", id=%" PRId32
+1 -1
View File
@@ -1241,7 +1241,7 @@ static int KYTY_SYSV_ABI KernelRaiseException(Pthread thread, int signum) {
if (thread == PthreadSelfOrNull()) { if (thread == PthreadSelfOrNull()) {
SignalDispatchScope scope; SignalDispatchScope scope;
auto ctx = CreateCurrentGuestCallSignalUcontext( auto ctx = CreateCurrentGuestCallSignalUcontext(
reinterpret_cast<uint64_t>(__builtin_return_address(0))); reinterpret_cast<uint64_t>(__builtin_return_address(0)));
handler(signum, &ctx); handler(signum, &ctx);
return OK; return OK;
} }
+10 -8
View File
@@ -438,11 +438,12 @@ int KYTY_SYSV_ABI SaveDataMount3(const SaveDataMount3* mount, SaveDataMountResul
*mount_result = {}; *mount_result = {};
Common::LockGuard lock(g_mount_mutex); Common::LockGuard lock(g_mount_mutex);
const std::string dir_name = mount->dir_name->data; const std::string dir_name = mount->dir_name->data;
const std::string mount_dir = std::string(SAVE_DATA_DIR) + "/" + get_title_id() + "/" + dir_name; const std::string mount_dir =
const bool create = ((mount->mount_mode & 4u) != 0); std::string(SAVE_DATA_DIR) + "/" + get_title_id() + "/" + dir_name;
const bool create2 = ((mount->mount_mode & 32u) != 0); const bool create = ((mount->mount_mode & 4u) != 0);
const bool open = (!create && !create2 && ((mount->mount_mode & 3u) != 0)); const bool create2 = ((mount->mount_mode & 32u) != 0);
const bool open = (!create && !create2 && ((mount->mount_mode & 3u) != 0));
const int slot = g_mount_slots.FindAvailable(dir_name); const int slot = g_mount_slots.FindAvailable(dir_name);
if (slot == SaveDataMountSlots::BUSY) { if (slot == SaveDataMountSlots::BUSY) {
@@ -594,9 +595,10 @@ int KYTY_SYSV_ABI SaveDataTransferringMount(const SaveDataTransferringMount* mou
*mount_result = {}; *mount_result = {};
Common::LockGuard lock(g_mount_mutex); Common::LockGuard lock(g_mount_mutex);
const std::string dir_name = mount->dir_name->data; const std::string dir_name = mount->dir_name->data;
const std::string mount_dir = std::string(SAVE_DATA_DIR) + "/" + get_title_id() + "/" + dir_name; const std::string mount_dir =
const int slot = g_mount_slots.FindAvailable(dir_name); std::string(SAVE_DATA_DIR) + "/" + get_title_id() + "/" + dir_name;
const int slot = g_mount_slots.FindAvailable(dir_name);
if (slot == SaveDataMountSlots::BUSY) { if (slot == SaveDataMountSlots::BUSY) {
return SAVE_DATA_ERROR_BUSY; return SAVE_DATA_ERROR_BUSY;
} }
+252 -36
View File
@@ -1,10 +1,12 @@
#include "common/abi.h" #include "common/abi.h"
#include "libs/errno.h" #include "libs/errno.h"
#include "libs/libs.h" #include "libs/libs.h"
#include "libs/videoDec2Decoder.h"
#include "loader/symbolDatabase.h" #include "loader/symbolDatabase.h"
#include <cstddef> #include <cstddef>
#include <cstdint> #include <cstdint>
#include <cstring>
#include <mutex> #include <mutex>
#include <unordered_set> #include <unordered_set>
@@ -14,6 +16,7 @@ LIB_VERSION("Videodec2", 1, "Videodec2", 1, 1);
namespace VideoDec2 { namespace VideoDec2 {
constexpr int32_t VIDEODEC2_ERROR_API_FAIL = -2128805632; // 0x811d0100
constexpr int32_t VIDEODEC2_ERROR_STRUCT_SIZE = -2128805631; // 0x811d0101 constexpr int32_t VIDEODEC2_ERROR_STRUCT_SIZE = -2128805631; // 0x811d0101
constexpr int32_t VIDEODEC2_ERROR_ARGUMENT_POINTER = -2128805630; // 0x811d0102 constexpr int32_t VIDEODEC2_ERROR_ARGUMENT_POINTER = -2128805630; // 0x811d0102
constexpr int32_t VIDEODEC2_ERROR_DECODER_INSTANCE = -2128805629; // 0x811d0103 constexpr int32_t VIDEODEC2_ERROR_DECODER_INSTANCE = -2128805629; // 0x811d0103
@@ -21,13 +24,20 @@ constexpr int32_t VIDEODEC2_ERROR_MEMORY_SIZE = -2128805628; // 0x811d0
constexpr int32_t VIDEODEC2_ERROR_MEMORY_POINTER = -2128805627; // 0x811d0105 constexpr int32_t VIDEODEC2_ERROR_MEMORY_POINTER = -2128805627; // 0x811d0105
constexpr int32_t VIDEODEC2_ERROR_FRAME_BUFFER_SIZE = -2128805626; // 0x811d0106 constexpr int32_t VIDEODEC2_ERROR_FRAME_BUFFER_SIZE = -2128805626; // 0x811d0106
constexpr int32_t VIDEODEC2_ERROR_FRAME_BUFFER_POINTER = -2128805625; // 0x811d0107 constexpr int32_t VIDEODEC2_ERROR_FRAME_BUFFER_POINTER = -2128805625; // 0x811d0107
constexpr int32_t VIDEODEC2_ERROR_ACCESS_UNIT_SIZE = -2128805619; // 0x811d010d
constexpr int32_t VIDEODEC2_ERROR_ACCESS_UNIT_POINTER = -2128805618; // 0x811d010e
constexpr int32_t VIDEODEC2_ERROR_OUTPUT_INFO = -2128805617; // 0x811d010f
constexpr int32_t VIDEODEC2_ERROR_COMPUTE_QUEUE = -2128805616; // 0x811d0110
constexpr int32_t VIDEODEC2_ERROR_CONFIG_INFO = -2128805376; // 0x811d0200 constexpr int32_t VIDEODEC2_ERROR_CONFIG_INFO = -2128805376; // 0x811d0200
constexpr int32_t VIDEODEC2_ERROR_COMPUTE_PIPE_ID = -2128805375; // 0x811d0201 constexpr int32_t VIDEODEC2_ERROR_COMPUTE_PIPE_ID = -2128805375; // 0x811d0201
constexpr int32_t VIDEODEC2_ERROR_COMPUTE_QUEUE_ID = -2128805374; // 0x811d0202 constexpr int32_t VIDEODEC2_ERROR_COMPUTE_QUEUE_ID = -2128805374; // 0x811d0202
constexpr int32_t VIDEODEC2_ERROR_RESOURCE_TYPE = -2128805373; // 0x811d0203 constexpr int32_t VIDEODEC2_ERROR_RESOURCE_TYPE = -2128805373; // 0x811d0203
constexpr int32_t VIDEODEC2_ERROR_CODEC_TYPE = -2128805372; // 0x811d0204
constexpr int32_t VIDEODEC2_ERROR_INPUT_QUEUE_DEPTH = -2128805370; // 0x811d0206 constexpr int32_t VIDEODEC2_ERROR_INPUT_QUEUE_DEPTH = -2128805370; // 0x811d0206
constexpr int32_t VIDEODEC2_ERROR_DPB_FRAME_COUNT = -2128805367; // 0x811d0209 constexpr int32_t VIDEODEC2_ERROR_DPB_FRAME_COUNT = -2128805367; // 0x811d0209
constexpr int32_t VIDEODEC2_ERROR_FRAME_WIDTH_HEIGHT = -2128805366; // 0x811d020a constexpr int32_t VIDEODEC2_ERROR_FRAME_WIDTH_HEIGHT = -2128805366; // 0x811d020a
constexpr int32_t VIDEODEC2_ERROR_ACCESS_UNIT = -2128805119; // 0x811d0301
constexpr int32_t VIDEODEC2_ERROR_OVERSIZE_DECODE = -2128805118; // 0x811d0302
constexpr uint32_t VIDEODEC2_RESOURCE_TYPE_COMPUTE = 1; constexpr uint32_t VIDEODEC2_RESOURCE_TYPE_COMPUTE = 1;
constexpr size_t VIDEODEC2_MIN_MEMORY_SIZE = 16ull * 1024ull * 1024ull; constexpr size_t VIDEODEC2_MIN_MEMORY_SIZE = 16ull * 1024ull * 1024ull;
@@ -101,6 +111,70 @@ struct Videodec2FrameBuffer {
bool is_accepted; bool is_accepted;
}; };
struct Videodec2AvcPictureInfo {
size_t this_size;
bool is_valid;
uint64_t pts_data;
uint64_t dts_data;
uint64_t attached_data;
uint8_t idr_picture_flag;
uint8_t profile_idc;
uint8_t level_idc;
uint32_t pic_width_in_mbs_minus1;
uint32_t pic_height_in_map_units_minus1;
uint8_t frame_mbs_only_flag;
uint8_t frame_cropping_flag;
uint32_t frame_crop_left_offset;
uint32_t frame_crop_right_offset;
uint32_t frame_crop_top_offset;
uint32_t frame_crop_bottom_offset;
uint8_t aspect_ratio_info_present_flag;
uint8_t aspect_ratio_idc;
uint16_t sar_width;
uint16_t sar_height;
uint8_t video_signal_type_present_flag;
uint8_t video_format;
uint8_t video_full_range_flag;
uint8_t colour_description_present_flag;
uint8_t colour_primaries;
uint8_t transfer_characteristics;
uint8_t matrix_coefficients;
uint8_t timing_info_present_flag;
uint32_t num_units_in_tick;
uint32_t time_scale;
uint8_t fixed_frame_rate_flag;
uint8_t bitstream_restriction_flag;
uint8_t max_dec_frame_buffering;
uint8_t pic_struct_present_flag;
uint8_t pic_struct;
uint8_t field_pic_flag;
uint8_t bottom_field_flag;
uint8_t sequence_parameter_set_present_flag;
uint8_t picture_parameter_set_present_flag;
uint8_t au_delimiter_present_flag;
uint8_t end_of_sequence_present_flag;
uint8_t end_of_stream_present_flag;
uint8_t filler_data_present_flag;
uint8_t picture_timing_sei_present_flag;
uint8_t buffering_period_sei_present_flag;
uint8_t constraint_set0_flag;
uint8_t constraint_set1_flag;
uint8_t constraint_set2_flag;
uint8_t constraint_set3_flag;
uint8_t constraint_set4_flag;
uint8_t constraint_set5_flag;
};
struct Videodec2ComputeMemoryInfo { struct Videodec2ComputeMemoryInfo {
size_t this_size; size_t this_size;
size_t cpu_gpu_memory_size; size_t cpu_gpu_memory_size;
@@ -116,10 +190,7 @@ struct Videodec2ComputeConfigInfo {
uint16_t reserved1; uint16_t reserved1;
}; };
struct DecoderState { using DecoderState = Decoder::Instance;
uint64_t magic;
uint32_t codec_type;
};
static_assert(sizeof(Videodec2ComputeMemoryInfo) == 24); static_assert(sizeof(Videodec2ComputeMemoryInfo) == 24);
static_assert(sizeof(Videodec2ComputeConfigInfo) == 16); static_assert(sizeof(Videodec2ComputeConfigInfo) == 16);
@@ -128,8 +199,7 @@ static_assert(sizeof(Videodec2DecoderMemoryInfo) == 72);
static_assert(sizeof(Videodec2InputData) == 48); static_assert(sizeof(Videodec2InputData) == 48);
static_assert(sizeof(Videodec2OutputInfo) == 56); static_assert(sizeof(Videodec2OutputInfo) == 56);
static_assert(sizeof(Videodec2FrameBuffer) == 32); static_assert(sizeof(Videodec2FrameBuffer) == 32);
static_assert(sizeof(Videodec2AvcPictureInfo) == 120);
constexpr uint64_t DECODER_MAGIC = 0x4b59545956444543ull; // KYTYVDEC
static std::mutex g_decoder_mutex; static std::mutex g_decoder_mutex;
static std::unordered_set<void*> g_decoders; static std::unordered_set<void*> g_decoders;
@@ -156,15 +226,54 @@ static void FillNoPictureOutput(const Videodec2FrameBuffer* frame_buffer,
output_info->frame_height = 0; output_info->frame_height = 0;
output_info->frame_buffer = frame_buffer != nullptr ? frame_buffer->frame_buffer : nullptr; output_info->frame_buffer = frame_buffer != nullptr ? frame_buffer->frame_buffer : nullptr;
output_info->frame_buffer_size = frame_buffer != nullptr ? frame_buffer->frame_buffer_size : 0; output_info->frame_buffer_size = frame_buffer != nullptr ? frame_buffer->frame_buffer_size : 0;
output_info->frame_format = VIDEODEC2_FRAME_FORMAT_DEFAULT; if (output_info->this_size == sizeof(Videodec2OutputInfo)) {
output_info->frame_pitch_in_bytes = 0; output_info->frame_format = VIDEODEC2_FRAME_FORMAT_DEFAULT;
output_info->frame_pitch_in_bytes = 0;
}
} }
static int32_t ValidateDecoderConfig(const Videodec2DecoderConfigInfo* config) { static int32_t MapDecoderResult(Decoder::Result result) {
switch (result) {
case Decoder::Result::Ok: return OK;
case Decoder::Result::ApiFail: return VIDEODEC2_ERROR_API_FAIL;
case Decoder::Result::AccessUnit: return VIDEODEC2_ERROR_ACCESS_UNIT;
case Decoder::Result::FrameBufferSize: return VIDEODEC2_ERROR_FRAME_BUFFER_SIZE;
case Decoder::Result::OversizeDecode: return VIDEODEC2_ERROR_OVERSIZE_DECODE;
}
return VIDEODEC2_ERROR_API_FAIL;
}
static void ApplyDecodedOutput(const Decoder::Output& decoded, Videodec2FrameBuffer* frame_buffer,
Videodec2OutputInfo* output_info) {
frame_buffer->is_accepted = decoded.buffer_accepted;
if (!decoded.valid) {
return;
}
output_info->is_valid = true;
output_info->is_error_frame = decoded.error_frame;
output_info->picture_count = 1;
output_info->codec_type = decoded.codec_type;
output_info->frame_width = decoded.width;
output_info->frame_pitch = decoded.pitch;
output_info->frame_height = decoded.height;
output_info->frame_buffer = decoded.buffer;
output_info->frame_buffer_size = decoded.buffer_size;
if (output_info->this_size == sizeof(Videodec2OutputInfo)) {
output_info->frame_format = VIDEODEC2_FRAME_FORMAT_DEFAULT;
output_info->frame_pitch_in_bytes = decoded.pitch;
}
}
static int32_t ValidateDecoderConfig(const Videodec2DecoderConfigInfo* config,
bool require_compute_queue) {
if (config->resource_type != VIDEODEC2_RESOURCE_TYPE_COMPUTE) { if (config->resource_type != VIDEODEC2_RESOURCE_TYPE_COMPUTE) {
return VIDEODEC2_ERROR_RESOURCE_TYPE; return VIDEODEC2_ERROR_RESOURCE_TYPE;
} }
if (!Decoder::IsCodecSupported(config->codec_type)) {
return VIDEODEC2_ERROR_CODEC_TYPE;
}
if (config->reserved0 != 0 || config->reserved1 != 0) { if (config->reserved0 != 0 || config->reserved1 != 0) {
return VIDEODEC2_ERROR_CONFIG_INFO; return VIDEODEC2_ERROR_CONFIG_INFO;
} }
@@ -182,8 +291,8 @@ static int32_t ValidateDecoderConfig(const Videodec2DecoderConfigInfo* config) {
return VIDEODEC2_ERROR_FRAME_WIDTH_HEIGHT; return VIDEODEC2_ERROR_FRAME_WIDTH_HEIGHT;
} }
if (config->compute_queue == nullptr) { if (require_compute_queue && config->compute_queue == nullptr) {
return VIDEODEC2_ERROR_CONFIG_INFO; return VIDEODEC2_ERROR_COMPUTE_QUEUE;
} }
return OK; return OK;
@@ -243,7 +352,6 @@ static int32_t KYTY_SYSV_ABI AllocateComputeQueue(
} }
*compute_queue = compute_memory_info->cpu_gpu_memory; *compute_queue = compute_memory_info->cpu_gpu_memory;
return OK; return OK;
} }
@@ -266,7 +374,7 @@ static int32_t KYTY_SYSV_ABI QueryDecoderMemoryInfo(const Videodec2DecoderConfig
return VIDEODEC2_ERROR_STRUCT_SIZE; return VIDEODEC2_ERROR_STRUCT_SIZE;
} }
const auto validation_result = ValidateDecoderConfig(config); const auto validation_result = ValidateDecoderConfig(config, false);
if (validation_result != OK) { if (validation_result != OK) {
return validation_result; return validation_result;
} }
@@ -298,7 +406,7 @@ static int32_t KYTY_SYSV_ABI CreateDecoder(const Videodec2DecoderConfigInfo* con
return VIDEODEC2_ERROR_STRUCT_SIZE; return VIDEODEC2_ERROR_STRUCT_SIZE;
} }
const auto validation_result = ValidateDecoderConfig(config); const auto validation_result = ValidateDecoderConfig(config, true);
if (validation_result != OK) { if (validation_result != OK) {
return validation_result; return validation_result;
} }
@@ -315,9 +423,11 @@ static int32_t KYTY_SYSV_ABI CreateDecoder(const Videodec2DecoderConfigInfo* con
return VIDEODEC2_ERROR_MEMORY_POINTER; return VIDEODEC2_ERROR_MEMORY_POINTER;
} }
auto* state = new DecoderState {}; auto* state =
state->magic = DECODER_MAGIC; Decoder::Create({config->codec_type, config->max_frame_width, config->max_frame_height});
state->codec_type = config->codec_type; if (state == nullptr) {
return VIDEODEC2_ERROR_API_FAIL;
}
{ {
std::scoped_lock lock(g_decoder_mutex); std::scoped_lock lock(g_decoder_mutex);
@@ -325,7 +435,6 @@ static int32_t KYTY_SYSV_ABI CreateDecoder(const Videodec2DecoderConfigInfo* con
} }
*decoder = state; *decoder = state;
return OK; return OK;
} }
@@ -343,7 +452,7 @@ static int32_t KYTY_SYSV_ABI DeleteDecoder(Videodec2Decoder decoder) {
g_decoders.erase(it); g_decoders.erase(it);
} }
delete state; Decoder::Destroy(state);
return OK; return OK;
} }
@@ -353,8 +462,8 @@ static int32_t KYTY_SYSV_ABI Decode(Videodec2Decoder decoder, const Videodec2Inp
Videodec2OutputInfo* output_info) { Videodec2OutputInfo* output_info) {
PRINT_NAME(); PRINT_NAME();
const auto* state = GetDecoder(decoder); auto* state = GetDecoder(decoder);
if (state == nullptr || state->magic != DECODER_MAGIC) { if (state == nullptr) {
return VIDEODEC2_ERROR_DECODER_INSTANCE; return VIDEODEC2_ERROR_DECODER_INSTANCE;
} }
@@ -368,8 +477,12 @@ static int32_t KYTY_SYSV_ABI Decode(Videodec2Decoder decoder, const Videodec2Inp
return VIDEODEC2_ERROR_STRUCT_SIZE; return VIDEODEC2_ERROR_STRUCT_SIZE;
} }
if (input_data->au_size != 0 && input_data->au_data == nullptr) { if (input_data->au_size == 0) {
return VIDEODEC2_ERROR_ARGUMENT_POINTER; return VIDEODEC2_ERROR_ACCESS_UNIT_SIZE;
}
if (input_data->au_data == nullptr) {
return VIDEODEC2_ERROR_ACCESS_UNIT_POINTER;
} }
if (frame_buffer->frame_buffer_size == 0) { if (frame_buffer->frame_buffer_size == 0) {
@@ -381,17 +494,24 @@ static int32_t KYTY_SYSV_ABI Decode(Videodec2Decoder decoder, const Videodec2Inp
} }
frame_buffer->is_accepted = false; frame_buffer->is_accepted = false;
FillNoPictureOutput(frame_buffer, output_info, state->codec_type); FillNoPictureOutput(frame_buffer, output_info, Decoder::GetCodecType(state));
return OK; Decoder::Output decoded {};
const auto result =
Decoder::Decode(state,
{input_data->au_data, input_data->au_size, input_data->pts_data,
input_data->dts_data, input_data->attached_data},
{frame_buffer->frame_buffer, frame_buffer->frame_buffer_size}, &decoded);
ApplyDecodedOutput(decoded, frame_buffer, output_info);
return MapDecoderResult(result);
} }
static int32_t KYTY_SYSV_ABI Flush(Videodec2Decoder decoder, Videodec2FrameBuffer* frame_buffer, static int32_t KYTY_SYSV_ABI Flush(Videodec2Decoder decoder, Videodec2FrameBuffer* frame_buffer,
Videodec2OutputInfo* output_info) { Videodec2OutputInfo* output_info) {
PRINT_NAME(); PRINT_NAME();
const auto* state = GetDecoder(decoder); auto* state = GetDecoder(decoder);
if (state == nullptr || state->magic != DECODER_MAGIC) { if (state == nullptr) {
return VIDEODEC2_ERROR_DECODER_INSTANCE; return VIDEODEC2_ERROR_DECODER_INSTANCE;
} }
@@ -404,26 +524,40 @@ static int32_t KYTY_SYSV_ABI Flush(Videodec2Decoder decoder, Videodec2FrameBuffe
return VIDEODEC2_ERROR_STRUCT_SIZE; return VIDEODEC2_ERROR_STRUCT_SIZE;
} }
frame_buffer->is_accepted = false; if (frame_buffer->frame_buffer_size == 0) {
FillNoPictureOutput(frame_buffer, output_info, state->codec_type); return VIDEODEC2_ERROR_FRAME_BUFFER_SIZE;
}
return OK; if (frame_buffer->frame_buffer == nullptr) {
return VIDEODEC2_ERROR_FRAME_BUFFER_POINTER;
}
frame_buffer->is_accepted = false;
FillNoPictureOutput(frame_buffer, output_info, Decoder::GetCodecType(state));
Decoder::Output decoded {};
const auto result = Decoder::Flush(
state, {frame_buffer->frame_buffer, frame_buffer->frame_buffer_size}, &decoded);
ApplyDecodedOutput(decoded, frame_buffer, output_info);
return MapDecoderResult(result);
} }
static int32_t KYTY_SYSV_ABI Reset(Videodec2Decoder decoder) { static int32_t KYTY_SYSV_ABI Reset(Videodec2Decoder decoder) {
PRINT_NAME(); PRINT_NAME();
const auto* state = GetDecoder(decoder); auto* state = GetDecoder(decoder);
return state != nullptr && state->magic == DECODER_MAGIC ? OK if (state == nullptr) {
: VIDEODEC2_ERROR_DECODER_INSTANCE; return VIDEODEC2_ERROR_DECODER_INSTANCE;
}
Decoder::Reset(state);
return OK;
} }
static int32_t KYTY_SYSV_ABI GetPictureInfo(const Videodec2OutputInfo* output_info, static int32_t KYTY_SYSV_ABI GetPictureInfo(const Videodec2OutputInfo* output_info,
void* /*first_picture_info*/, void* first_picture_info, void* second_picture_info) {
void* /*second_picture_info*/) {
PRINT_NAME(); PRINT_NAME();
if (output_info == nullptr) { if (output_info == nullptr || first_picture_info == nullptr) {
return VIDEODEC2_ERROR_ARGUMENT_POINTER; return VIDEODEC2_ERROR_ARGUMENT_POINTER;
} }
@@ -431,6 +565,88 @@ static int32_t KYTY_SYSV_ABI GetPictureInfo(const Videodec2OutputInfo* output_in
return VIDEODEC2_ERROR_STRUCT_SIZE; return VIDEODEC2_ERROR_STRUCT_SIZE;
} }
if (!output_info->is_valid || output_info->picture_count == 0 ||
output_info->frame_buffer == nullptr) {
return VIDEODEC2_ERROR_OUTPUT_INFO;
}
Decoder::PictureInfo decoded {};
if (!Decoder::GetPictureInfo(output_info->frame_buffer, &decoded) ||
decoded.codec_type != output_info->codec_type) {
return VIDEODEC2_ERROR_OUTPUT_INFO;
}
auto fill_common = [&decoded](void* destination, bool valid) -> int32_t {
auto* bytes = static_cast<uint8_t*>(destination);
const auto size = *static_cast<const size_t*>(destination);
if (size < 40 || size > 256) {
return VIDEODEC2_ERROR_STRUCT_SIZE;
}
std::memset(bytes + sizeof(size_t), 0, size - sizeof(size_t));
bytes[8] = valid ? 1 : 0;
if (valid) {
std::memcpy(bytes + 16, &decoded.pts, sizeof(decoded.pts));
std::memcpy(bytes + 24, &decoded.dts, sizeof(decoded.dts));
std::memcpy(bytes + 32, &decoded.attached_data, sizeof(decoded.attached_data));
}
return OK;
};
if (output_info->codec_type == 1) {
const auto requested_size = *static_cast<const size_t*>(first_picture_info);
if (requested_size != sizeof(Videodec2AvcPictureInfo) &&
(requested_size | 16u) != sizeof(Videodec2AvcPictureInfo)) {
return VIDEODEC2_ERROR_STRUCT_SIZE;
}
Videodec2AvcPictureInfo picture {};
picture.this_size = requested_size;
picture.is_valid = true;
picture.pts_data = decoded.pts;
picture.dts_data = decoded.dts;
picture.attached_data = decoded.attached_data;
picture.idr_picture_flag = decoded.key_frame ? 1 : 0;
picture.profile_idc = static_cast<uint8_t>(decoded.profile);
picture.level_idc = static_cast<uint8_t>(decoded.level);
picture.pic_width_in_mbs_minus1 = (decoded.width + 15u) / 16u - 1u;
picture.pic_height_in_map_units_minus1 = (decoded.height + 15u) / 16u - 1u;
picture.frame_mbs_only_flag = 1;
picture.frame_cropping_flag = decoded.crop_left != 0 || decoded.crop_right != 0 ||
decoded.crop_top != 0 || decoded.crop_bottom != 0
? 1
: 0;
picture.frame_crop_left_offset = decoded.crop_left;
picture.frame_crop_right_offset = decoded.crop_right;
picture.frame_crop_top_offset = decoded.crop_top;
picture.frame_crop_bottom_offset = decoded.crop_bottom;
picture.aspect_ratio_info_present_flag =
decoded.sar_width != 0 && decoded.sar_height != 0 ? 1 : 0;
picture.aspect_ratio_idc = picture.aspect_ratio_info_present_flag ? 255 : 0;
picture.sar_width = decoded.sar_width;
picture.sar_height = decoded.sar_height;
picture.video_signal_type_present_flag = 1;
picture.video_format = 5;
picture.video_full_range_flag = decoded.color_range == 2 ? 1 : 0;
picture.colour_description_present_flag =
decoded.color_primaries != 0 || decoded.color_trc != 0 || decoded.color_space != 0 ? 1
: 0;
picture.colour_primaries = decoded.color_primaries;
picture.transfer_characteristics = decoded.color_trc;
picture.matrix_coefficients = decoded.color_space;
std::memcpy(first_picture_info, &picture, requested_size);
} else {
const auto result = fill_common(first_picture_info, true);
if (result != OK) {
return result;
}
}
if (second_picture_info != nullptr) {
const auto result = fill_common(second_picture_info, false);
if (result != OK) {
return result;
}
}
return OK; return OK;
} }
+412
View File
@@ -0,0 +1,412 @@
#include "libs/videoDec2Decoder.h"
#include "common/logging/log.h"
#include <algorithm>
#include <cstddef>
#include <cstdint>
#include <cstring>
#include <limits>
#include <mutex>
#include <unordered_map>
#include <unordered_set>
extern "C" {
#include <libavcodec/avcodec.h>
#include <libavutil/buffer.h>
#include <libavutil/error.h>
#include <libavutil/frame.h>
#include <libavutil/pixfmt.h>
#include <libswscale/swscale.h>
}
namespace Libs::VideoDec2::Decoder {
namespace {
constexpr uint32_t CODEC_TYPE_AVC = 1;
constexpr uint32_t CODEC_TYPE_HEVC = 974921;
constexpr uint32_t CODEC_TYPE_VP9 = 2382845;
struct PacketMetadata {
uint64_t pts = TIMESTAMP_INVALID;
uint64_t dts = TIMESTAMP_INVALID;
uint64_t attached_data = 0;
};
struct StoredPicture {
const Instance* owner = nullptr;
PictureInfo info;
};
std::mutex g_picture_mutex;
std::unordered_map<void*, StoredPicture> g_picture_infos;
AVCodecID GetAvCodecId(uint32_t codec_type) {
switch (codec_type) {
case CODEC_TYPE_AVC: return AV_CODEC_ID_H264;
case CODEC_TYPE_HEVC: return AV_CODEC_ID_HEVC;
case CODEC_TYPE_VP9: return AV_CODEC_ID_VP9;
default: return AV_CODEC_ID_NONE;
}
}
const char* AvErrorString(int error) {
thread_local char text[AV_ERROR_MAX_STRING_SIZE] {};
if (av_strerror(error, text, sizeof(text)) != 0) {
std::strcpy(text, "unknown FFmpeg error");
}
return text;
}
uint32_t AlignUp(uint32_t value, uint32_t alignment) {
return (value + alignment - 1u) & ~(alignment - 1u);
}
int64_t ToAvTimestamp(uint64_t timestamp) {
return timestamp == TIMESTAMP_INVALID ||
timestamp > static_cast<uint64_t>(std::numeric_limits<int64_t>::max())
? AV_NOPTS_VALUE
: static_cast<int64_t>(timestamp);
}
} // namespace
class Instance {
public:
explicit Instance(const Config& config): m_config(config) {}
~Instance() {
ClearPictureMetadata();
if (m_sws != nullptr) {
sws_freeContext(m_sws);
}
if (m_codec != nullptr) {
avcodec_free_context(&m_codec);
}
}
Instance(const Instance&) = delete;
Instance& operator=(const Instance&) = delete;
[[nodiscard]] bool Initialize() {
const AVCodec* decoder = avcodec_find_decoder(GetAvCodecId(m_config.codec_type));
if (decoder == nullptr) {
LOGF("Videodec2: FFmpeg decoder is unavailable for codec type %u\n",
m_config.codec_type);
return false;
}
m_codec = avcodec_alloc_context3(decoder);
if (m_codec == nullptr) {
LOGF("Videodec2: avcodec_alloc_context3 failed\n");
return false;
}
// This carries PTS/DTS/attachedData through codecs that reorder B frames.
m_codec->flags |= AV_CODEC_FLAG_COPY_OPAQUE;
const int result = avcodec_open2(m_codec, decoder, nullptr);
if (result < 0) {
LOGF("Videodec2: avcodec_open2 failed: %s (%d)\n", AvErrorString(result), result);
return false;
}
return true;
}
[[nodiscard]] uint32_t CodecType() const { return m_config.codec_type; }
[[nodiscard]] Result DecodeInput(const Input& input, const FrameBuffer& frame_buffer,
Output* output) {
std::scoped_lock lock(m_mutex);
*output = {};
m_draining = false;
AVPacket* packet = av_packet_alloc();
AVFrame* frame = av_frame_alloc();
if (packet == nullptr || frame == nullptr ||
input.size > static_cast<size_t>(std::numeric_limits<int>::max())) {
av_packet_free(&packet);
av_frame_free(&frame);
return Result::ApiFail;
}
int result = av_new_packet(packet, static_cast<int>(input.size));
if (result < 0) {
LOGF("Videodec2: av_new_packet failed: %s (%d)\n", AvErrorString(result), result);
av_packet_free(&packet);
av_frame_free(&frame);
return Result::ApiFail;
}
std::memcpy(packet->data, input.data, input.size);
packet->pts = ToAvTimestamp(input.pts);
packet->dts = ToAvTimestamp(input.dts);
packet->opaque_ref = av_buffer_alloc(sizeof(PacketMetadata));
if (packet->opaque_ref == nullptr) {
av_packet_free(&packet);
av_frame_free(&frame);
return Result::ApiFail;
}
const PacketMetadata metadata {input.pts, input.dts, input.attached_data};
std::memcpy(packet->opaque_ref->data, &metadata, sizeof(metadata));
bool have_pending_frame = false;
result = avcodec_send_packet(m_codec, packet);
if (result == AVERROR(EAGAIN)) {
result = avcodec_receive_frame(m_codec, frame);
if (result < 0) {
LOGF("Videodec2: decoder rejected an AU while no output was available: %s (%d)\n",
AvErrorString(result), result);
av_packet_free(&packet);
av_frame_free(&frame);
return Result::AccessUnit;
}
have_pending_frame = true;
result = avcodec_send_packet(m_codec, packet);
}
if (result < 0) {
LOGF("Videodec2: avcodec_send_packet failed: %s (%d)\n", AvErrorString(result), result);
av_packet_free(&packet);
av_frame_free(&frame);
return Result::AccessUnit;
}
Result decode_result = Result::Ok;
if (!have_pending_frame) {
result = avcodec_receive_frame(m_codec, frame);
if (result != AVERROR(EAGAIN) && result != AVERROR_EOF) {
if (result < 0) {
LOGF("Videodec2: avcodec_receive_frame failed: %s (%d)\n",
AvErrorString(result), result);
decode_result = Result::AccessUnit;
} else {
decode_result = CopyFrame(frame, frame_buffer, output);
}
}
} else {
decode_result = CopyFrame(frame, frame_buffer, output);
}
av_packet_free(&packet);
av_frame_free(&frame);
return decode_result;
}
[[nodiscard]] Result FlushOutput(const FrameBuffer& frame_buffer, Output* output) {
std::scoped_lock lock(m_mutex);
*output = {};
AVFrame* frame = av_frame_alloc();
if (frame == nullptr) {
return Result::ApiFail;
}
if (!m_draining) {
const int send_result = avcodec_send_packet(m_codec, nullptr);
if (send_result == 0 || send_result == AVERROR_EOF) {
m_draining = true;
} else if (send_result != AVERROR(EAGAIN)) {
LOGF("Videodec2: flushing decoder failed: %s (%d)\n", AvErrorString(send_result),
send_result);
av_frame_free(&frame);
return Result::ApiFail;
}
}
const int receive_result = avcodec_receive_frame(m_codec, frame);
if (receive_result == AVERROR(EAGAIN) || receive_result == AVERROR_EOF) {
av_frame_free(&frame);
return Result::Ok;
}
if (receive_result < 0) {
LOGF("Videodec2: receiving a flushed frame failed: %s (%d)\n",
AvErrorString(receive_result), receive_result);
av_frame_free(&frame);
return Result::ApiFail;
}
const auto result = CopyFrame(frame, frame_buffer, output);
av_frame_free(&frame);
return result;
}
void ResetDecoder() {
std::scoped_lock lock(m_mutex);
avcodec_flush_buffers(m_codec);
m_draining = false;
ClearPictureMetadata();
}
private:
[[nodiscard]] PictureInfo MakePictureInfo(const AVFrame* frame) const {
PictureInfo result {};
if (frame->opaque_ref != nullptr && frame->opaque_ref->size >= sizeof(PacketMetadata)) {
PacketMetadata metadata {};
std::memcpy(&metadata, frame->opaque_ref->data, sizeof(metadata));
result.pts = metadata.pts;
result.dts = metadata.dts;
result.attached_data = metadata.attached_data;
} else {
result.pts = frame->pts == AV_NOPTS_VALUE ? TIMESTAMP_INVALID
: static_cast<uint64_t>(frame->pts);
result.dts = frame->pkt_dts == AV_NOPTS_VALUE ? TIMESTAMP_INVALID
: static_cast<uint64_t>(frame->pkt_dts);
}
result.codec_type = m_config.codec_type;
result.width = static_cast<uint32_t>(frame->width);
result.height = static_cast<uint32_t>(frame->height);
result.crop_left = static_cast<uint32_t>(frame->crop_left);
result.crop_right = static_cast<uint32_t>(frame->crop_right);
result.crop_top = static_cast<uint32_t>(frame->crop_top);
result.crop_bottom = static_cast<uint32_t>(frame->crop_bottom);
result.profile = m_codec->profile > 0 ? static_cast<uint32_t>(m_codec->profile) : 0;
result.level = m_codec->level > 0 ? static_cast<uint32_t>(m_codec->level) : 0;
result.sar_width =
frame->sample_aspect_ratio.num > 0
? static_cast<uint16_t>(std::min(frame->sample_aspect_ratio.num, 65535))
: 0;
result.sar_height =
frame->sample_aspect_ratio.den > 0
? static_cast<uint16_t>(std::min(frame->sample_aspect_ratio.den, 65535))
: 0;
result.color_range = static_cast<uint8_t>(frame->color_range);
result.color_primaries = static_cast<uint8_t>(frame->color_primaries);
result.color_trc = static_cast<uint8_t>(frame->color_trc);
result.color_space = static_cast<uint8_t>(frame->colorspace);
result.key_frame = (frame->flags & AV_FRAME_FLAG_KEY) != 0;
return result;
}
[[nodiscard]] Result CopyFrame(const AVFrame* frame, const FrameBuffer& frame_buffer,
Output* output) {
if (frame->width <= 0 || frame->height <= 0) {
return Result::ApiFail;
}
if ((m_config.max_width > 0 && frame->width > m_config.max_width) ||
(m_config.max_height > 0 && frame->height > m_config.max_height)) {
return Result::OversizeDecode;
}
const auto width = static_cast<uint32_t>(frame->width);
const auto height = static_cast<uint32_t>(frame->height);
const auto pitch = AlignUp(width, 256);
const auto chroma_rows = (static_cast<uint64_t>(height) + 1u) / 2u;
const auto required =
static_cast<uint64_t>(pitch) * height + static_cast<uint64_t>(pitch) * chroma_rows;
if (required > frame_buffer.size) {
return Result::FrameBufferSize;
}
auto* dst = static_cast<uint8_t*>(frame_buffer.data);
std::memset(dst, 0, static_cast<size_t>(required));
if (frame->format == AV_PIX_FMT_NV12) {
for (uint32_t y = 0; y < height; y++) {
std::memcpy(dst + static_cast<size_t>(y) * pitch,
frame->data[0] + static_cast<ptrdiff_t>(y) * frame->linesize[0], width);
}
auto* chroma = dst + static_cast<size_t>(pitch) * height;
for (uint32_t y = 0; y < chroma_rows; y++) {
std::memcpy(chroma + static_cast<size_t>(y) * pitch,
frame->data[1] + static_cast<ptrdiff_t>(y) * frame->linesize[1], width);
}
} else {
m_sws = sws_getCachedContext(m_sws, frame->width, frame->height,
static_cast<AVPixelFormat>(frame->format), frame->width,
frame->height, AV_PIX_FMT_NV12, SWS_FAST_BILINEAR, nullptr,
nullptr, nullptr);
if (m_sws == nullptr) {
return Result::ApiFail;
}
uint8_t* output_planes[4] = {dst, dst + static_cast<size_t>(pitch) * height, nullptr,
nullptr};
int output_strides[4] = {static_cast<int>(pitch), static_cast<int>(pitch), 0, 0};
if (sws_scale(m_sws, frame->data, frame->linesize, 0, frame->height, output_planes,
output_strides) != frame->height) {
return Result::ApiFail;
}
}
output->valid = true;
output->error_frame = (frame->flags & AV_FRAME_FLAG_CORRUPT) != 0;
output->buffer_accepted = true;
output->codec_type = m_config.codec_type;
output->width = width;
output->pitch = pitch;
output->height = height;
output->buffer = frame_buffer.data;
output->buffer_size = frame_buffer.size;
{
std::scoped_lock lock(g_picture_mutex);
g_picture_infos[frame_buffer.data] = {this, MakePictureInfo(frame)};
m_picture_buffers.insert(frame_buffer.data);
}
return Result::Ok;
}
void ClearPictureMetadata() {
std::scoped_lock lock(g_picture_mutex);
for (auto* buffer: m_picture_buffers) {
const auto it = g_picture_infos.find(buffer);
if (it != g_picture_infos.end() && it->second.owner == this) {
g_picture_infos.erase(it);
}
}
m_picture_buffers.clear();
}
Config m_config;
AVCodecContext* m_codec = nullptr;
SwsContext* m_sws = nullptr;
bool m_draining = false;
std::mutex m_mutex;
std::unordered_set<void*> m_picture_buffers;
};
bool IsCodecSupported(uint32_t codec_type) {
return GetAvCodecId(codec_type) != AV_CODEC_ID_NONE;
}
Instance* Create(const Config& config) {
if (!IsCodecSupported(config.codec_type)) {
return nullptr;
}
auto* instance = new Instance(config);
if (!instance->Initialize()) {
delete instance;
return nullptr;
}
return instance;
}
void Destroy(Instance* instance) {
delete instance;
}
uint32_t GetCodecType(const Instance* instance) {
return instance->CodecType();
}
Result Decode(Instance* instance, const Input& input, const FrameBuffer& frame_buffer,
Output* output) {
return instance->DecodeInput(input, frame_buffer, output);
}
Result Flush(Instance* instance, const FrameBuffer& frame_buffer, Output* output) {
return instance->FlushOutput(frame_buffer, output);
}
void Reset(Instance* instance) {
instance->ResetDecoder();
}
bool GetPictureInfo(void* frame_buffer, PictureInfo* picture_info) {
std::scoped_lock lock(g_picture_mutex);
const auto it = g_picture_infos.find(frame_buffer);
if (it == g_picture_infos.end()) {
return false;
}
*picture_info = it->second.info;
return true;
}
} // namespace Libs::VideoDec2::Decoder
+86
View File
@@ -0,0 +1,86 @@
#ifndef EMULATOR_INCLUDE_EMULATOR_LIBS_VIDEODEC2DECODER_H_
#define EMULATOR_INCLUDE_EMULATOR_LIBS_VIDEODEC2DECODER_H_
#include <cstddef>
#include <cstdint>
namespace Libs::VideoDec2::Decoder {
constexpr uint64_t TIMESTAMP_INVALID = UINT64_MAX;
enum class Result {
Ok,
ApiFail,
AccessUnit,
FrameBufferSize,
OversizeDecode,
};
struct Config {
uint32_t codec_type = 0;
int32_t max_width = -1;
int32_t max_height = -1;
};
struct Input {
const void* data = nullptr;
size_t size = 0;
uint64_t pts = TIMESTAMP_INVALID;
uint64_t dts = TIMESTAMP_INVALID;
uint64_t attached_data = 0;
};
struct FrameBuffer {
void* data = nullptr;
size_t size = 0;
};
struct Output {
bool valid = false;
bool error_frame = false;
bool buffer_accepted = false;
uint32_t codec_type = 0;
uint32_t width = 0;
uint32_t pitch = 0;
uint32_t height = 0;
void* buffer = nullptr;
size_t buffer_size = 0;
};
struct PictureInfo {
uint64_t pts = TIMESTAMP_INVALID;
uint64_t dts = TIMESTAMP_INVALID;
uint64_t attached_data = 0;
uint32_t codec_type = 0;
uint32_t width = 0;
uint32_t height = 0;
uint32_t crop_left = 0;
uint32_t crop_right = 0;
uint32_t crop_top = 0;
uint32_t crop_bottom = 0;
uint32_t profile = 0;
uint32_t level = 0;
uint16_t sar_width = 0;
uint16_t sar_height = 0;
uint8_t color_range = 0;
uint8_t color_primaries = 0;
uint8_t color_trc = 0;
uint8_t color_space = 0;
bool key_frame = false;
};
class Instance;
[[nodiscard]] bool IsCodecSupported(uint32_t codec_type);
[[nodiscard]] Instance* Create(const Config& config);
void Destroy(Instance* instance);
[[nodiscard]] uint32_t GetCodecType(const Instance* instance);
[[nodiscard]] Result Decode(Instance* instance, const Input& input, const FrameBuffer& frame_buffer,
Output* output);
[[nodiscard]] Result Flush(Instance* instance, const FrameBuffer& frame_buffer, Output* output);
void Reset(Instance* instance);
[[nodiscard]] bool GetPictureInfo(void* frame_buffer, PictureInfo* picture_info);
} // namespace Libs::VideoDec2::Decoder
#endif // EMULATOR_INCLUDE_EMULATOR_LIBS_VIDEODEC2DECODER_H_
+27 -25
View File
@@ -79,10 +79,10 @@ static void Sha1Msg1(XmmWords& dest, const XmmWords& src2) {
const uint32_t w3 = dest.w[0]; const uint32_t w3 = dest.w[0];
const uint32_t w4 = src2.w[3]; const uint32_t w4 = src2.w[3];
const uint32_t w5 = src2.w[2]; const uint32_t w5 = src2.w[2];
dest.w[3] = w2 ^ w0; dest.w[3] = w2 ^ w0;
dest.w[2] = w3 ^ w1; dest.w[2] = w3 ^ w1;
dest.w[1] = w4 ^ w2; dest.w[1] = w4 ^ w2;
dest.w[0] = w5 ^ w3; dest.w[0] = w5 ^ w3;
} }
static void Sha1Msg2(XmmWords& dest, const XmmWords& src2) { static void Sha1Msg2(XmmWords& dest, const XmmWords& src2) {
@@ -93,18 +93,18 @@ static void Sha1Msg2(XmmWords& dest, const XmmWords& src2) {
const uint32_t w17 = Rol32(dest.w[2] ^ w14, 1u); const uint32_t w17 = Rol32(dest.w[2] ^ w14, 1u);
const uint32_t w18 = Rol32(dest.w[1] ^ w15, 1u); const uint32_t w18 = Rol32(dest.w[1] ^ w15, 1u);
const uint32_t w19 = Rol32(dest.w[0] ^ w16, 1u); const uint32_t w19 = Rol32(dest.w[0] ^ w16, 1u);
dest.w[3] = w16; dest.w[3] = w16;
dest.w[2] = w17; dest.w[2] = w17;
dest.w[1] = w18; dest.w[1] = w18;
dest.w[0] = w19; dest.w[0] = w19;
} }
static void Sha1Nexte(XmmWords& dest, const XmmWords& src2) { static void Sha1Nexte(XmmWords& dest, const XmmWords& src2) {
const uint32_t tmp = Rol32(dest.w[3], 30u); const uint32_t tmp = Rol32(dest.w[3], 30u);
dest.w[3] = src2.w[3] + tmp; dest.w[3] = src2.w[3] + tmp;
dest.w[2] = src2.w[2]; dest.w[2] = src2.w[2];
dest.w[1] = src2.w[1]; dest.w[1] = src2.w[1];
dest.w[0] = src2.w[0]; dest.w[0] = src2.w[0];
} }
static uint32_t Sha1RoundFunc(uint8_t group, uint32_t b, uint32_t c, uint32_t d) { static uint32_t Sha1RoundFunc(uint8_t group, uint32_t b, uint32_t c, uint32_t d) {
@@ -185,10 +185,10 @@ static void Sha256Msg1(XmmWords& dest, const XmmWords& src2) {
const uint32_t w2 = dest.w[2]; const uint32_t w2 = dest.w[2];
const uint32_t w1 = dest.w[1]; const uint32_t w1 = dest.w[1];
const uint32_t w0 = dest.w[0]; const uint32_t w0 = dest.w[0];
dest.w[3] = w3 + Sha256Sigma0(w4); dest.w[3] = w3 + Sha256Sigma0(w4);
dest.w[2] = w2 + Sha256Sigma0(w3); dest.w[2] = w2 + Sha256Sigma0(w3);
dest.w[1] = w1 + Sha256Sigma0(w2); dest.w[1] = w1 + Sha256Sigma0(w2);
dest.w[0] = w0 + Sha256Sigma0(w1); dest.w[0] = w0 + Sha256Sigma0(w1);
} }
static void Sha256Msg2(XmmWords& dest, const XmmWords& src2) { static void Sha256Msg2(XmmWords& dest, const XmmWords& src2) {
@@ -312,7 +312,9 @@ static bool DecodeShaNiInsn(const uint8_t* rip, ShaNiInsn& insn) {
return true; return true;
} }
static bool ShaNiModrmIsRegister(uint8_t modrm) { return (modrm & 0xc0u) == 0xc0u; } static bool ShaNiModrmIsRegister(uint8_t modrm) {
return (modrm & 0xc0u) == 0xc0u;
}
static uint8_t ShaNiRegIndex(uint8_t modrm, uint8_t rex, bool reg_field) { static uint8_t ShaNiRegIndex(uint8_t modrm, uint8_t rex, bool reg_field) {
if (reg_field) { if (reg_field) {
@@ -321,7 +323,7 @@ static uint8_t ShaNiRegIndex(uint8_t modrm, uint8_t rex, bool reg_field) {
return (modrm & 0x07u) | ((rex & 0x01u) << 3u); return (modrm & 0x07u) | ((rex & 0x01u) << 3u);
} }
static bool ResolveShaNiMemoryAddress(const uint8_t* rip, const ShaNiInsn& insn, static bool ResolveShaNiMemoryAddress(const uint8_t* rip, const ShaNiInsn& insn,
const uint64_t (&gpr)[16], const void*& address) { const uint64_t (&gpr)[16], const void*& address) {
const uint8_t modrm = rip[insn.modrm_offset]; const uint8_t modrm = rip[insn.modrm_offset];
const uint8_t mod = modrm >> 6u; const uint8_t mod = modrm >> 6u;
@@ -334,12 +336,12 @@ static bool ResolveShaNiMemoryAddress(const uint8_t* rip, const ShaNiInsn& insn,
uint64_t result = 0; uint64_t result = 0;
if (rm == 4u) { if (rm == 4u) {
const uint8_t sib = rip[offset++]; const uint8_t sib = rip[offset++];
const uint8_t scale = sib >> 6u; const uint8_t scale = sib >> 6u;
const uint8_t index_low = (sib >> 3u) & 0x07u; const uint8_t index_low = (sib >> 3u) & 0x07u;
const uint8_t base_low = sib & 0x07u; const uint8_t base_low = sib & 0x07u;
const bool has_index = index_low != 4u || (insn.rex & 0x02u) != 0; const bool has_index = index_low != 4u || (insn.rex & 0x02u) != 0;
const bool has_base = mod != 0u || base_low != 5u; const bool has_base = mod != 0u || base_low != 5u;
if (has_base) { if (has_base) {
const uint8_t base = base_low | ((insn.rex & 0x01u) << 3u); const uint8_t base = base_low | ((insn.rex & 0x01u) << 3u);
+339 -396
View File
@@ -1,5 +1,4 @@
#include "kernel/eventQueue.h" #include "kernel/eventQueue.h"
#include "libs/errno.h" #include "libs/errno.h"
#include <algorithm> #include <algorithm>
@@ -17,484 +16,428 @@ namespace EventQueue = Libs::LibKernel::EventQueue;
using Libs::LibKernel::KERNEL_ERROR_EBADF; using Libs::LibKernel::KERNEL_ERROR_EBADF;
using Libs::LibKernel::KERNEL_ERROR_ENOENT; using Libs::LibKernel::KERNEL_ERROR_ENOENT;
void Check(bool value, const char *text) { void Check(bool value, const char* text) {
if (!value) { if (!value) {
std::fprintf(stderr, "EventQueueLifetimeTests: failed: %s\n", text); std::fprintf(stderr, "EventQueueLifetimeTests: failed: %s\n", text);
std::abort(); std::abort();
} }
} }
void CheckConcurrentResult(int result, const char *text) { void CheckConcurrentResult(int result, const char* text) {
Check(result == OK || result == KERNEL_ERROR_EBADF || Check(result == OK || result == KERNEL_ERROR_EBADF || result == KERNEL_ERROR_ENOENT, text);
result == KERNEL_ERROR_ENOENT,
text);
} }
void CountDeletedEvent(EventQueue::KernelEqueue, void CountDeletedEvent(EventQueue::KernelEqueue, EventQueue::KernelEqueueEvent* event) {
EventQueue::KernelEqueueEvent *event) { auto* count = static_cast<std::atomic_uint32_t*>(event->filter.data);
auto *count = static_cast<std::atomic_uint32_t *>(event->filter.data); count->fetch_add(1, std::memory_order_relaxed);
count->fetch_add(1, std::memory_order_relaxed);
} }
struct DuplicateEventOwner { struct DuplicateEventOwner {
std::atomic_uint32_t delete_count{0}; std::atomic_uint32_t delete_count {0};
}; };
void QueueDuplicateEvent(EventQueue::KernelEqueueEvent *event, void QueueDuplicateEvent(EventQueue::KernelEqueueEvent* event, void* trigger_data) {
void *trigger_data) { auto next = event->event;
auto next = event->event; next.data = reinterpret_cast<intptr_t>(trigger_data);
next.data = reinterpret_cast<intptr_t>(trigger_data); if (event->triggered) {
if (event->triggered) { event->pending_events.push_back(next);
event->pending_events.push_back(next); } else {
} else { event->event = next;
event->event = next; event->triggered = true;
event->triggered = true; }
}
} }
void ResetDuplicateEvent(EventQueue::KernelEqueueEvent *event) { void ResetDuplicateEvent(EventQueue::KernelEqueueEvent* event) {
event->triggered = false; event->triggered = false;
event->event.data = 0; event->event.data = 0;
} }
void DeleteDuplicateEvent(EventQueue::KernelEqueue, void DeleteDuplicateEvent(EventQueue::KernelEqueue, EventQueue::KernelEqueueEvent* event) {
EventQueue::KernelEqueueEvent *event) { auto* owner = static_cast<DuplicateEventOwner*>(event->filter.data);
auto *owner = static_cast<DuplicateEventOwner *>(event->filter.data); owner->delete_count.fetch_add(1, std::memory_order_relaxed);
owner->delete_count.fetch_add(1, std::memory_order_relaxed);
} }
void PoisonDuplicateEvent(EventQueue::KernelEqueueEvent *, void *) { void PoisonDuplicateEvent(EventQueue::KernelEqueueEvent*, void*) {
Check(false, "duplicate add replaced trigger callback"); Check(false, "duplicate add replaced trigger callback");
} }
void TestDuplicateAddPreservesEventState() { void TestDuplicateAddPreservesEventState() {
EventQueue::KernelEqueue queue = EventQueue::KERNEL_EQUEUE_INVALID; EventQueue::KernelEqueue queue = EventQueue::KERNEL_EQUEUE_INVALID;
Check(EventQueue::KernelCreateEqueue(&queue, "duplicate-add") == OK, Check(EventQueue::KernelCreateEqueue(&queue, "duplicate-add") == OK,
"create duplicate add queue"); "create duplicate add queue");
auto original_owner = std::make_shared<DuplicateEventOwner>(); auto original_owner = std::make_shared<DuplicateEventOwner>();
std::weak_ptr<DuplicateEventOwner> weak_original = original_owner; std::weak_ptr<DuplicateEventOwner> weak_original = original_owner;
EventQueue::KernelEqueueEvent original{}; EventQueue::KernelEqueueEvent original {};
original.event.ident = 17; original.event.ident = 17;
original.event.filter = EventQueue::KERNEL_EVFILT_VIDEO_OUT; original.event.filter = EventQueue::KERNEL_EVFILT_VIDEO_OUT;
original.event.udata = reinterpret_cast<void *>(0x1111); original.event.udata = reinterpret_cast<void*>(0x1111);
original.filter.data = original_owner.get(); original.filter.data = original_owner.get();
original.filter.owner = original_owner; original.filter.owner = original_owner;
original.filter.trigger_func = QueueDuplicateEvent; original.filter.trigger_func = QueueDuplicateEvent;
original.filter.reset_func = ResetDuplicateEvent; original.filter.reset_func = ResetDuplicateEvent;
original.filter.delete_event_func = DeleteDuplicateEvent; original.filter.delete_event_func = DeleteDuplicateEvent;
Check(EventQueue::KernelAddEvent(queue, original) == OK, Check(EventQueue::KernelAddEvent(queue, original) == OK, "add original duplicate event");
"add original duplicate event"); Check(EventQueue::KernelTriggerEvent(queue, 17, EventQueue::KERNEL_EVFILT_VIDEO_OUT,
Check(EventQueue::KernelTriggerEvent(queue, 17, reinterpret_cast<void*>(0x1234)) == OK,
EventQueue::KERNEL_EVFILT_VIDEO_OUT, "queue first trigger");
reinterpret_cast<void *>(0x1234)) == OK, Check(EventQueue::KernelTriggerEvent(queue, 17, EventQueue::KERNEL_EVFILT_VIDEO_OUT,
"queue first trigger"); reinterpret_cast<void*>(0x5678)) == OK,
Check(EventQueue::KernelTriggerEvent(queue, 17, "queue pending trigger");
EventQueue::KERNEL_EVFILT_VIDEO_OUT,
reinterpret_cast<void *>(0x5678)) == OK,
"queue pending trigger");
auto replacement_owner = std::make_shared<DuplicateEventOwner>(); auto replacement_owner = std::make_shared<DuplicateEventOwner>();
std::weak_ptr<DuplicateEventOwner> weak_replacement = replacement_owner; std::weak_ptr<DuplicateEventOwner> weak_replacement = replacement_owner;
EventQueue::KernelEqueueEvent duplicate{}; EventQueue::KernelEqueueEvent duplicate {};
duplicate.triggered = false; duplicate.triggered = false;
duplicate.deadline_ns = 1; duplicate.deadline_ns = 1;
duplicate.event.ident = 17; duplicate.event.ident = 17;
duplicate.event.filter = EventQueue::KERNEL_EVFILT_VIDEO_OUT; duplicate.event.filter = EventQueue::KERNEL_EVFILT_VIDEO_OUT;
duplicate.event.data = 0x7fffffff; duplicate.event.data = 0x7fffffff;
duplicate.event.udata = reinterpret_cast<void *>(0x2222); duplicate.event.udata = reinterpret_cast<void*>(0x2222);
duplicate.filter.data = replacement_owner.get(); duplicate.filter.data = replacement_owner.get();
duplicate.filter.owner = replacement_owner; duplicate.filter.owner = replacement_owner;
duplicate.filter.trigger_func = PoisonDuplicateEvent; duplicate.filter.trigger_func = PoisonDuplicateEvent;
Check(EventQueue::KernelAddEvent(queue, duplicate) == OK, Check(EventQueue::KernelAddEvent(queue, duplicate) == OK, "update duplicate event");
"update duplicate event");
duplicate.filter.owner.reset(); duplicate.filter.owner.reset();
replacement_owner.reset(); replacement_owner.reset();
Check(weak_replacement.expired(), "duplicate owner is not retained"); Check(weak_replacement.expired(), "duplicate owner is not retained");
original.filter.owner.reset(); original.filter.owner.reset();
original_owner.reset(); original_owner.reset();
Check(!weak_original.expired(), "original event owner remains retained"); Check(!weak_original.expired(), "original event owner remains retained");
EventQueue::KernelEvent events[2]{}; EventQueue::KernelEvent events[2] {};
int out = 0; int out = 0;
Libs::LibKernel::KernelUseconds timeout = 0; Libs::LibKernel::KernelUseconds timeout = 0;
Check(EventQueue::KernelWaitEqueue(queue, events, 2, &out, &timeout) == OK, Check(EventQueue::KernelWaitEqueue(queue, events, 2, &out, &timeout) == OK,
"read queued duplicate triggers"); "read queued duplicate triggers");
Check(out == 2, "duplicate add preserves pending event count"); Check(out == 2, "duplicate add preserves pending event count");
Check(events[0].data == 0x1234 && events[1].data == 0x5678, Check(events[0].data == 0x1234 && events[1].data == 0x5678,
"duplicate add preserves current and pending event data"); "duplicate add preserves current and pending event data");
Check(events[0].udata == reinterpret_cast<void *>(0x2222), Check(events[0].udata == reinterpret_cast<void*>(0x2222),
"duplicate add updates current user data"); "duplicate add updates current user data");
Check(events[1].udata == reinterpret_cast<void *>(0x2222), Check(events[1].udata == reinterpret_cast<void*>(0x2222),
"duplicate add updates pending user data"); "duplicate add updates pending user data");
EventQueue::KernelEvent timer_event{}; EventQueue::KernelEvent timer_event {};
Check(EventQueue::KernelWaitEqueue(queue, &timer_event, 1, &out, &timeout) == Check(EventQueue::KernelWaitEqueue(queue, &timer_event, 1, &out, &timeout) == OK && out == 1,
OK && "duplicate add updates deadline metadata");
out == 1, Check(timer_event.data == 0 && timer_event.udata == reinterpret_cast<void*>(0x2222),
"duplicate add updates deadline metadata"); "deadline trigger retains updated duplicate metadata");
Check(timer_event.data == 0 &&
timer_event.udata == reinterpret_cast<void *>(0x2222),
"deadline trigger retains updated duplicate metadata");
auto retained_owner = weak_original.lock(); auto retained_owner = weak_original.lock();
Check(retained_owner != nullptr, "original owner alive before delete"); Check(retained_owner != nullptr, "original owner alive before delete");
Check(EventQueue::KernelDeleteEvent(queue, 17, Check(EventQueue::KernelDeleteEvent(queue, 17, EventQueue::KERNEL_EVFILT_VIDEO_OUT) == OK,
EventQueue::KERNEL_EVFILT_VIDEO_OUT) == "delete duplicate event");
OK, Check(retained_owner->delete_count.load(std::memory_order_relaxed) == 1,
"delete duplicate event"); "duplicate add preserves delete callback");
Check(retained_owner->delete_count.load(std::memory_order_relaxed) == 1, retained_owner.reset();
"duplicate add preserves delete callback"); Check(weak_original.expired(), "original owner released on delete");
retained_owner.reset(); Check(EventQueue::KernelDeleteEqueue(queue) == OK, "delete duplicate add queue");
Check(weak_original.expired(), "original owner released on delete");
Check(EventQueue::KernelDeleteEqueue(queue) == OK,
"delete duplicate add queue");
} }
struct SimulatedVideoOutEventState; struct SimulatedVideoOutEventState;
struct SimulatedVideoOutRegistration { struct SimulatedVideoOutRegistration {
EventQueue::KernelEqueue handle = EventQueue::KERNEL_EQUEUE_INVALID; EventQueue::KernelEqueue handle = EventQueue::KERNEL_EQUEUE_INVALID;
std::shared_ptr<SimulatedVideoOutEventState> state; std::shared_ptr<SimulatedVideoOutEventState> state;
uint64_t marker = 0x123456789abcdef0ull; uint64_t marker = 0x123456789abcdef0ull;
}; };
struct SimulatedVideoOutEventState { struct SimulatedVideoOutEventState {
SimulatedVideoOutEventState(std::atomic_uint32_t &stage, SimulatedVideoOutEventState(std::atomic_uint32_t& stage, std::atomic_uint32_t& destroy_count)
std::atomic_uint32_t &destroy_count) : stage(stage), destroy_count(destroy_count) {}
: stage(stage), destroy_count(destroy_count) {}
~SimulatedVideoOutEventState() { ~SimulatedVideoOutEventState() { destroy_count.fetch_add(1, std::memory_order_relaxed); }
destroy_count.fetch_add(1, std::memory_order_relaxed);
}
std::mutex mutex; std::mutex mutex;
std::vector<std::shared_ptr<SimulatedVideoOutRegistration>> queues; std::vector<std::shared_ptr<SimulatedVideoOutRegistration>> queues;
std::atomic_uint32_t &stage; std::atomic_uint32_t& stage;
std::atomic_uint32_t &destroy_count; std::atomic_uint32_t& destroy_count;
uint64_t marker = 0xfedcba9876543210ull; uint64_t marker = 0xfedcba9876543210ull;
}; };
void DetachSimulatedVideoOutEvent(EventQueue::KernelEqueue queue, void DetachSimulatedVideoOutEvent(EventQueue::KernelEqueue queue,
EventQueue::KernelEqueueEvent *event) { EventQueue::KernelEqueueEvent* event) {
auto *registration = auto* registration = static_cast<SimulatedVideoOutRegistration*>(event->filter.data);
static_cast<SimulatedVideoOutRegistration *>(event->filter.data); Check(registration != nullptr && registration->handle == queue,
Check(registration != nullptr && registration->handle == queue, "simulated registration identity");
"simulated registration identity"); auto state = registration->state;
auto state = registration->state; Check(state != nullptr, "simulated event owns shared state");
Check(state != nullptr, "simulated event owns shared state"); state->stage.store(1, std::memory_order_release);
state->stage.store(1, std::memory_order_release); while (state->stage.load(std::memory_order_acquire) != 2) {
while (state->stage.load(std::memory_order_acquire) != 2) { std::this_thread::yield();
std::this_thread::yield(); }
}
{ {
std::lock_guard lock(state->mutex); std::lock_guard lock(state->mutex);
const auto entry = const auto entry = std::find_if(
std::find_if(state->queues.begin(), state->queues.end(), state->queues.begin(), state->queues.end(),
[registration](const auto &candidate) { [registration](const auto& candidate) { return candidate.get() == registration; });
return candidate.get() == registration; if (entry != state->queues.end()) {
}); state->queues.erase(entry);
if (entry != state->queues.end()) { }
state->queues.erase(entry); }
} event->filter.owner.reset();
} Check(state->marker == 0xfedcba9876543210ull && registration->marker == 0x123456789abcdef0ull,
event->filter.owner.reset(); "callback state survives simulated port destruction");
Check(state->marker == 0xfedcba9876543210ull &&
registration->marker == 0x123456789abcdef0ull,
"callback state survives simulated port destruction");
} }
void TestCallbackStateOutlivesPort() { void TestCallbackStateOutlivesPort() {
EventQueue::KernelEqueue queue = EventQueue::KERNEL_EQUEUE_INVALID; EventQueue::KernelEqueue queue = EventQueue::KERNEL_EQUEUE_INVALID;
Check(EventQueue::KernelCreateEqueue(&queue, "shared-port-state") == OK, Check(EventQueue::KernelCreateEqueue(&queue, "shared-port-state") == OK,
"create shared port state queue"); "create shared port state queue");
std::atomic_uint32_t stage{0}; std::atomic_uint32_t stage {0};
std::atomic_uint32_t destroy_count{0}; std::atomic_uint32_t destroy_count {0};
auto port_state = auto port_state = std::make_shared<SimulatedVideoOutEventState>(stage, destroy_count);
std::make_shared<SimulatedVideoOutEventState>(stage, destroy_count); std::weak_ptr<SimulatedVideoOutEventState> weak_state = port_state;
std::weak_ptr<SimulatedVideoOutEventState> weak_state = port_state; auto registration = std::make_shared<SimulatedVideoOutRegistration>();
auto registration = std::make_shared<SimulatedVideoOutRegistration>(); std::weak_ptr<SimulatedVideoOutRegistration> weak_registration = registration;
std::weak_ptr<SimulatedVideoOutRegistration> weak_registration = registration->handle = queue;
registration; registration->state = port_state;
registration->handle = queue; port_state->queues.push_back(registration);
registration->state = port_state;
port_state->queues.push_back(registration);
EventQueue::KernelEqueueEvent event{}; EventQueue::KernelEqueueEvent event {};
event.event.ident = 8; event.event.ident = 8;
event.event.filter = EventQueue::KERNEL_EVFILT_VIDEO_OUT; event.event.filter = EventQueue::KERNEL_EVFILT_VIDEO_OUT;
event.filter.data = registration.get(); event.filter.data = registration.get();
event.filter.owner = registration; event.filter.owner = registration;
event.filter.delete_event_func = DetachSimulatedVideoOutEvent; event.filter.delete_event_func = DetachSimulatedVideoOutEvent;
Check(EventQueue::KernelAddEvent(queue, event) == OK, Check(EventQueue::KernelAddEvent(queue, event) == OK, "add shared port state event");
"add shared port state event"); event.filter.owner.reset();
event.filter.owner.reset();
std::jthread close([&] { std::jthread close([&] {
Check(EventQueue::KernelDeleteEqueue(queue) == OK, Check(EventQueue::KernelDeleteEqueue(queue) == OK, "delete shared port state queue");
"delete shared port state queue"); });
}); while (stage.load(std::memory_order_acquire) != 1) {
while (stage.load(std::memory_order_acquire) != 1) { std::this_thread::yield();
std::this_thread::yield(); }
}
std::vector<std::shared_ptr<SimulatedVideoOutRegistration>> detached; std::vector<std::shared_ptr<SimulatedVideoOutRegistration>> detached;
{ {
std::lock_guard lock(port_state->mutex); std::lock_guard lock(port_state->mutex);
detached = std::move(port_state->queues); detached = std::move(port_state->queues);
} }
registration.reset(); registration.reset();
detached.clear(); detached.clear();
port_state.reset(); port_state.reset();
Check(!weak_state.expired(), Check(!weak_state.expired(), "callback state outlives simulated port object");
"callback state outlives simulated port object"); Check(!weak_registration.expired(), "registration outlives simulated port object");
Check(!weak_registration.expired(),
"registration outlives simulated port object");
stage.store(2, std::memory_order_release); stage.store(2, std::memory_order_release);
close.join(); close.join();
Check(weak_registration.expired(), "detached registration is released"); Check(weak_registration.expired(), "detached registration is released");
Check(weak_state.expired(), "detached event state is released"); Check(weak_state.expired(), "detached event state is released");
Check(destroy_count.load(std::memory_order_relaxed) == 1, Check(destroy_count.load(std::memory_order_relaxed) == 1,
"shared event state is destroyed exactly once"); "shared event state is destroyed exactly once");
} }
struct OwnedCallbackPayload { struct OwnedCallbackPayload {
OwnedCallbackPayload(std::atomic_uint32_t &stage, OwnedCallbackPayload(std::atomic_uint32_t& stage, std::atomic_uint32_t& delete_count,
std::atomic_uint32_t &delete_count, std::atomic_uint32_t& destroy_count)
std::atomic_uint32_t &destroy_count) : stage(stage), delete_count(delete_count), destroy_count(destroy_count) {}
: stage(stage), delete_count(delete_count), destroy_count(destroy_count) {}
std::atomic_uint32_t &stage; std::atomic_uint32_t& stage;
std::atomic_uint32_t &delete_count; std::atomic_uint32_t& delete_count;
std::atomic_uint32_t &destroy_count; std::atomic_uint32_t& destroy_count;
uint64_t marker = 0xc0dec0dec0dec0deull; uint64_t marker = 0xc0dec0dec0dec0deull;
~OwnedCallbackPayload() { ~OwnedCallbackPayload() { destroy_count.fetch_add(1, std::memory_order_relaxed); }
destroy_count.fetch_add(1, std::memory_order_relaxed);
}
}; };
void DeleteOwnedEvent(EventQueue::KernelEqueue queue, void DeleteOwnedEvent(EventQueue::KernelEqueue queue, EventQueue::KernelEqueueEvent* event) {
EventQueue::KernelEqueueEvent *event) { auto* payload = static_cast<OwnedCallbackPayload*>(event->filter.data);
auto *payload = static_cast<OwnedCallbackPayload *>(event->filter.data); Check(payload != nullptr, "owned callback payload");
Check(payload != nullptr, "owned callback payload"); Check(!EventQueue::KernelPinEqueue(queue), "owned callback runs after registry removal");
Check(!EventQueue::KernelPinEqueue(queue), payload->delete_count.fetch_add(1, std::memory_order_relaxed);
"owned callback runs after registry removal"); event->filter.owner.reset();
payload->delete_count.fetch_add(1, std::memory_order_relaxed); payload->stage.store(1, std::memory_order_release);
event->filter.owner.reset(); while (payload->stage.load(std::memory_order_acquire) != 2) {
payload->stage.store(1, std::memory_order_release); std::this_thread::yield();
while (payload->stage.load(std::memory_order_acquire) != 2) { }
std::this_thread::yield(); Check(payload->marker == 0xc0dec0dec0dec0deull, "owned callback payload remains valid");
}
Check(payload->marker == 0xc0dec0dec0dec0deull,
"owned callback payload remains valid");
} }
void TestCallbackOwnsPayload() { void TestCallbackOwnsPayload() {
EventQueue::KernelEqueue queue = EventQueue::KERNEL_EQUEUE_INVALID; EventQueue::KernelEqueue queue = EventQueue::KERNEL_EQUEUE_INVALID;
Check(EventQueue::KernelCreateEqueue(&queue, "owned-callback") == OK, Check(EventQueue::KernelCreateEqueue(&queue, "owned-callback") == OK,
"create owned callback queue"); "create owned callback queue");
std::atomic_uint32_t stage{0}; std::atomic_uint32_t stage {0};
std::atomic_uint32_t delete_count{0}; std::atomic_uint32_t delete_count {0};
std::atomic_uint32_t destroy_count{0}; std::atomic_uint32_t destroy_count {0};
auto registration = auto registration = std::make_shared<OwnedCallbackPayload>(stage, delete_count, destroy_count);
std::make_shared<OwnedCallbackPayload>(stage, delete_count, destroy_count); std::weak_ptr<OwnedCallbackPayload> weak_registration = registration;
std::weak_ptr<OwnedCallbackPayload> weak_registration = registration; std::vector<std::shared_ptr<OwnedCallbackPayload>> port_registrations {registration};
std::vector<std::shared_ptr<OwnedCallbackPayload>> port_registrations{ {
registration}; EventQueue::KernelEqueueEvent event {};
{ event.event.ident = 2;
EventQueue::KernelEqueueEvent event{}; event.event.filter = EventQueue::KERNEL_EVFILT_VIDEO_OUT;
event.event.ident = 2; event.filter.data = registration.get();
event.event.filter = EventQueue::KERNEL_EVFILT_VIDEO_OUT; event.filter.owner = registration;
event.filter.data = registration.get(); event.filter.delete_event_func = DeleteOwnedEvent;
event.filter.owner = registration; Check(EventQueue::KernelAddEvent(queue, event) == OK, "add owned callback event");
event.filter.delete_event_func = DeleteOwnedEvent; }
Check(EventQueue::KernelAddEvent(queue, event) == OK,
"add owned callback event");
}
std::jthread close([&] { std::jthread close(
Check(EventQueue::KernelDeleteEqueue(queue) == OK, [&] { Check(EventQueue::KernelDeleteEqueue(queue) == OK, "delete owned callback queue"); });
"delete owned callback queue"); while (stage.load(std::memory_order_acquire) != 1) {
}); std::this_thread::yield();
while (stage.load(std::memory_order_acquire) != 1) { }
std::this_thread::yield();
}
Check(!EventQueue::KernelPinEqueue(queue), Check(!EventQueue::KernelPinEqueue(queue),
"owned callback queue removed while callback blocked"); "owned callback queue removed while callback blocked");
port_registrations.clear(); port_registrations.clear();
registration.reset(); registration.reset();
Check(!weak_registration.expired(), Check(!weak_registration.expired(), "delete callback retains detached payload");
"delete callback retains detached payload");
stage.store(2, std::memory_order_release); stage.store(2, std::memory_order_release);
close.join(); close.join();
Check(delete_count.load(std::memory_order_relaxed) == 1, Check(delete_count.load(std::memory_order_relaxed) == 1, "owned callback runs exactly once");
"owned callback runs exactly once"); Check(weak_registration.expired(), "owned callback payload released with event");
Check(weak_registration.expired(), Check(destroy_count.load(std::memory_order_relaxed) == 1,
"owned callback payload released with event"); "owned callback payload destroyed exactly once");
Check(destroy_count.load(std::memory_order_relaxed) == 1,
"owned callback payload destroyed exactly once");
} }
void TestPinnedClose() { void TestPinnedClose() {
EventQueue::KernelEqueue queue = EventQueue::KERNEL_EQUEUE_INVALID; EventQueue::KernelEqueue queue = EventQueue::KERNEL_EQUEUE_INVALID;
Check(EventQueue::KernelCreateEqueue(&queue, "pinned-close") == OK, Check(EventQueue::KernelCreateEqueue(&queue, "pinned-close") == OK, "create pinned queue");
"create pinned queue");
std::atomic_uint32_t delete_count{0}; std::atomic_uint32_t delete_count {0};
EventQueue::KernelEqueueEvent event{}; EventQueue::KernelEqueueEvent event {};
event.event.ident = 1; event.event.ident = 1;
event.event.filter = EventQueue::KERNEL_EVFILT_VIDEO_OUT; event.event.filter = EventQueue::KERNEL_EVFILT_VIDEO_OUT;
event.filter.data = &delete_count; event.filter.data = &delete_count;
event.filter.delete_event_func = CountDeletedEvent; event.filter.delete_event_func = CountDeletedEvent;
Check(EventQueue::KernelAddEvent(queue, event) == OK, "add callback event"); Check(EventQueue::KernelAddEvent(queue, event) == OK, "add callback event");
auto owner = EventQueue::KernelPinEqueue(queue); auto owner = EventQueue::KernelPinEqueue(queue);
Check(owner != nullptr, "pin live queue"); Check(owner != nullptr, "pin live queue");
Check(EventQueue::KernelDeleteEqueue(queue) == OK, "delete pinned queue"); Check(EventQueue::KernelDeleteEqueue(queue) == OK, "delete pinned queue");
Check(delete_count.load(std::memory_order_relaxed) == 1, Check(delete_count.load(std::memory_order_relaxed) == 1, "close invokes callback once");
"close invokes callback once"); Check(!EventQueue::KernelPinEqueue(queue), "deleted queue leaves registry");
Check(!EventQueue::KernelPinEqueue(queue), "deleted queue leaves registry"); Check(EventQueue::KernelTriggerEvent(queue, 1, EventQueue::KERNEL_EVFILT_VIDEO_OUT, nullptr) ==
Check(EventQueue::KernelTriggerEvent(queue, 1, KERNEL_ERROR_EBADF,
EventQueue::KERNEL_EVFILT_VIDEO_OUT, "stale trigger rejected");
nullptr) == KERNEL_ERROR_EBADF, Check(EventQueue::KernelDeleteEqueue(queue) == KERNEL_ERROR_EBADF,
"stale trigger rejected"); "second queue delete rejected");
Check(EventQueue::KernelDeleteEqueue(queue) == KERNEL_ERROR_EBADF,
"second queue delete rejected");
owner.reset(); owner.reset();
Check(delete_count.load(std::memory_order_relaxed) == 1, Check(delete_count.load(std::memory_order_relaxed) == 1,
"deferred destruction does not repeat callback"); "deferred destruction does not repeat callback");
} }
void TestStaleHandleNeverAliasesNewQueue() { void TestStaleHandleNeverAliasesNewQueue() {
EventQueue::KernelEqueue stale = EventQueue::KERNEL_EQUEUE_INVALID; EventQueue::KernelEqueue stale = EventQueue::KERNEL_EQUEUE_INVALID;
Check(EventQueue::KernelCreateEqueue(&stale, "stale-handle") == OK, Check(EventQueue::KernelCreateEqueue(&stale, "stale-handle") == OK, "create stale queue");
"create stale queue"); Check(EventQueue::KernelDeleteEqueue(stale) == OK, "delete stale queue");
Check(EventQueue::KernelDeleteEqueue(stale) == OK, "delete stale queue");
EventQueue::KernelEqueue replacement = EventQueue::KERNEL_EQUEUE_INVALID; EventQueue::KernelEqueue replacement = EventQueue::KERNEL_EQUEUE_INVALID;
Check(EventQueue::KernelCreateEqueue(&replacement, "replacement") == OK, Check(EventQueue::KernelCreateEqueue(&replacement, "replacement") == OK,
"create replacement queue"); "create replacement queue");
Check(stale != replacement, "queue handles are never recycled"); Check(stale != replacement, "queue handles are never recycled");
Check(!EventQueue::KernelPinEqueue(stale), "stale handle does not pin"); Check(!EventQueue::KernelPinEqueue(stale), "stale handle does not pin");
Check(EventQueue::KernelAddUserEvent(stale, 11) == KERNEL_ERROR_EBADF, Check(EventQueue::KernelAddUserEvent(stale, 11) == KERNEL_ERROR_EBADF,
"stale handle cannot mutate replacement"); "stale handle cannot mutate replacement");
Check(EventQueue::KernelAddUserEvent(replacement, 11) == OK, Check(EventQueue::KernelAddUserEvent(replacement, 11) == OK,
"replacement handle remains valid"); "replacement handle remains valid");
Check(EventQueue::KernelTriggerUserEvent(stale, 11, nullptr) == Check(EventQueue::KernelTriggerUserEvent(stale, 11, nullptr) == KERNEL_ERROR_EBADF,
KERNEL_ERROR_EBADF, "stale handle cannot trigger replacement");
"stale handle cannot trigger replacement"); Check(EventQueue::KernelTriggerUserEvent(replacement, 11, nullptr) == OK,
Check(EventQueue::KernelTriggerUserEvent(replacement, 11, nullptr) == OK, "replacement event triggers");
"replacement event triggers"); Check(EventQueue::KernelDeleteEqueue(replacement) == OK, "delete replacement queue");
Check(EventQueue::KernelDeleteEqueue(replacement) == OK,
"delete replacement queue");
} }
void TestConcurrentCloseCallback() { void TestConcurrentCloseCallback() {
for (uint32_t iteration = 0; iteration < 64; iteration++) { for (uint32_t iteration = 0; iteration < 64; iteration++) {
EventQueue::KernelEqueue queue = EventQueue::KERNEL_EQUEUE_INVALID; EventQueue::KernelEqueue queue = EventQueue::KERNEL_EQUEUE_INVALID;
Check(EventQueue::KernelCreateEqueue(&queue, "callback-race") == OK, Check(EventQueue::KernelCreateEqueue(&queue, "callback-race") == OK,
"create callback race queue"); "create callback race queue");
std::atomic_uint32_t delete_count{0}; std::atomic_uint32_t delete_count {0};
EventQueue::KernelEqueueEvent callback_event{}; EventQueue::KernelEqueueEvent callback_event {};
callback_event.event.ident = 9; callback_event.event.ident = 9;
callback_event.event.filter = EventQueue::KERNEL_EVFILT_GRAPHICS; callback_event.event.filter = EventQueue::KERNEL_EVFILT_GRAPHICS;
callback_event.filter.data = &delete_count; callback_event.filter.data = &delete_count;
callback_event.filter.delete_event_func = CountDeletedEvent; callback_event.filter.delete_event_func = CountDeletedEvent;
Check(EventQueue::KernelAddEvent(queue, callback_event) == OK, Check(EventQueue::KernelAddEvent(queue, callback_event) == OK, "add callback race event");
"add callback race event");
std::atomic_bool start{false}; std::atomic_bool start {false};
std::jthread trigger([&] { std::jthread trigger([&] {
while (!start.load(std::memory_order_acquire)) { while (!start.load(std::memory_order_acquire)) {
std::this_thread::yield(); std::this_thread::yield();
} }
for (uint32_t i = 0; i < 256; i++) { for (uint32_t i = 0; i < 256; i++) {
CheckConcurrentResult( CheckConcurrentResult(EventQueue::KernelTriggerEvent(
EventQueue::KernelTriggerEvent( queue, 9, EventQueue::KERNEL_EVFILT_GRAPHICS, nullptr),
queue, 9, EventQueue::KERNEL_EVFILT_GRAPHICS, nullptr), "callback race trigger result");
"callback race trigger result"); }
} });
});
start.store(true, std::memory_order_release); start.store(true, std::memory_order_release);
Check(EventQueue::KernelDeleteEqueue(queue) == OK, Check(EventQueue::KernelDeleteEqueue(queue) == OK, "callback race queue delete");
"callback race queue delete"); trigger.join();
trigger.join(); Check(delete_count.load(std::memory_order_relaxed) == 1,
Check(delete_count.load(std::memory_order_relaxed) == 1, "concurrent close invokes callback exactly once");
"concurrent close invokes callback exactly once"); }
}
} }
void TestConcurrentDelete() { void TestConcurrentDelete() {
for (uint32_t iteration = 0; iteration < 64; iteration++) { for (uint32_t iteration = 0; iteration < 64; iteration++) {
EventQueue::KernelEqueue queue = EventQueue::KERNEL_EQUEUE_INVALID; EventQueue::KernelEqueue queue = EventQueue::KERNEL_EQUEUE_INVALID;
Check(EventQueue::KernelCreateEqueue(&queue, "concurrent-delete") == OK, Check(EventQueue::KernelCreateEqueue(&queue, "concurrent-delete") == OK,
"create concurrent queue"); "create concurrent queue");
EventQueue::KernelEqueueEvent event{}; EventQueue::KernelEqueueEvent event {};
event.event.ident = 7; event.event.ident = 7;
event.event.filter = EventQueue::KERNEL_EVFILT_USER; event.event.filter = EventQueue::KERNEL_EVFILT_USER;
Check(EventQueue::KernelAddEvent(queue, event) == OK, Check(EventQueue::KernelAddEvent(queue, event) == OK, "add concurrent event");
"add concurrent event");
std::atomic_bool start{false}; std::atomic_bool start {false};
std::jthread mutate([&] { std::jthread mutate([&] {
while (!start.load(std::memory_order_acquire)) { while (!start.load(std::memory_order_acquire)) {
std::this_thread::yield(); std::this_thread::yield();
} }
for (uint32_t i = 0; i < 64; i++) { for (uint32_t i = 0; i < 64; i++) {
CheckConcurrentResult(EventQueue::KernelAddEvent(queue, event), CheckConcurrentResult(EventQueue::KernelAddEvent(queue, event),
"concurrent add result"); "concurrent add result");
CheckConcurrentResult( CheckConcurrentResult(EventQueue::KernelTriggerEvent(
EventQueue::KernelTriggerEvent( queue, 7, EventQueue::KERNEL_EVFILT_USER, nullptr),
queue, 7, EventQueue::KERNEL_EVFILT_USER, nullptr), "concurrent trigger result");
"concurrent trigger result"); CheckConcurrentResult(
CheckConcurrentResult(EventQueue::KernelDeleteEvent( EventQueue::KernelDeleteEvent(queue, 7, EventQueue::KERNEL_EVFILT_USER),
queue, 7, EventQueue::KERNEL_EVFILT_USER), "concurrent event delete result");
"concurrent event delete result"); }
} });
}); std::jthread trigger([&] {
std::jthread trigger([&] { while (!start.load(std::memory_order_acquire)) {
while (!start.load(std::memory_order_acquire)) { std::this_thread::yield();
std::this_thread::yield(); }
} for (uint32_t i = 0; i < 128; i++) {
for (uint32_t i = 0; i < 128; i++) { CheckConcurrentResult(EventQueue::KernelTriggerEvent(
CheckConcurrentResult( queue, 7, EventQueue::KERNEL_EVFILT_USER, nullptr),
EventQueue::KernelTriggerEvent( "parallel trigger result");
queue, 7, EventQueue::KERNEL_EVFILT_USER, nullptr), }
"parallel trigger result"); });
}
});
start.store(true, std::memory_order_release); start.store(true, std::memory_order_release);
Check(EventQueue::KernelDeleteEqueue(queue) == OK, Check(EventQueue::KernelDeleteEqueue(queue) == OK, "concurrent queue delete");
"concurrent queue delete"); mutate.join();
mutate.join(); trigger.join();
trigger.join(); Check(!EventQueue::KernelPinEqueue(queue), "concurrent queue removed from registry");
Check(!EventQueue::KernelPinEqueue(queue), }
"concurrent queue removed from registry");
}
} }
} // namespace } // namespace
int main() { int main() {
TestDuplicateAddPreservesEventState(); TestDuplicateAddPreservesEventState();
TestCallbackStateOutlivesPort(); TestCallbackStateOutlivesPort();
TestCallbackOwnsPayload(); TestCallbackOwnsPayload();
TestPinnedClose(); TestPinnedClose();
TestStaleHandleNeverAliasesNewQueue(); TestStaleHandleNeverAliasesNewQueue();
TestConcurrentCloseCallback(); TestConcurrentCloseCallback();
TestConcurrentDelete(); TestConcurrentDelete();
std::printf("EventQueueLifetimeTests: all cases passed\n"); std::printf("EventQueueLifetimeTests: all cases passed\n");
return 0; return 0;
} }
+41 -24
View File
@@ -7,8 +7,8 @@
namespace { namespace {
using Owners = std::vector<uint32_t>; using Owners = std::vector<uint32_t>;
using Table = Libs::Graphics::MultiLevelPageTable<Owners>; using Table = Libs::Graphics::MultiLevelPageTable<Owners>;
using OwnerIndex = Libs::Graphics::MultiRangePageOwnerIndex<uint32_t>; using OwnerIndex = Libs::Graphics::MultiRangePageOwnerIndex<uint32_t>;
void Check(bool value, const char* text) { void Check(bool value, const char* text) {
@@ -24,23 +24,26 @@ void TestMultiOwnerAndExactErase() {
owners.push_back(11); owners.push_back(11);
owners.push_back(22); owners.push_back(22);
Check(table.Find(17) != nullptr && table.Find(17)->size() == 2, "both page owners are retained"); Check(table.Find(17) != nullptr && table.Find(17)->size() == 2,
"both page owners are retained");
Check(Libs::Graphics::EraseExact(owners, 11U), "registered owner is erased"); Check(Libs::Graphics::EraseExact(owners, 11U), "registered owner is erased");
Check(owners.size() == 1 && owners.front() == 22, "erasing one owner preserves its neighbor"); Check(owners.size() == 1 && owners.front() == 22, "erasing one owner preserves its neighbor");
Check(!Libs::Graphics::EraseExact(owners, 33U), "missing owner is reported without mutation"); Check(!Libs::Graphics::EraseExact(owners, 33U), "missing owner is reported without mutation");
} }
void TestCrossBucketRange() { void TestCrossBucketRange() {
Table::PageRange range{}; Table::PageRange range {};
constexpr uint64_t bucket_boundary = uint64_t{Table::kBucketEntries} << Table::kPageBits; constexpr uint64_t bucket_boundary = uint64_t {Table::kBucketEntries} << Table::kPageBits;
Check(Table::TryGetPageRange(bucket_boundary - 1, 2, range), "cross-bucket range is valid"); Check(Table::TryGetPageRange(bucket_boundary - 1, 2, range), "cross-bucket range is valid");
Check(range.first == Table::kBucketEntries - 1 && range.last_exclusive == Table::kBucketEntries + 1, Check(range.first == Table::kBucketEntries - 1 &&
range.last_exclusive == Table::kBucketEntries + 1,
"cross-bucket range covers both pages"); "cross-bucket range covers both pages");
Table table; Table table;
table[range.first].push_back(1); table[range.first].push_back(1);
table[range.last_exclusive - 1].push_back(2); table[range.last_exclusive - 1].push_back(2);
Check(table.AllocatedBucketCount() == 2, "pages across the L1 boundary use distinct sparse buckets"); Check(table.AllocatedBucketCount() == 2,
"pages across the L1 boundary use distinct sparse buckets");
} }
void TestQueriesDoNotAllocate() { void TestQueriesDoNotAllocate() {
@@ -55,27 +58,34 @@ void TestQueriesDoNotAllocate() {
} }
void TestAddressSpaceBoundaries() { void TestAddressSpaceBoundaries() {
Table::PageRange range{}; Table::PageRange range {};
Check(Table::TryGetPageRange(Table::kAddressSpaceSize - 1, 1, range), "last guest byte is valid"); Check(Table::TryGetPageRange(Table::kAddressSpaceSize - 1, 1, range),
"last guest byte is valid");
Check(range.first == Table::kPageCount - 1 && range.last_exclusive == Table::kPageCount, Check(range.first == Table::kPageCount - 1 && range.last_exclusive == Table::kPageCount,
"last guest byte maps to the final page"); "last guest byte maps to the final page");
Check(!Table::TryGetPageRange(0, 0, range), "empty ranges are rejected"); Check(!Table::TryGetPageRange(0, 0, range), "empty ranges are rejected");
Check(!Table::TryGetPageRange(Table::kAddressSpaceSize, 1, range), "first out-of-range byte is rejected"); Check(!Table::TryGetPageRange(Table::kAddressSpaceSize, 1, range),
Check(!Table::TryGetPageRange(Table::kAddressSpaceSize - 1, 2, range), "crossing the address-space end is rejected"); "first out-of-range byte is rejected");
Check(!Table::TryGetPageRange(Table::kAddressSpaceSize - 1, 2, range),
"crossing the address-space end is rejected");
Check(!Table::TryGetPageRange(UINT64_MAX - 1, 4, range), "wrapping input is rejected"); Check(!Table::TryGetPageRange(UINT64_MAX - 1, 4, range), "wrapping input is rejected");
Table table; Table table;
table.GetOrCreate(Table::kPageCount - 1).push_back(99); table.GetOrCreate(Table::kPageCount - 1).push_back(99);
Check(table.Find(Table::kPageCount - 1) != nullptr && table.Find(Table::kPageCount - 1)->front() == 99, Check(table.Find(Table::kPageCount - 1) != nullptr &&
table.Find(Table::kPageCount - 1)->front() == 99,
"final page supports allocating and nonallocating access"); "final page supports allocating and nonallocating access");
} }
void TestMultiRangeRegistrationDeduplicatesPages() { void TestMultiRangeRegistrationDeduplicatesPages() {
OwnerIndex index; OwnerIndex index;
// Depth and stencil-like planes overlap tracking pages and share one 1 MiB bucket. // Depth and stencil-like planes overlap tracking pages and share one 1 MiB bucket.
Check(index.Register(7, {{0x101000, 0x2800}, {0x102000, 0x3000}}), "multi-range owner registers"); Check(index.Register(7, {{0x101000, 0x2800}, {0x102000, 0x3000}}),
Check(index.CoarseMembershipCount(1) == 1, "one owner is inserted once in a shared 1 MiB bucket"); "multi-range owner registers");
Check(index.TrackingMembershipCount(0x102) == 1, "overlapping planes insert one 4 KiB membership"); Check(index.CoarseMembershipCount(1) == 1,
"one owner is inserted once in a shared 1 MiB bucket");
Check(index.TrackingMembershipCount(0x102) == 1,
"overlapping planes insert one 4 KiB membership");
Check(!index.Register(7, {{0x101000, 0x1000}}), "duplicate owner registration hard-fails"); Check(!index.Register(7, {{0x101000, 0x1000}}), "duplicate owner registration hard-fails");
const auto owners = index.Query(0x100000, 0x10000); const auto owners = index.Query(0x100000, 0x10000);
@@ -83,9 +93,10 @@ void TestMultiRangeRegistrationDeduplicatesPages() {
} }
void TestSharedPageUnregisterLifecycle() { void TestSharedPageUnregisterLifecycle() {
OwnerIndex index; OwnerIndex index;
const std::vector<OwnerIndex::ByteRange> ranges{{0x202000, 0x2000}}; const std::vector<OwnerIndex::ByteRange> ranges {{0x202000, 0x2000}};
Check(index.Register(11, ranges) && index.Register(22, ranges), "two owners register on identical pages"); Check(index.Register(11, ranges) && index.Register(22, ranges),
"two owners register on identical pages");
Check(index.CoarseMembershipCount(2) == 2 && index.TrackingMembershipCount(0x202) == 2, Check(index.CoarseMembershipCount(2) == 2 && index.TrackingMembershipCount(0x202) == 2,
"coarse and tracking pages retain both owners"); "coarse and tracking pages retain both owners");
@@ -97,7 +108,8 @@ void TestSharedPageUnregisterLifecycle() {
Check(!index.Unregister(11, releases), "missing membership hard-fails without mutation"); Check(!index.Unregister(11, releases), "missing membership hard-fails without mutation");
Check(index.Unregister(22, releases), "final owner unregisters"); Check(index.Unregister(22, releases), "final owner unregisters");
Check(releases.size() == 1 && releases.front().address == 0x202000 && releases.front().size == 0x2000, Check(releases.size() == 1 && releases.front().address == 0x202000 &&
releases.front().size == 0x2000,
"adjacent final-owner tracking pages return one contiguous release"); "adjacent final-owner tracking pages return one contiguous release");
} }
@@ -105,14 +117,19 @@ void TestStrictByteFilteringAndPredicate() {
OwnerIndex index; OwnerIndex index;
Check(index.Register(31, {{0x300100, 0x100}}), "first byte-disjoint owner registers"); Check(index.Register(31, {{0x300100, 0x100}}), "first byte-disjoint owner registers");
Check(index.Register(32, {{0x300800, 0x100}}), "second byte-disjoint owner registers"); Check(index.Register(32, {{0x300800, 0x100}}), "second byte-disjoint owner registers");
Check(index.TrackingMembershipCount(0x300) == 2, "byte-disjoint owners share one tracking page"); Check(index.TrackingMembershipCount(0x300) == 2,
"byte-disjoint owners share one tracking page");
Check(index.Query(0x300400, 0x40).empty(), "page hit without byte overlap is filtered out"); Check(index.Query(0x300400, 0x40).empty(), "page hit without byte overlap is filtered out");
const auto page_candidates = index.QueryCandidates(0x300400, 0x40); const auto page_candidates = index.QueryCandidates(0x300400, 0x40);
Check(page_candidates.size() == 2, "fault candidate query retains byte-disjoint owners on the touched page"); Check(page_candidates.size() == 2,
"fault candidate query retains byte-disjoint owners on the touched page");
const auto first = index.Query(0x300180, 0x10); const auto first = index.Query(0x300180, 0x10);
Check(first.size() == 1 && first.front() == 31, "strict byte overlap selects only the matching owner"); Check(first.size() == 1 && first.front() == 31,
const auto predicate_filtered = index.Query(0x300000, 0x1000, [](uint32_t owner) { return owner == 32; }); "strict byte overlap selects only the matching owner");
Check(predicate_filtered.size() == 1 && predicate_filtered.front() == 32, "supplied predicate filters query owners"); const auto predicate_filtered =
index.Query(0x300000, 0x1000, [](uint32_t owner) { return owner == 32; });
Check(predicate_filtered.size() == 1 && predicate_filtered.front() == 32,
"supplied predicate filters query owners");
} }
} // namespace } // namespace

Some files were not shown because too many files have changed in this diff Show More