mirror of
https://github.com/KytyPS5/KytyPS5.git
synced 2026-08-03 11:23:49 +00:00
Compare commits
8
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
e91dd39cb0 | ||
|
|
cc76827e63 | ||
|
|
832bc84100 | ||
|
|
65a0f0baa7 | ||
|
|
b9ae2537ef | ||
|
|
0b9edaa721 | ||
|
|
8a244677d7 | ||
|
|
f6e01e5403 |
@@ -432,11 +432,6 @@ uint64_t SysVirtualReserveAligned(uint64_t address, uint64_t size, uint64_t alig
|
||||
|
||||
pthread_mutex_lock(&g_virtual_mutex);
|
||||
record_alloc(ret_addr, size);
|
||||
uintptr_t page_start = ret_addr >> 12u;
|
||||
uintptr_t page_end = (ret_addr + size - 1) >> 12u;
|
||||
for (uintptr_t page = page_start; page <= page_end; page++) {
|
||||
(*g_protects)[page] = PROT_NONE;
|
||||
}
|
||||
pthread_mutex_unlock(&g_virtual_mutex);
|
||||
|
||||
return ret_addr;
|
||||
@@ -470,11 +465,6 @@ bool SysVirtualReserveFixed(uint64_t address, uint64_t size) {
|
||||
if (ptr != MAP_FAILED) {
|
||||
pthread_mutex_lock(&g_virtual_mutex);
|
||||
record_alloc(ret_addr, size);
|
||||
uintptr_t page_start = ret_addr >> 12u;
|
||||
uintptr_t page_end = (ret_addr + size - 1) >> 12u;
|
||||
for (uintptr_t page = page_start; page <= page_end; page++) {
|
||||
(*g_protects)[page] = PROT_NONE;
|
||||
}
|
||||
pthread_mutex_unlock(&g_virtual_mutex);
|
||||
|
||||
return true;
|
||||
|
||||
@@ -38,10 +38,51 @@ public:
|
||||
bool downloaded) noexcept;
|
||||
[[nodiscard]] bool InvalidateRegion(uint64_t vaddr, uint64_t size,
|
||||
PageFaultPhase phase) noexcept;
|
||||
template <typename Flush>
|
||||
void InvalidateRegion(uint64_t vaddr, uint64_t size, Flush&& on_flush) {
|
||||
static_assert(std::is_invocable_v<Flush&>);
|
||||
CheckNotInUploadCallback();
|
||||
ValidateRange(vaddr, size);
|
||||
|
||||
const auto update_cpu_state = [this, vaddr, size] {
|
||||
std::lock_guard access(m_access_mutex);
|
||||
std::vector<RegionManager*> managers;
|
||||
Iterate<false>(vaddr, size, [&](RegionManager* manager, uint64_t, uint64_t) {
|
||||
managers.push_back(manager);
|
||||
});
|
||||
std::vector<std::unique_lock<TrackingSpinLock>> locks;
|
||||
locks.reserve(managers.size());
|
||||
for (auto* manager: managers) {
|
||||
locks.emplace_back(manager->lock);
|
||||
}
|
||||
const bool gpu_modified = Iterate<false>(
|
||||
vaddr, size, [](RegionManager* manager, uint64_t offset, uint64_t bytes) {
|
||||
return manager->IsModified<DirtySource::Gpu>(offset, bytes);
|
||||
});
|
||||
if (gpu_modified) {
|
||||
return true;
|
||||
}
|
||||
Iterate<false>(vaddr, size,
|
||||
[](RegionManager* manager, uint64_t offset, uint64_t bytes) {
|
||||
const auto changed = manager->ChangeState<DirtySource::Cpu, true>(
|
||||
manager->GetCpuAddr() + offset, bytes);
|
||||
manager->ApplyProtection(changed, false);
|
||||
});
|
||||
return false;
|
||||
};
|
||||
|
||||
if (!update_cpu_state()) {
|
||||
return;
|
||||
}
|
||||
std::forward<Flush>(on_flush)();
|
||||
if (update_cpu_state()) {
|
||||
EXIT("memory invalidation retained GPU-owned pages\n");
|
||||
}
|
||||
}
|
||||
[[nodiscard]] bool InvalidateVirtualGpuWrite(PageFaultAccess access, uint64_t vaddr,
|
||||
uint64_t size, PageFaultPhase phase) noexcept;
|
||||
void ValidateGpuDirtyPages(const RangeSet& dirty, uint64_t vaddr, uint64_t size,
|
||||
const char* operation) const noexcept;
|
||||
void ValidateGpuDirtyPages(const RangeSet& dirty, uint64_t vaddr, uint64_t size,
|
||||
const char* operation) const noexcept;
|
||||
void ValidateGpuDirtyOwnership(const RangeSet& dirty, uint64_t vaddr, uint64_t size,
|
||||
const char* operation);
|
||||
|
||||
|
||||
@@ -699,26 +699,6 @@ bool PageManager::IsMapped(uint64_t vaddr, uint64_t size) const noexcept {
|
||||
return true;
|
||||
}
|
||||
|
||||
bool PageManager::HasAnyMapping(uint64_t vaddr, uint64_t size) const noexcept {
|
||||
if (g_in_fault_resolution || vaddr == 0 || size == 0 || vaddr >= ADDRESS_SIZE ||
|
||||
size > ADDRESS_SIZE - vaddr) {
|
||||
return false;
|
||||
}
|
||||
const auto end = PageEnd(vaddr, size);
|
||||
for (auto page_vaddr = PageStart(vaddr); page_vaddr < end; page_vaddr += PAGE_SIZE) {
|
||||
auto* region = m_impl->FindRegion(page_vaddr);
|
||||
if (region == nullptr) {
|
||||
continue;
|
||||
}
|
||||
auto& page = m_impl->GetPage(*region, page_vaddr);
|
||||
SpinGuard lock(page.lock);
|
||||
if (page.mappings != 0) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
bool PageManager::HasGpuAccess(uint64_t vaddr, uint64_t size, GpuAccess access) const noexcept {
|
||||
if (access != GpuAccess::Read && access != GpuAccess::Write && access != GpuAccess::ReadWrite) {
|
||||
FailFast("HasGpuAccess received an invalid GPU access mode");
|
||||
@@ -1150,22 +1130,4 @@ bool PageManager::HandleFault(PageFaultAccess access, uint64_t fault_vaddr) noex
|
||||
return true;
|
||||
}
|
||||
|
||||
bool PageManager::HandleWriteRange(uint64_t vaddr, uint64_t size) noexcept {
|
||||
if (g_in_fault_resolution || vaddr == 0 || size == 0 || vaddr >= ADDRESS_SIZE ||
|
||||
size > ADDRESS_SIZE - vaddr) {
|
||||
return false;
|
||||
}
|
||||
const auto end = PageEnd(vaddr, size);
|
||||
for (auto page_vaddr = PageStart(vaddr); page_vaddr < end; page_vaddr += PAGE_SIZE) {
|
||||
if (!IsMapped(page_vaddr, 1)) {
|
||||
continue;
|
||||
}
|
||||
const auto fault_vaddr = std::max(page_vaddr, vaddr);
|
||||
if (!HandleFault(PageFaultAccess::Write, fault_vaddr)) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
} // namespace Libs::Graphics
|
||||
|
||||
@@ -41,7 +41,6 @@ public:
|
||||
[[nodiscard]] uint64_t GetPageSize() const;
|
||||
[[nodiscard]] bool IsTracked(uint64_t vaddr) const noexcept;
|
||||
[[nodiscard]] bool IsMapped(uint64_t vaddr, uint64_t size) const noexcept;
|
||||
[[nodiscard]] bool HasAnyMapping(uint64_t vaddr, uint64_t size) const noexcept;
|
||||
[[nodiscard]] bool HasGpuAccess(uint64_t vaddr, uint64_t size, GpuAccess access) const noexcept;
|
||||
|
||||
void UpdatePageWatchers(bool track, uint64_t vaddr, uint64_t size,
|
||||
@@ -50,9 +49,8 @@ public:
|
||||
void OnGpuUnmap(uint64_t vaddr, uint64_t size, GpuAccess access = GpuAccess::ReadWrite);
|
||||
|
||||
[[nodiscard]] bool HandleFault(PageFaultAccess access, uint64_t fault_vaddr) noexcept;
|
||||
[[nodiscard]] bool HandleWriteRange(uint64_t vaddr, uint64_t size) noexcept;
|
||||
[[nodiscard]] std::vector<std::unique_ptr<BackingWrite>>
|
||||
ReserveBackingWrites(std::span<const RangeSet::Range> ranges);
|
||||
ReserveBackingWrites(std::span<const RangeSet::Range> ranges);
|
||||
|
||||
private:
|
||||
void BeginBackingWrite(uint64_t vaddr, uint64_t size) noexcept;
|
||||
|
||||
@@ -59,6 +59,16 @@ public:
|
||||
return result;
|
||||
}
|
||||
|
||||
[[nodiscard]] bool Contains(uint64_t address, uint64_t size) const {
|
||||
const auto end = End(address, size);
|
||||
auto it = m_ranges.upper_bound(address);
|
||||
if (it == m_ranges.begin()) {
|
||||
return false;
|
||||
}
|
||||
--it;
|
||||
return it->first <= address && it->second >= end;
|
||||
}
|
||||
|
||||
template <typename Func>
|
||||
void ForEachIntersection(uint64_t address, uint64_t size, Func&& func) const {
|
||||
const auto end = End(address, size);
|
||||
|
||||
+132
-72
@@ -4,10 +4,10 @@
|
||||
#include "common/logging/log.h"
|
||||
#include "common/profiler.h"
|
||||
#include "graphics/host_gpu/graphicContext.h"
|
||||
#include "graphics/host_gpu/renderer/commandScheduler.h"
|
||||
#include "graphics/host_gpu/renderer/render.h"
|
||||
#include "graphics/host_gpu/renderer/cache/resourceMutex.h"
|
||||
#include "graphics/host_gpu/renderer/cache/textureCache.h"
|
||||
#include "graphics/host_gpu/renderer/commandScheduler.h"
|
||||
#include "graphics/host_gpu/renderer/render.h"
|
||||
#include "kernel/memory.h"
|
||||
|
||||
#include <algorithm>
|
||||
@@ -183,9 +183,8 @@ BufferCache::RecordDownloads(std::span<const DownloadCopy> copies) {
|
||||
return {};
|
||||
}
|
||||
|
||||
auto& download = m_download_buffer;
|
||||
const auto [mapped, base_offset] =
|
||||
download.Map(reservation_size, DOWNLOAD_ALIGNMENT);
|
||||
auto& download = m_download_buffer;
|
||||
const auto [mapped, base_offset] = download.Map(reservation_size, DOWNLOAD_ALIGNMENT);
|
||||
if (mapped == nullptr) {
|
||||
EXIT("BufferCache: download batch could not reserve the shared stream\n");
|
||||
}
|
||||
@@ -215,40 +214,38 @@ void BufferCache::PublishDownloads(std::span<const DownloadRange> downloads) {
|
||||
}
|
||||
}
|
||||
|
||||
void BufferCache::QueueGarbageDownload(std::span<const DownloadCopy> copies,
|
||||
RetiredBuffer retire) {
|
||||
void BufferCache::QueueGarbageDownload(std::span<const DownloadCopy> copies, RetiredBuffer retire) {
|
||||
if (copies.empty()) {
|
||||
return;
|
||||
}
|
||||
auto downloads = RecordDownloads(copies);
|
||||
const auto tick = m_scheduler.CurrentTick();
|
||||
auto downloads = RecordDownloads(copies);
|
||||
const auto tick = m_scheduler.CurrentTick();
|
||||
BeginBackingPublication(retire.address, retire.size, tick);
|
||||
m_scheduler.DeferOperation(
|
||||
[this, downloads = std::move(downloads), retire = std::move(retire), tick]() mutable {
|
||||
PublishDownloads(downloads);
|
||||
{
|
||||
FaultSafeCacheLock lock(this, m_mutex);
|
||||
if (m_memory_tracker.IsRegionGpuModified(retire.address, retire.size)) {
|
||||
m_memory_tracker.ForEachDownloadRange<true>(
|
||||
retire.address, retire.size,
|
||||
[&](uint64_t address, uint64_t size) noexcept {
|
||||
m_memory_tracker.ValidateGpuDirtyPages(
|
||||
m_gpu_modified_ranges, address, size,
|
||||
"asynchronous garbage retirement");
|
||||
},
|
||||
[](uint64_t, uint64_t) noexcept {});
|
||||
}
|
||||
for (const auto& range: downloads) {
|
||||
m_gpu_modified_ranges.Subtract(range.address, range.size);
|
||||
}
|
||||
if (m_memory_tracker.IsRegionGpuModified(retire.address, retire.size) ||
|
||||
!m_gpu_modified_ranges.Intersections(retire.address, retire.size).empty()) {
|
||||
EXIT("BufferCache: asynchronous garbage collection retained GPU ownership\n");
|
||||
}
|
||||
m_memory_tracker.UntrackMemory(retire.address, retire.size);
|
||||
}
|
||||
CompleteBackingPublication(retire.address, retire.size, tick);
|
||||
});
|
||||
m_scheduler.DeferOperation([this, downloads = std::move(downloads), retire = std::move(retire),
|
||||
tick]() mutable {
|
||||
PublishDownloads(downloads);
|
||||
{
|
||||
FaultSafeCacheLock lock(this, m_mutex);
|
||||
if (m_memory_tracker.IsRegionGpuModified(retire.address, retire.size)) {
|
||||
m_memory_tracker.ForEachDownloadRange<true>(
|
||||
retire.address, retire.size,
|
||||
[&](uint64_t address, uint64_t size) noexcept {
|
||||
m_memory_tracker.ValidateGpuDirtyPages(m_gpu_modified_ranges, address, size,
|
||||
"asynchronous garbage retirement");
|
||||
},
|
||||
[](uint64_t, uint64_t) noexcept {});
|
||||
}
|
||||
for (const auto& range: downloads) {
|
||||
m_gpu_modified_ranges.Subtract(range.address, range.size);
|
||||
}
|
||||
if (m_memory_tracker.IsRegionGpuModified(retire.address, retire.size) ||
|
||||
!m_gpu_modified_ranges.Intersections(retire.address, retire.size).empty()) {
|
||||
EXIT("BufferCache: asynchronous garbage collection retained GPU ownership\n");
|
||||
}
|
||||
m_memory_tracker.UntrackMemory(retire.address, retire.size);
|
||||
}
|
||||
CompleteBackingPublication(retire.address, retire.size, tick);
|
||||
});
|
||||
}
|
||||
|
||||
BufferCache::BufferCache(GraphicContext& graphics, CommandScheduler& scheduler,
|
||||
@@ -301,14 +298,13 @@ BufferCache::~BufferCache() {
|
||||
bool BufferCache::SynchronizeBacking(uint64_t vaddr, uint64_t size) {
|
||||
bool waited = false;
|
||||
for (;;) {
|
||||
uint64_t tick = 0;
|
||||
uint64_t tick = 0;
|
||||
const auto page_begin = vaddr & ~(TRACKER_PAGE_SIZE - 1);
|
||||
const auto page_end =
|
||||
(vaddr + size + TRACKER_PAGE_SIZE - 1) & ~(TRACKER_PAGE_SIZE - 1);
|
||||
const auto page_end = (vaddr + size + TRACKER_PAGE_SIZE - 1) & ~(TRACKER_PAGE_SIZE - 1);
|
||||
CacheRange affected {.address = page_begin, .size = page_end - page_begin};
|
||||
{
|
||||
FaultSafeCacheLock lock(this, m_mutex);
|
||||
bool changed = true;
|
||||
bool changed = true;
|
||||
while (changed) {
|
||||
changed = false;
|
||||
for (const auto& [address, cached]: m_buffers) {
|
||||
@@ -384,6 +380,73 @@ BufferBinding BufferCache::UploadTransient(const void* data, uint64_t size, uint
|
||||
return {owner, owner->Handle(), 0};
|
||||
}
|
||||
|
||||
void BufferCache::InvalidateMemory(uint64_t vaddr, uint64_t size) {
|
||||
if (vaddr == 0 || size == 0 || vaddr >= TRACKER_ADDRESS_SIZE ||
|
||||
size > TRACKER_ADDRESS_SIZE - vaddr) {
|
||||
EXIT("BufferCache: invalid memory-invalidation range\n");
|
||||
}
|
||||
(void)SynchronizeBacking(vaddr, size);
|
||||
if (!HasPageOverlap(vaddr, size)) {
|
||||
return;
|
||||
}
|
||||
m_memory_tracker.InvalidateRegion(vaddr, size,
|
||||
[this, vaddr, size] { ReadMemory(vaddr, size); });
|
||||
}
|
||||
|
||||
void BufferCache::ReadMemory(uint64_t vaddr, uint64_t size) {
|
||||
(void)SynchronizeBacking(vaddr, size);
|
||||
std::vector<DownloadCopy> copies;
|
||||
{
|
||||
FaultSafeCacheLock lock(this, m_mutex);
|
||||
m_memory_tracker.ForEachDownloadRange<false>(
|
||||
vaddr, size,
|
||||
[&](uint64_t address, uint64_t bytes) noexcept {
|
||||
m_memory_tracker.ValidateGpuDirtyPages(m_gpu_modified_ranges, address, bytes,
|
||||
"memory invalidation");
|
||||
},
|
||||
[&](uint64_t address, uint64_t bytes) noexcept {
|
||||
for (const auto range: m_gpu_modified_ranges.Intersections(address, bytes)) {
|
||||
for (uint64_t copied = 0; copied < range.size;) {
|
||||
const auto copy_address = range.address + copied;
|
||||
auto owner = m_buffers.upper_bound(copy_address);
|
||||
if (owner == m_buffers.begin()) {
|
||||
EXIT("BufferCache: invalidation readback has no buffer owner\n");
|
||||
}
|
||||
auto& cached = *std::prev(owner)->second;
|
||||
if (!cached.buffer->IsInBounds(copy_address, 1)) {
|
||||
EXIT(
|
||||
"BufferCache: invalidation readback is outside its buffer owner\n");
|
||||
}
|
||||
const auto copy_size = std::min(range.size - copied,
|
||||
cached.vaddr + cached.size - copy_address);
|
||||
copies.push_back({cached.buffer, cached.buffer->Offset(copy_address),
|
||||
copy_address, copy_size});
|
||||
copied += copy_size;
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
if (copies.empty()) {
|
||||
return;
|
||||
}
|
||||
auto downloads = RecordDownloads(copies);
|
||||
m_scheduler.FinishCurrent();
|
||||
PublishDownloads(downloads);
|
||||
{
|
||||
FaultSafeCacheLock lock(this, m_mutex);
|
||||
m_memory_tracker.ForEachDownloadRange<true>(
|
||||
vaddr, size,
|
||||
[&](uint64_t address, uint64_t bytes) noexcept {
|
||||
m_memory_tracker.ValidateGpuDirtyPages(m_gpu_modified_ranges, address, bytes,
|
||||
"memory invalidation completion");
|
||||
},
|
||||
[](uint64_t, uint64_t) noexcept {});
|
||||
for (const auto& range: downloads) {
|
||||
m_gpu_modified_ranges.Subtract(range.address, range.size);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
bool BufferCache::InvalidateMemory(PageFaultAccess access, uint64_t vaddr, uint64_t size,
|
||||
PageFaultPhase phase) noexcept {
|
||||
const auto page = vaddr & ~(TRACKER_PAGE_SIZE - 1);
|
||||
@@ -544,8 +607,8 @@ void BufferCache::UnmapMemory(uint64_t vaddr, uint64_t size) {
|
||||
m_memory_tracker.ForEachDownloadRange<true>(
|
||||
begin, bytes,
|
||||
[&](uint64_t address, uint64_t download_size) noexcept {
|
||||
m_memory_tracker.ValidateGpuDirtyPages(
|
||||
m_gpu_modified_ranges, address, download_size, "unmap retirement");
|
||||
m_memory_tracker.ValidateGpuDirtyPages(m_gpu_modified_ranges, address,
|
||||
download_size, "unmap retirement");
|
||||
},
|
||||
[](uint64_t, uint64_t) noexcept {});
|
||||
}
|
||||
@@ -722,12 +785,11 @@ ImageBufferSource BufferCache::ObtainBufferForImage(uint64_t vaddr, uint64_t siz
|
||||
|
||||
{
|
||||
FaultSafeCacheLock lock(this, m_mutex);
|
||||
const bool cpu_modified = m_memory_tracker.IsRegionCpuModified(vaddr, size);
|
||||
const bool gpu_modified = m_memory_tracker.IsRegionGpuModified(vaddr, size);
|
||||
const auto dirty = m_gpu_modified_ranges.Intersections(vaddr, size);
|
||||
const bool invalidated =
|
||||
!m_image_invalidated_ranges.Intersections(vaddr, size).empty();
|
||||
const bool requested_gpu_owned = !dirty.empty();
|
||||
const bool cpu_modified = m_memory_tracker.IsRegionCpuModified(vaddr, size);
|
||||
const bool gpu_modified = m_memory_tracker.IsRegionGpuModified(vaddr, size);
|
||||
const auto dirty = m_gpu_modified_ranges.Intersections(vaddr, size);
|
||||
const bool invalidated = !m_image_invalidated_ranges.Intersections(vaddr, size).empty();
|
||||
const bool requested_gpu_owned = !dirty.empty();
|
||||
m_memory_tracker.ValidateGpuDirtyOwnership(m_gpu_modified_ranges, vaddr, size,
|
||||
"image source");
|
||||
|
||||
@@ -807,8 +869,8 @@ ImageBufferSource BufferCache::ObtainBufferForImage(uint64_t vaddr, uint64_t siz
|
||||
}
|
||||
|
||||
FaultSafeCacheLock lock(this, m_mutex);
|
||||
const auto dirty = m_gpu_modified_ranges.Intersections(vaddr, size);
|
||||
const bool invalidated = !m_image_invalidated_ranges.Intersections(vaddr, size).empty();
|
||||
const auto dirty = m_gpu_modified_ranges.Intersections(vaddr, size);
|
||||
const bool invalidated = !m_image_invalidated_ranges.Intersections(vaddr, size).empty();
|
||||
const bool requested_gpu_owned = !dirty.empty();
|
||||
auto owner = find_owner();
|
||||
if (requested_gpu_owned && owner == m_buffers.end()) {
|
||||
@@ -831,9 +893,8 @@ ImageBufferSource BufferCache::ObtainBufferForImage(uint64_t vaddr, uint64_t siz
|
||||
[&]() noexcept {
|
||||
for (const auto& [address, upload_size]: uploads) {
|
||||
cached.buffer->CopyFrom(
|
||||
m_scheduler.Current(), m_staging_buffer,
|
||||
stage_offset + address - stage_address, cached.buffer->Offset(address),
|
||||
upload_size, vk::AccessFlagBits::eHostWrite);
|
||||
m_scheduler.Current(), m_staging_buffer, stage_offset + address - stage_address,
|
||||
cached.buffer->Offset(address), upload_size, vk::AccessFlagBits::eHostWrite);
|
||||
}
|
||||
});
|
||||
DiscardGpuDirtyBytesLocked(vaddr, size, "staged image source transfer");
|
||||
@@ -917,9 +978,8 @@ std::pair<std::shared_ptr<Buffer>, uint64_t> BufferCache::ObtainBufferForImageWr
|
||||
[&]() noexcept {
|
||||
for (const auto& [address, upload_size]: uploads) {
|
||||
cached.buffer->CopyFrom(
|
||||
m_scheduler.Current(), m_staging_buffer,
|
||||
stage_offset + address - stage_address, cached.buffer->Offset(address),
|
||||
upload_size, vk::AccessFlagBits::eHostWrite);
|
||||
m_scheduler.Current(), m_staging_buffer, stage_offset + address - stage_address,
|
||||
cached.buffer->Offset(address), upload_size, vk::AccessFlagBits::eHostWrite);
|
||||
}
|
||||
});
|
||||
return {cached.buffer, cached.buffer->Offset(vaddr)};
|
||||
@@ -946,12 +1006,12 @@ void BufferCache::FillBuffer(uint64_t vaddr, uint64_t size, uint32_t value, bool
|
||||
const auto region = m_texture_cache.QueryRegion(vaddr, size);
|
||||
if (!HasGpuDirtyBytes(vaddr, size) && !region.gpu_image_bytes) {
|
||||
if (region.image_bytes) {
|
||||
m_texture_cache.PrepareHostWrite(vaddr, size);
|
||||
m_texture_cache.InvalidateMemory(vaddr, size);
|
||||
}
|
||||
std::array<uint32_t, 4096> values;
|
||||
values.fill(value);
|
||||
const std::span<const uint8_t> bytes {
|
||||
reinterpret_cast<const uint8_t*>(values.data()), sizeof(values)};
|
||||
const std::span<const uint8_t> bytes {reinterpret_cast<const uint8_t*>(values.data()),
|
||||
sizeof(values)};
|
||||
for (uint64_t offset = 0; offset < size;) {
|
||||
const auto chunk = std::min<uint64_t>(size - offset, bytes.size());
|
||||
WriteHostMemory(vaddr + offset, bytes.first(chunk));
|
||||
@@ -992,7 +1052,7 @@ void BufferCache::CopyBuffer(uint64_t dst_vaddr, uint64_t src_vaddr, uint64_t si
|
||||
if (src_memory) {
|
||||
(void)SynchronizeBacking(src_vaddr, size);
|
||||
}
|
||||
const auto src_region =
|
||||
const auto src_region =
|
||||
src_memory ? m_texture_cache.QueryRegion(src_vaddr, size) : TextureCache::RegionInfo {};
|
||||
const auto dst_region =
|
||||
dst_memory ? m_texture_cache.QueryRegion(dst_vaddr, size) : TextureCache::RegionInfo {};
|
||||
@@ -1004,7 +1064,7 @@ void BufferCache::CopyBuffer(uint64_t dst_vaddr, uint64_t src_vaddr, uint64_t si
|
||||
!HasGpuDirtyBytes(dst_vaddr, size) && !src_region.gpu_image_bytes &&
|
||||
!dst_region.gpu_image_bytes) {
|
||||
if (dst_region.image_bytes) {
|
||||
m_texture_cache.PrepareHostWrite(dst_vaddr, size);
|
||||
m_texture_cache.InvalidateMemory(dst_vaddr, size);
|
||||
}
|
||||
std::array<uint8_t, 64 * 1024> bytes;
|
||||
for (uint64_t offset = 0; offset < size;) {
|
||||
@@ -1037,10 +1097,10 @@ void BufferCache::CopyBuffer(uint64_t dst_vaddr, uint64_t src_vaddr, uint64_t si
|
||||
EXIT("BufferCache: resolved Vulkan copy ranges overlap\n");
|
||||
}
|
||||
auto& source = src.owner != nullptr ? *std::static_pointer_cast<Buffer>(src.owner)
|
||||
: src_gds ? m_gds_buffer
|
||||
: m_stream_buffer;
|
||||
auto& destination = dst.owner != nullptr ? *std::static_pointer_cast<Buffer>(dst.owner)
|
||||
: m_gds_buffer;
|
||||
: src_gds ? m_gds_buffer
|
||||
: m_stream_buffer;
|
||||
auto& destination =
|
||||
dst.owner != nullptr ? *std::static_pointer_cast<Buffer>(dst.owner) : m_gds_buffer;
|
||||
if (source.Handle() != src.buffer || destination.Handle() != dst.buffer) {
|
||||
EXIT("BufferCache: resolved copy owner does not match its Vulkan handle\n");
|
||||
}
|
||||
@@ -1107,7 +1167,7 @@ void BufferCache::BeginBackingPublication(uint64_t vaddr, uint64_t size, uint64_
|
||||
|
||||
void BufferCache::CompleteBackingPublication(uint64_t vaddr, uint64_t size, uint64_t tick) {
|
||||
std::lock_guard lock(m_publication_mutex);
|
||||
const auto publication =
|
||||
const auto publication =
|
||||
std::ranges::find_if(m_pending_backing_publications, [&](const auto& pending) {
|
||||
return pending.address == vaddr && pending.size == size && pending.tick == tick;
|
||||
});
|
||||
@@ -1170,7 +1230,7 @@ void BufferCache::RunGarbageCollector() {
|
||||
const uint64_t age = std::min<uint64_t>(aggressive ? 80 : 160, tick);
|
||||
const size_t limit = aggressive ? 64 : 32;
|
||||
|
||||
std::vector<RetiredBuffer> retires;
|
||||
std::vector<RetiredBuffer> retires;
|
||||
std::vector<std::pair<RetiredBuffer, std::vector<DownloadCopy>>> dirty_retires;
|
||||
{
|
||||
FaultSafeCacheLock lock(this, m_mutex);
|
||||
@@ -1190,8 +1250,8 @@ void BufferCache::RunGarbageCollector() {
|
||||
}
|
||||
for (const auto address: candidates) {
|
||||
auto& cached = *m_buffers.at(address);
|
||||
m_memory_tracker.ValidateGpuDirtyOwnership(
|
||||
m_gpu_modified_ranges, cached.vaddr, cached.size, "garbage collection");
|
||||
m_memory_tracker.ValidateGpuDirtyOwnership(m_gpu_modified_ranges, cached.vaddr,
|
||||
cached.size, "garbage collection");
|
||||
retires.push_back({address, cached.size, cached.buffer});
|
||||
// GC runs immediately before submission. Preserve every source referenced by commands
|
||||
// already recorded in the active batch.
|
||||
@@ -1205,13 +1265,13 @@ void BufferCache::RunGarbageCollector() {
|
||||
m_memory_tracker.ForEachDownloadRange<false>(
|
||||
retire.address, retire.size,
|
||||
[&](uint64_t address, uint64_t size) noexcept {
|
||||
m_memory_tracker.ValidateGpuDirtyPages(
|
||||
m_gpu_modified_ranges, address, size, "garbage collection");
|
||||
m_memory_tracker.ValidateGpuDirtyPages(m_gpu_modified_ranges, address, size,
|
||||
"garbage collection");
|
||||
},
|
||||
[&](uint64_t address, uint64_t size) noexcept {
|
||||
for (const auto range: m_gpu_modified_ranges.Intersections(address, size)) {
|
||||
copies.push_back({retire.owner, range.address - retire.address, range.address,
|
||||
range.size});
|
||||
copies.push_back({retire.owner, range.address - retire.address,
|
||||
range.address, range.size});
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
+11
-11
@@ -49,6 +49,8 @@ public:
|
||||
|
||||
[[nodiscard]] bool InvalidateMemory(PageFaultAccess access, uint64_t vaddr, uint64_t size,
|
||||
PageFaultPhase phase) noexcept;
|
||||
void InvalidateMemory(uint64_t vaddr, uint64_t size);
|
||||
void ReadMemory(uint64_t vaddr, uint64_t size);
|
||||
void UnmapMemory(uint64_t vaddr, uint64_t size);
|
||||
[[nodiscard]] BufferBinding ObtainBuffer(CommandBuffer& command, uint64_t vaddr, uint64_t size,
|
||||
bool is_written = false, bool is_read = true,
|
||||
@@ -71,8 +73,8 @@ public:
|
||||
[[nodiscard]] bool IsRegionCpuModified(uint64_t vaddr, uint64_t size);
|
||||
[[nodiscard]] bool IsRegionGpuModified(uint64_t vaddr, uint64_t size);
|
||||
void InvalidateImageAliases(uint64_t vaddr, uint64_t size);
|
||||
void BeginBackingPublication(uint64_t vaddr, uint64_t size, uint64_t tick);
|
||||
void CompleteBackingPublication(uint64_t vaddr, uint64_t size, uint64_t tick);
|
||||
void BeginBackingPublication(uint64_t vaddr, uint64_t size, uint64_t tick);
|
||||
void CompleteBackingPublication(uint64_t vaddr, uint64_t size, uint64_t tick);
|
||||
[[nodiscard]] bool SynchronizeBacking(uint64_t vaddr, uint64_t size);
|
||||
void PublishImageBuffer(uint64_t vaddr, uint64_t size);
|
||||
void ValidateGpuAccess(uint64_t vaddr, uint64_t size, bool is_read, bool is_written) const;
|
||||
@@ -91,23 +93,21 @@ private:
|
||||
struct RetiredBuffer;
|
||||
struct FaultReadback;
|
||||
struct PendingBackingPublication;
|
||||
static constexpr uint64_t DOWNLOAD_ALIGNMENT = 64;
|
||||
[[nodiscard]] static uint64_t AlignDown(uint64_t value) noexcept;
|
||||
[[nodiscard]] static uint64_t AlignUp(uint64_t value);
|
||||
static constexpr uint64_t DOWNLOAD_ALIGNMENT = 64;
|
||||
[[nodiscard]] static uint64_t AlignDown(uint64_t value) noexcept;
|
||||
[[nodiscard]] static uint64_t AlignUp(uint64_t value);
|
||||
[[nodiscard]] static constexpr uint64_t AlignDownload(uint64_t size) noexcept {
|
||||
return (size + DOWNLOAD_ALIGNMENT - 1) & ~(DOWNLOAD_ALIGNMENT - 1);
|
||||
}
|
||||
[[nodiscard]] static bool PageOverlaps(uint64_t left, uint64_t left_size, uint64_t right,
|
||||
uint64_t right_size) noexcept;
|
||||
[[nodiscard]] static std::pair<uint64_t, uint64_t>
|
||||
DownloadEnvelope(const DownloadCopy& copy);
|
||||
[[nodiscard]] static bool ResolveOverlap(CacheRange& merged, CacheRange candidate) noexcept;
|
||||
uint64_t right_size) noexcept;
|
||||
[[nodiscard]] static std::pair<uint64_t, uint64_t> DownloadEnvelope(const DownloadCopy& copy);
|
||||
[[nodiscard]] static bool ResolveOverlap(CacheRange& merged, CacheRange candidate) noexcept;
|
||||
void Upload(CommandBuffer& command, Buffer& destination, uint64_t destination_offset,
|
||||
const void* source, uint64_t size);
|
||||
[[nodiscard]] CachedBuffer& GetOrCreateBuffer(CommandBuffer& command, uint64_t vaddr,
|
||||
uint64_t size);
|
||||
[[nodiscard]] std::vector<DownloadRange>
|
||||
RecordDownloads(std::span<const DownloadCopy> copies);
|
||||
[[nodiscard]] std::vector<DownloadRange> RecordDownloads(std::span<const DownloadCopy> copies);
|
||||
void PublishDownloads(std::span<const DownloadRange> downloads);
|
||||
void QueueGarbageDownload(std::span<const DownloadCopy> copies, RetiredBuffer retire);
|
||||
void RefreshInvalidatedRanges(CommandBuffer& command, CachedBuffer& cached, uint64_t vaddr,
|
||||
|
||||
+33
-20
@@ -36,7 +36,8 @@ bool GpuResourceManager::InvalidateMemory(PageFaultAccess access, uint64_t vaddr
|
||||
}
|
||||
|
||||
bool GpuResourceManager::HandleFault(PageFaultAccess access, uint64_t fault_vaddr) noexcept {
|
||||
if (!m_page_manager.IsMapped(fault_vaddr, 1)) {
|
||||
constexpr uint64_t fault_size = 8;
|
||||
if (!IsMapped(fault_vaddr, fault_size)) {
|
||||
return false;
|
||||
}
|
||||
if (CommandScheduler::InDeferredOperation()) {
|
||||
@@ -47,10 +48,15 @@ bool GpuResourceManager::HandleFault(PageFaultAccess access, uint64_t fault_vadd
|
||||
bool handled = false;
|
||||
const auto resolve = [this, access, fault_vaddr, &handled](CommandProcessor& cp) {
|
||||
cp.BeginReadbackTransaction();
|
||||
(void)m_buffer_cache.SynchronizeBacking(fault_vaddr, 1);
|
||||
{
|
||||
ResourceMutex::FaultScope fault(m_resource_mutex);
|
||||
handled = m_page_manager.HandleFault(access, fault_vaddr);
|
||||
if (access == PageFaultAccess::Write) {
|
||||
m_buffer_cache.InvalidateMemory(fault_vaddr, fault_size);
|
||||
m_texture_cache.InvalidateMemory(fault_vaddr, fault_size);
|
||||
} else {
|
||||
m_buffer_cache.ReadMemory(fault_vaddr, fault_size);
|
||||
}
|
||||
handled = true;
|
||||
}
|
||||
cp.EndReadbackTransaction();
|
||||
};
|
||||
@@ -68,47 +74,52 @@ bool GpuResourceManager::HandleFault(PageFaultAccess access, uint64_t fault_vadd
|
||||
return handled;
|
||||
}
|
||||
|
||||
void GpuResourceManager::PrepareHostWrite(uint64_t vaddr, uint64_t size) {
|
||||
if (!m_page_manager.HasAnyMapping(vaddr, size)) {
|
||||
return;
|
||||
bool GpuResourceManager::InvalidateMemory(uint64_t vaddr, uint64_t size) {
|
||||
if (!IsMapped(vaddr, size)) {
|
||||
return false;
|
||||
}
|
||||
if (CommandScheduler::InDeferredOperation()) {
|
||||
EXIT("unsupported host write from an asynchronous GPU completion, addr=0x%016" PRIx64
|
||||
" size=0x%016" PRIx64 "\n",
|
||||
EXIT("unsupported memory invalidation from an asynchronous GPU completion, "
|
||||
"addr=0x%016" PRIx64 " size=0x%016" PRIx64 "\n",
|
||||
vaddr, size);
|
||||
}
|
||||
const auto handle_range = [this, vaddr, size] {
|
||||
if (!m_page_manager.HandleWriteRange(vaddr, size)) {
|
||||
EXIT("failed to prepare host write, addr=0x%016" PRIx64 " size=0x%016" PRIx64 "\n",
|
||||
vaddr, size);
|
||||
}
|
||||
};
|
||||
const auto resolve = [this, &handle_range](CommandProcessor& cp) {
|
||||
const auto resolve = [this, vaddr, size](CommandProcessor& cp) {
|
||||
cp.BeginReadbackTransaction();
|
||||
{
|
||||
ResourceMutex::FaultScope fault(m_resource_mutex);
|
||||
handle_range();
|
||||
m_buffer_cache.InvalidateMemory(vaddr, size);
|
||||
m_texture_cache.InvalidateMemory(vaddr, size);
|
||||
}
|
||||
cp.EndReadbackTransaction();
|
||||
};
|
||||
if (auto* cp = Gpu::CurrentCommandProcessor(); cp != nullptr) {
|
||||
resolve(*cp);
|
||||
return;
|
||||
return true;
|
||||
}
|
||||
if (m_resource_mutex.IsOwnedByCurrentThread()) {
|
||||
EXIT("unsupported host write from a pre-owned resource transaction, addr=0x%016" PRIx64
|
||||
" size=0x%016" PRIx64 "\n",
|
||||
EXIT("unsupported memory invalidation from a pre-owned resource transaction, "
|
||||
"addr=0x%016" PRIx64 " size=0x%016" PRIx64 "\n",
|
||||
vaddr, size);
|
||||
}
|
||||
EXIT_IF(m_gpu == nullptr);
|
||||
m_gpu->SendCommandSyncWithProcessor(resolve);
|
||||
return true;
|
||||
}
|
||||
|
||||
bool GpuResourceManager::IsMapped(uint64_t vaddr, uint64_t size) const noexcept {
|
||||
return m_page_manager.IsMapped(vaddr, size);
|
||||
if (vaddr == 0 || size == 0 || vaddr >= TRACKER_ADDRESS_SIZE ||
|
||||
size > TRACKER_ADDRESS_SIZE - vaddr) {
|
||||
return false;
|
||||
}
|
||||
std::shared_lock lock(m_mapped_ranges_mutex);
|
||||
return m_mapped_ranges.Contains(vaddr, size);
|
||||
}
|
||||
|
||||
void GpuResourceManager::MapMemory(uint64_t vaddr, uint64_t size, GpuAccess access) {
|
||||
{
|
||||
std::lock_guard lock(m_mapped_ranges_mutex);
|
||||
m_mapped_ranges.Add(vaddr, size);
|
||||
}
|
||||
m_page_manager.OnGpuMap(vaddr, size, access);
|
||||
}
|
||||
|
||||
@@ -120,6 +131,8 @@ void GpuResourceManager::UnmapMemory(uint64_t vaddr, uint64_t size, GpuAccess ac
|
||||
m_texture_cache.UnmapMemory(vaddr, size);
|
||||
m_buffer_cache.UnmapMemory(vaddr, size);
|
||||
m_page_manager.OnGpuUnmap(vaddr, size, access);
|
||||
std::lock_guard lock(m_mapped_ranges_mutex);
|
||||
m_mapped_ranges.Subtract(vaddr, size);
|
||||
};
|
||||
if (m_gpu == nullptr) {
|
||||
if (m_resource_mutex.IsOwnedByCurrentThread()) {
|
||||
|
||||
@@ -9,6 +9,7 @@
|
||||
#include "graphics/host_gpu/renderer/cache/textureCache.h"
|
||||
|
||||
#include <cstdint>
|
||||
#include <shared_mutex>
|
||||
|
||||
namespace Libs::Graphics {
|
||||
|
||||
@@ -26,7 +27,7 @@ public:
|
||||
void SetGpu(Gpu* gpu) noexcept { m_gpu = gpu; }
|
||||
|
||||
[[nodiscard]] bool HandleFault(PageFaultAccess access, uint64_t fault_vaddr) noexcept;
|
||||
void PrepareHostWrite(uint64_t vaddr, uint64_t size);
|
||||
[[nodiscard]] bool InvalidateMemory(uint64_t vaddr, uint64_t size);
|
||||
[[nodiscard]] bool IsMapped(uint64_t vaddr, uint64_t size) const noexcept;
|
||||
void MapMemory(uint64_t vaddr, uint64_t size, GpuAccess access);
|
||||
void UnmapMemory(uint64_t vaddr, uint64_t size, GpuAccess access);
|
||||
@@ -38,11 +39,13 @@ private:
|
||||
[[nodiscard]] bool InvalidateMemory(PageFaultAccess access, uint64_t vaddr, uint64_t size,
|
||||
PageFaultPhase phase) noexcept;
|
||||
|
||||
PageManager m_page_manager;
|
||||
ResourceMutex m_resource_mutex;
|
||||
BufferCache m_buffer_cache;
|
||||
TextureCache m_texture_cache;
|
||||
Gpu* m_gpu = nullptr;
|
||||
PageManager m_page_manager;
|
||||
ResourceMutex m_resource_mutex;
|
||||
BufferCache m_buffer_cache;
|
||||
TextureCache m_texture_cache;
|
||||
mutable std::shared_mutex m_mapped_ranges_mutex;
|
||||
RangeSet m_mapped_ranges;
|
||||
Gpu* m_gpu = nullptr;
|
||||
};
|
||||
|
||||
} // namespace Libs::Graphics
|
||||
|
||||
+2
-2
@@ -1439,9 +1439,9 @@ bool TextureCache::ClearImageFromBuffer(CommandBuffer& command, uint64_t address
|
||||
return true;
|
||||
}
|
||||
|
||||
void TextureCache::PrepareHostWrite(uint64_t address, uint64_t size) {
|
||||
void TextureCache::InvalidateMemory(uint64_t address, uint64_t size) {
|
||||
if (!GuestRange {address, size}.Valid()) {
|
||||
EXIT("TextureCache: invalid host-write range\n");
|
||||
EXIT("TextureCache: invalid memory-invalidation range\n");
|
||||
}
|
||||
CacheLock lock(*this, m_lock);
|
||||
InvalidateCpuAliases(address, size);
|
||||
|
||||
+1
-1
@@ -65,7 +65,7 @@ public:
|
||||
|
||||
[[nodiscard]] bool ClearImageFromBuffer(CommandBuffer& command, uint64_t address, uint64_t size,
|
||||
uint32_t packed_clear);
|
||||
void PrepareHostWrite(uint64_t address, uint64_t size);
|
||||
void InvalidateMemory(uint64_t address, uint64_t size);
|
||||
[[nodiscard]] bool SynchronizeImageToBuffer(uint64_t address, uint64_t size);
|
||||
[[nodiscard]] bool InvalidateMemoryFromGPU(uint64_t address, uint64_t size,
|
||||
bool formatted_buffer_write = false);
|
||||
|
||||
@@ -103,10 +103,7 @@ IsSupportedSampledDepthUintResource(const ShaderRecompiler::IR::ImageResource& r
|
||||
|
||||
inline void ValidateStorageColorView(vk::Format image_format, vk::Format view_format,
|
||||
uint32_t swizzle) noexcept {
|
||||
const auto srgb_view = SrgbStorageViewFormat(image_format);
|
||||
const bool srgb_storage_view =
|
||||
srgb_view != vk::Format::eUndefined && view_format == srgb_view;
|
||||
if ((image_format != view_format && !srgb_storage_view) ||
|
||||
if (!ImageViewOps::FormatsCompatible(image_format, view_format) ||
|
||||
!IsValidImageSwizzle(swizzle)) {
|
||||
UnsupportedColorView("storage", image_format, view_format, swizzle);
|
||||
}
|
||||
@@ -122,7 +119,10 @@ IsSupportedStorageImageResource(const ShaderRecompiler::IR::ImageResource& resou
|
||||
resource.dimension == ShaderRecompiler::Decoder::ImageDimension::Dim3D ||
|
||||
resource.dimension == ShaderRecompiler::Decoder::ImageDimension::Dim2DArray) &&
|
||||
resource.mip_mode == ShaderRecompiler::IR::ImageMipMode::None && resource.written &&
|
||||
!resource.atomic && !resource.depth_compare;
|
||||
(!resource.atomic ||
|
||||
(resource.kind == ShaderRecompiler::IR::ResourceKind::StorageImageUint &&
|
||||
resource.read)) &&
|
||||
!resource.depth_compare;
|
||||
}
|
||||
|
||||
inline void
|
||||
|
||||
@@ -14,13 +14,13 @@
|
||||
#include "graphics/guest_gpu/tile.h"
|
||||
#include "graphics/host_gpu/graphicContext.h"
|
||||
#include "graphics/host_gpu/hostMemory.h"
|
||||
#include "graphics/host_gpu/renderer/image/textureCommon.h"
|
||||
#include "graphics/host_gpu/renderer/debug.h"
|
||||
#include "graphics/host_gpu/renderer/pipeline/descriptorCache.h"
|
||||
#include "graphics/host_gpu/renderer/image/imageView.h"
|
||||
#include "graphics/host_gpu/renderer/image/textureCommon.h"
|
||||
#include "graphics/host_gpu/renderer/pipeline/descriptorCache.h"
|
||||
#include "graphics/host_gpu/renderer/pipeline/shaderResourceBarrier.h"
|
||||
#include "graphics/host_gpu/renderer/render.h"
|
||||
#include "graphics/host_gpu/renderer/renderContext.h"
|
||||
#include "graphics/host_gpu/renderer/pipeline/shaderResourceBarrier.h"
|
||||
#include "graphics/host_gpu/vma.h"
|
||||
#include "graphics/host_gpu/vulkanCommon.h"
|
||||
#include "graphics/shader/recompiler/ir/BindingLayout.h"
|
||||
@@ -239,11 +239,11 @@ bool IsSupportedDepthTextureEncoding(const ShaderTextureResource& descriptor, co
|
||||
const uint32_t field3_expected =
|
||||
(descriptor.Type() << 28u) | field3_common | descriptor.DstSelXYZW();
|
||||
const uint32_t field4_expected = descriptor.Depth() | (descriptor.BaseArray5() << 16u);
|
||||
const bool common = (descriptor.fields[1] & field1_reserved_mask) == 0 &&
|
||||
(descriptor.fields[2] & field2_reserved_mask) == 0 &&
|
||||
descriptor.fields[3] == field3_expected &&
|
||||
descriptor.fields[4] == field4_expected &&
|
||||
descriptor.fields[5] == field5_expected;
|
||||
const bool common = (descriptor.fields[1] & field1_reserved_mask) == 0 &&
|
||||
(descriptor.fields[2] & field2_reserved_mask) == 0 &&
|
||||
descriptor.fields[3] == field3_expected &&
|
||||
descriptor.fields[4] == field4_expected &&
|
||||
descriptor.fields[5] == field5_expected;
|
||||
if (!common || (descriptor.fields[6] == 0 && descriptor.fields[7] != 0)) {
|
||||
return false;
|
||||
}
|
||||
@@ -318,8 +318,8 @@ static bool IsSupportedStorageTextureDescriptor(const ShaderRecompiler::IR::Imag
|
||||
const bool valid_2d_slice =
|
||||
(is_color_2d && descriptor.Depth() == 0 && descriptor.BaseArray5() == 0) ||
|
||||
(is_color_2d_array && descriptor.BaseArray5() <= descriptor.Depth());
|
||||
const bool is_2d = resource.dimension == ShaderRecompiler::Decoder::ImageDimension::Dim2D &&
|
||||
valid_2d_slice;
|
||||
const bool is_2d =
|
||||
resource.dimension == ShaderRecompiler::Decoder::ImageDimension::Dim2D && valid_2d_slice;
|
||||
const bool is_2d_array =
|
||||
resource.dimension == ShaderRecompiler::Decoder::ImageDimension::Dim2DArray &&
|
||||
is_color_2d_array && descriptor.BaseArray5() <= descriptor.Depth();
|
||||
@@ -340,13 +340,13 @@ static bool IsSupportedStorageTextureDescriptor(const ShaderRecompiler::IR::Imag
|
||||
const bool supported_tile = tile == Prospero::GpuEnumValue(Prospero::TileMode::kLinear) ||
|
||||
tile == Prospero::GpuEnumValue(Prospero::TileMode::kRenderTarget) ||
|
||||
supported_depth_tile || supported_standard_tile;
|
||||
const auto swizzle = descriptor.DstSelXYZW();
|
||||
const bool supported_swizzle =
|
||||
IsValidImageSwizzle(descriptor.DstSelXYZW()) &&
|
||||
(descriptor.DstSelXYZW() == DstSel(4, 5, 6, 7) || !resource.read);
|
||||
IsValidImageSwizzle(swizzle) &&
|
||||
(swizzle == DstSel(4, 5, 6, 7) || !resource.read || resource.atomic);
|
||||
const bool supported_mip_view = descriptor.BaseLevel() == 0 || is_1d || is_2d;
|
||||
return (is_1d || is_1d_array || is_2d || is_2d_array || is_3d) && supported_tile &&
|
||||
supported_mip_view &&
|
||||
descriptor.BaseLevel() == descriptor.LastLevel() &&
|
||||
supported_mip_view && descriptor.BaseLevel() == descriptor.LastLevel() &&
|
||||
descriptor.LastLevel() <= descriptor.MaxMip() && descriptor.MinLod() == 0 &&
|
||||
supported_swizzle && descriptor.BCSwizzle() == 0 && !descriptor.MsaaDepth();
|
||||
}
|
||||
@@ -377,8 +377,10 @@ void ValidateStorageTexture(const ShaderRecompiler::IR::ImageResource& resource,
|
||||
const bool encoding_ok = IsSupportedStorageTextureEncoding(descriptor);
|
||||
const bool uint_resource =
|
||||
resource.kind == ShaderRecompiler::IR::ResourceKind::StorageImageUint;
|
||||
const bool format_ok = Prospero::IsSupportedTextureFormat(format) &&
|
||||
uint_resource == Prospero::IsUintTextureFormat(format);
|
||||
const bool format_ok =
|
||||
Prospero::IsSupportedTextureFormat(format) &&
|
||||
uint_resource == Prospero::IsUintTextureFormat(format) &&
|
||||
(!resource.atomic || format == Prospero::GpuEnumValue(Prospero::BufferFormat::k32UInt));
|
||||
if (resource_ok && descriptor_ok && encoding_ok && format_ok && size != 0) {
|
||||
return;
|
||||
}
|
||||
@@ -618,12 +620,12 @@ RenderExecutor::ResolveTexture(const ShaderRecompiler::IR::ImageResource& reso
|
||||
resource.written);
|
||||
}
|
||||
|
||||
const auto pixel_format = TextureGetFormat(format);
|
||||
const auto pixel_format = TextureGetFormat(format);
|
||||
const auto storage_view_format = SrgbStorageViewFormat(pixel_format);
|
||||
const auto view_format =
|
||||
storage && storage_view_format != vk::Format::eUndefined ? storage_view_format
|
||||
: pixel_format;
|
||||
const auto block_bytes = Prospero::BlockCompressedBytesPerBlock(format);
|
||||
const auto view_format = storage && storage_view_format != vk::Format::eUndefined
|
||||
? storage_view_format
|
||||
: pixel_format;
|
||||
const auto block_bytes = Prospero::BlockCompressedBytesPerBlock(format);
|
||||
TextureCache::ImageDesc desc {};
|
||||
desc.info.data = {address, size.size};
|
||||
desc.info.pixel_format = pixel_format;
|
||||
|
||||
@@ -18,15 +18,16 @@
|
||||
#include "graphics/host_gpu/renderer/depthRenderTarget.h"
|
||||
#include "graphics/host_gpu/renderer/pipeline/descriptorCache.h"
|
||||
#include "graphics/host_gpu/renderer/pipeline/pipelineCache.h"
|
||||
#include "graphics/host_gpu/renderer/render.h"
|
||||
#include "graphics/host_gpu/renderer/renderContext.h"
|
||||
#include "graphics/host_gpu/renderer/pipeline/shaderResourceBarrier.h"
|
||||
#include "graphics/host_gpu/renderer/pipeline/shaderSubgroup.h"
|
||||
#include "graphics/host_gpu/renderer/render.h"
|
||||
#include "graphics/host_gpu/renderer/renderContext.h"
|
||||
#include "graphics/host_gpu/vulkanCommon.h"
|
||||
#include "graphics/shader/recompiler/ir/ResourceMaterialization.h"
|
||||
#include "graphics/shader/recompiler/ir/ShaderIR.h"
|
||||
#include "graphics/shader/shader.h"
|
||||
#include "kernel/eventQueue.h"
|
||||
#include "kernel/memory.h"
|
||||
#include "kernel/pthread.h"
|
||||
#include "libs/errno.h"
|
||||
|
||||
@@ -221,8 +222,7 @@ static void LogDrawTargetState(const char* draw_name, const RenderColorInfo& col
|
||||
LogMrtState(draw_name, buffer, ps_input_info);
|
||||
}
|
||||
|
||||
static void LogDrawInputState(const RenderCommandBuffer& buffer,
|
||||
const RenderColorInfo& color,
|
||||
static void LogDrawInputState(const RenderCommandBuffer& buffer, const RenderColorInfo& color,
|
||||
const ShaderVertexInputInfo& vs_input_info,
|
||||
uint32_t index_type_and_size, uint32_t index_count,
|
||||
const void* index_addr) {
|
||||
@@ -499,9 +499,9 @@ struct DrawCallInfo {
|
||||
};
|
||||
|
||||
RenderState RenderExecutor::AcquireRenderTargets(CommandBuffer& buffer, RenderColorInfo* colors,
|
||||
uint32_t color_count, RenderDepthInfo& depth) {
|
||||
uint32_t color_count, RenderDepthInfo& depth) {
|
||||
EXIT_IF(colors == nullptr || color_count > RENDER_COLOR_ATTACHMENTS_MAX);
|
||||
auto& cache = m_context.GetTextureCache();
|
||||
auto& cache = m_context.GetTextureCache();
|
||||
RenderState state {};
|
||||
state.width = std::numeric_limits<uint32_t>::max();
|
||||
state.height = std::numeric_limits<uint32_t>::max();
|
||||
@@ -512,8 +512,7 @@ RenderState RenderExecutor::AcquireRenderTargets(CommandBuffer& buffer, RenderCo
|
||||
auto& target = colors[i];
|
||||
EXIT_IF(!target.image_id);
|
||||
const auto old_image = cache.ResolveOwner(target.image_id);
|
||||
if (old_image == nullptr ||
|
||||
(!old_image->registered && !old_image->info.data.Empty()) ||
|
||||
if (old_image == nullptr || (!old_image->registered && !old_image->info.data.Empty()) ||
|
||||
old_image->binding.needs_rebind) {
|
||||
if (old_image != nullptr) {
|
||||
old_image->binding = {};
|
||||
@@ -522,7 +521,7 @@ RenderState RenderExecutor::AcquireRenderTargets(CommandBuffer& buffer, RenderCo
|
||||
BindRenderTarget(target.image_id);
|
||||
}
|
||||
target.image_view = cache.FindRenderTarget(target.image_id, target.desc);
|
||||
auto& image = cache.GetImage(target.image_id);
|
||||
auto& image = cache.GetImage(target.image_id);
|
||||
EXIT_IF(image.backing.samples != target.samples || target.image_view == nullptr);
|
||||
if (attachment_samples == 0) {
|
||||
attachment_samples = target.samples;
|
||||
@@ -530,20 +529,19 @@ RenderState RenderExecutor::AcquireRenderTargets(CommandBuffer& buffer, RenderCo
|
||||
EXIT("mixed color attachment sample counts are unsupported: %u and %u\n",
|
||||
attachment_samples, target.samples);
|
||||
}
|
||||
const auto& view = target.desc.view_info;
|
||||
const auto layout =
|
||||
image.binding.is_bound ? vk::ImageLayout::eGeneral
|
||||
: vk::ImageLayout::eColorAttachmentOptimal;
|
||||
const auto& view = target.desc.view_info;
|
||||
const auto layout = image.binding.is_bound ? vk::ImageLayout::eGeneral
|
||||
: vk::ImageLayout::eColorAttachmentOptimal;
|
||||
image.Transit(layout,
|
||||
vk::AccessFlagBits2::eColorAttachmentRead |
|
||||
vk::AccessFlagBits2::eColorAttachmentWrite,
|
||||
ImageSubresourceRange {view.base_level, view.level_count, view.base_layer,
|
||||
view.layer_count},
|
||||
buffer.Handle());
|
||||
state.width = std::min(state.width, target.extent.width);
|
||||
state.height = std::min(state.height, target.extent.height);
|
||||
state.num_layers = std::min(state.num_layers, view.layer_count);
|
||||
auto& attachment = state.color_attachments[i];
|
||||
state.width = std::min(state.width, target.extent.width);
|
||||
state.height = std::min(state.height, target.extent.height);
|
||||
state.num_layers = std::min(state.num_layers, view.layer_count);
|
||||
auto& attachment = state.color_attachments[i];
|
||||
attachment.image_view = target.image_view;
|
||||
attachment.image_layout = layout;
|
||||
attachment.clear_value = target.color_clear_value.uint32;
|
||||
@@ -561,8 +559,7 @@ RenderState RenderExecutor::AcquireRenderTargets(CommandBuffer& buffer, RenderCo
|
||||
depth.depth_meta_clear_enable =
|
||||
depth.htile &&
|
||||
cache.IsMetaCleared(depth.htile_buffer_vaddr, depth.desc.view_info.base_layer);
|
||||
depth.depth_load_clear_enable =
|
||||
depth.depth_clear_enable || depth.depth_meta_clear_enable;
|
||||
depth.depth_load_clear_enable = depth.depth_clear_enable || depth.depth_meta_clear_enable;
|
||||
if (depth.depth_meta_clear_enable &&
|
||||
!cache.TouchMeta(depth.htile_buffer_vaddr, depth.desc.view_info.base_layer, false)) {
|
||||
EXIT("failed to consume HTile clear state\n");
|
||||
@@ -572,12 +569,12 @@ RenderState RenderExecutor::AcquireRenderTargets(CommandBuffer& buffer, RenderCo
|
||||
if (attachment_samples == 0) {
|
||||
attachment_samples = depth.samples;
|
||||
} else if (attachment_samples != depth.samples) {
|
||||
EXIT("mixed color/depth sample counts are unsupported: %u and %u\n",
|
||||
attachment_samples, depth.samples);
|
||||
EXIT("mixed color/depth sample counts are unsupported: %u and %u\n", attachment_samples,
|
||||
depth.samples);
|
||||
}
|
||||
const auto layout = depth_attachment_layout(depth);
|
||||
const auto writes = depth.AttachmentWriteAspects();
|
||||
auto access = vk::AccessFlags2 {vk::AccessFlagBits2::eDepthStencilAttachmentRead};
|
||||
auto access = vk::AccessFlags2 {vk::AccessFlagBits2::eDepthStencilAttachmentRead};
|
||||
if (writes) {
|
||||
access |= vk::AccessFlagBits2::eDepthStencilAttachmentWrite;
|
||||
}
|
||||
@@ -586,21 +583,19 @@ RenderState RenderExecutor::AcquireRenderTargets(CommandBuffer& buffer, RenderCo
|
||||
ImageSubresourceRange {view.base_level, view.level_count, view.base_layer,
|
||||
view.layer_count},
|
||||
buffer.Handle());
|
||||
state.width = std::min(state.width, depth.width);
|
||||
state.height = std::min(state.height, depth.height);
|
||||
state.num_layers = std::min(state.num_layers, view.layer_count);
|
||||
const auto aspects = ImageViewOps::DepthAspectMask(depth.format);
|
||||
auto& attachment = state.depth_stencil_attachment;
|
||||
state.width = std::min(state.width, depth.width);
|
||||
state.height = std::min(state.height, depth.height);
|
||||
state.num_layers = std::min(state.num_layers, view.layer_count);
|
||||
const auto aspects = ImageViewOps::DepthAspectMask(depth.format);
|
||||
auto& attachment = state.depth_stencil_attachment;
|
||||
attachment.image_view = depth.image_view;
|
||||
attachment.image_layout = layout;
|
||||
attachment.clear_value[0] = std::bit_cast<uint32_t>(depth.depth_clear_value);
|
||||
attachment.clear_value[1] = depth.stencil_clear_value;
|
||||
attachment.has_depth =
|
||||
static_cast<bool>(aspects & vk::ImageAspectFlagBits::eDepth);
|
||||
attachment.depth_clear = depth.depth_load_clear_enable;
|
||||
attachment.has_stencil =
|
||||
static_cast<bool>(aspects & vk::ImageAspectFlagBits::eStencil);
|
||||
attachment.stencil_clear = depth.stencil_clear_enable;
|
||||
attachment.has_depth = static_cast<bool>(aspects & vk::ImageAspectFlagBits::eDepth);
|
||||
attachment.depth_clear = depth.depth_load_clear_enable;
|
||||
attachment.has_stencil = static_cast<bool>(aspects & vk::ImageAspectFlagBits::eStencil);
|
||||
attachment.stencil_clear = depth.stencil_clear_enable;
|
||||
}
|
||||
if (attachment_samples == 0 ||
|
||||
vulkan_sample_count(attachment_samples) == vk::SampleCountFlagBits {}) {
|
||||
@@ -685,6 +680,85 @@ static uint64_t VertexBufferDescriptorSize(const ShaderVertexInputBuffer& buffer
|
||||
: buffer.num_records);
|
||||
}
|
||||
|
||||
struct VertexBufferRange {
|
||||
uint64_t base_address = 0;
|
||||
uint64_t requested_end = 0;
|
||||
uint64_t acquired_end = 0;
|
||||
BufferBinding binding;
|
||||
|
||||
[[nodiscard]] uint64_t RequestedSize() const { return requested_end - base_address; }
|
||||
};
|
||||
|
||||
static std::vector<BufferBinding> AcquireVertexBuffers(RenderCommandBuffer& buffer,
|
||||
const ShaderVertexInputInfo& vs_input_info) {
|
||||
// Collect the non-empty guest vertex ranges.
|
||||
std::vector<VertexBufferRange> ranges;
|
||||
ranges.reserve(vs_input_info.buffers_num);
|
||||
for (int i = 0; i < vs_input_info.buffers_num; i++) {
|
||||
const auto& vertex = vs_input_info.buffers[i];
|
||||
const auto size = VertexBufferDescriptorSize(vertex);
|
||||
if (size == 0) {
|
||||
continue;
|
||||
}
|
||||
if (vertex.addr == 0 || size > UINT64_MAX - vertex.addr) {
|
||||
EXIT("invalid vertex buffer range: addr=0x%016" PRIx64 " size=0x%016" PRIx64 "\n",
|
||||
vertex.addr, size);
|
||||
}
|
||||
ranges.push_back({vertex.addr, vertex.addr + size});
|
||||
}
|
||||
|
||||
std::ranges::sort(ranges, [](const VertexBufferRange& left, const VertexBufferRange& right) {
|
||||
return left.base_address < right.base_address;
|
||||
});
|
||||
|
||||
// Merge overlapping or touching ranges before acquiring host buffers.
|
||||
std::vector<VertexBufferRange> merged_ranges;
|
||||
merged_ranges.reserve(ranges.size());
|
||||
for (const auto& range: ranges) {
|
||||
if (!merged_ranges.empty() && merged_ranges.back().requested_end >= range.base_address) {
|
||||
merged_ranges.back().requested_end =
|
||||
std::max(merged_ranges.back().requested_end, range.requested_end);
|
||||
continue;
|
||||
}
|
||||
merged_ranges.push_back(range);
|
||||
}
|
||||
|
||||
auto& cache = buffer.GetContext().GetBufferCache();
|
||||
for (auto& range: merged_ranges) {
|
||||
// PPSA20298
|
||||
const auto size =
|
||||
Libs::LibKernel::Memory::ClampRangeSize(range.base_address, range.RequestedSize());
|
||||
range.acquired_end = range.base_address + size;
|
||||
range.binding = cache.ObtainBuffer(buffer, range.base_address, size);
|
||||
}
|
||||
|
||||
// Rebuild slot bindings, offsetting non-empty slots into their acquired merged range.
|
||||
std::vector<BufferBinding> bindings;
|
||||
bindings.reserve(vs_input_info.buffers_num);
|
||||
for (int i = 0; i < vs_input_info.buffers_num; i++) {
|
||||
const auto& vertex = vs_input_info.buffers[i];
|
||||
const auto size = VertexBufferDescriptorSize(vertex);
|
||||
if (size == 0) {
|
||||
auto owner = cache.ObtainNullBuffer();
|
||||
bindings.push_back({owner, owner->Handle(), 0});
|
||||
continue;
|
||||
}
|
||||
|
||||
const auto range = std::ranges::find_if(merged_ranges, [&](const VertexBufferRange& value) {
|
||||
return vertex.addr >= value.base_address && vertex.addr < value.acquired_end;
|
||||
});
|
||||
if (range == merged_ranges.end()) {
|
||||
EXIT("vertex buffer address is outside the acquired range: addr=0x%016" PRIx64 "\n",
|
||||
vertex.addr);
|
||||
}
|
||||
|
||||
auto binding = range->binding;
|
||||
binding.offset += vertex.addr - range->base_address;
|
||||
bindings.push_back(std::move(binding));
|
||||
}
|
||||
return bindings;
|
||||
}
|
||||
|
||||
static void SetDrawDebugPhase(RenderCommandBuffer& buffer, uint64_t submit_id,
|
||||
const DrawCallInfo& draw, uint32_t phase) {
|
||||
EXIT_IF(draw.name == nullptr);
|
||||
@@ -736,9 +810,9 @@ static bool GetDrawTopology(const HW::UserConfig& ucfg, bool auto_draw, bool use
|
||||
}
|
||||
|
||||
bool RenderExecutor::PrepareDrawRenderState(uint64_t submit_id, RenderCommandBuffer& buffer,
|
||||
const DrawCallInfo& draw,
|
||||
uint32_t render_target_slice_offset,
|
||||
bool log_setup_phases, DrawRenderState& state) {
|
||||
const DrawCallInfo& draw,
|
||||
uint32_t render_target_slice_offset,
|
||||
bool log_setup_phases, DrawRenderState& state) {
|
||||
EXIT_IF(draw.name == nullptr);
|
||||
auto& ctx = buffer.GetRegisters();
|
||||
|
||||
@@ -823,37 +897,13 @@ static std::vector<BufferBinding> PrepareVertexBuffers(uint64_t
|
||||
(void)submit_id;
|
||||
|
||||
LogDrawPhase(draw.name, "PrepareVertexBuffers");
|
||||
std::vector<BufferBinding> bindings;
|
||||
bindings.reserve(vs_input_info.buffers_num);
|
||||
for (int i = 0; i < vs_input_info.buffers_num; i++) {
|
||||
const auto& b = vs_input_info.buffers[i];
|
||||
const auto size = VertexBufferDescriptorSize(b);
|
||||
if (size == 0) {
|
||||
auto owner = buffer.GetContext().GetBufferCache().ObtainNullBuffer();
|
||||
bindings.push_back({owner, owner->Handle(), 0});
|
||||
} else {
|
||||
bindings.push_back(
|
||||
buffer.GetContext().GetBufferCache().ObtainBuffer(buffer, b.addr, size));
|
||||
}
|
||||
}
|
||||
return bindings;
|
||||
return AcquireVertexBuffers(buffer, vs_input_info);
|
||||
}
|
||||
|
||||
static void RebindVertexBuffers(RenderCommandBuffer& buffer,
|
||||
const ShaderVertexInputInfo& vs_input_info,
|
||||
std::vector<BufferBinding>& bindings) {
|
||||
EXIT_IF(bindings.size() != static_cast<size_t>(vs_input_info.buffers_num));
|
||||
for (int i = 0; i < vs_input_info.buffers_num; i++) {
|
||||
const auto& vertex = vs_input_info.buffers[i];
|
||||
const auto size = VertexBufferDescriptorSize(vertex);
|
||||
if (size == 0) {
|
||||
auto owner = buffer.GetContext().GetBufferCache().ObtainNullBuffer();
|
||||
bindings[i] = {owner, owner->Handle(), 0};
|
||||
} else {
|
||||
bindings[i] =
|
||||
buffer.GetContext().GetBufferCache().ObtainBuffer(buffer, vertex.addr, size);
|
||||
}
|
||||
}
|
||||
bindings = AcquireVertexBuffers(buffer, vs_input_info);
|
||||
}
|
||||
|
||||
static PreparedIndexBuffer PrepareIndexBuffer(RenderCommandBuffer& buffer,
|
||||
@@ -1011,17 +1061,17 @@ static void EmitDrawPrimitives(const HW::UserConfig& ucfg, vk::CommandBuffer vk_
|
||||
}
|
||||
|
||||
void RenderExecutor::ExecutePreparedDraw(uint64_t submit_id, RenderCommandBuffer& buffer,
|
||||
const DrawCallInfo& draw, DrawRenderState& state,
|
||||
vk::PrimitiveTopology topology, const DrawEmitInfo& emit,
|
||||
const DrawIndexBufferSource& index_source,
|
||||
bool log_pipeline_phase, bool set_bind_debug,
|
||||
bool set_auto_debug) {
|
||||
const DrawCallInfo& draw, DrawRenderState& state,
|
||||
vk::PrimitiveTopology topology, const DrawEmitInfo& emit,
|
||||
const DrawIndexBufferSource& index_source,
|
||||
bool log_pipeline_phase, bool set_bind_debug,
|
||||
bool set_auto_debug) {
|
||||
EXIT_IF(draw.name == nullptr);
|
||||
auto& ucfg = buffer.GetUserConfig();
|
||||
|
||||
LogDrawPhase(draw.name, "PrepareBindings");
|
||||
auto bindings = PrepareGraphicsBindings(buffer, state.vs_input_info.stage,
|
||||
state.ps_input_info.stage, state.ps_active);
|
||||
auto bindings = PrepareGraphicsBindings(buffer, state.vs_input_info.stage,
|
||||
state.ps_input_info.stage, state.ps_active);
|
||||
auto vertex_bindings = PrepareVertexBuffers(submit_id, buffer, draw, state.vs_input_info);
|
||||
auto index_binding = PrepareIndexBuffer(buffer, index_source);
|
||||
RebindVertexBuffers(buffer, state.vs_input_info, vertex_bindings);
|
||||
@@ -1094,10 +1144,10 @@ void RenderExecutor::ExecutePreparedDraw(uint64_t submit_id, RenderCommandBuffer
|
||||
}
|
||||
|
||||
void RenderExecutor::DrawIndex(uint64_t submit_id, RenderCommandBuffer& buffer,
|
||||
uint32_t index_type_and_size, uint32_t index_count,
|
||||
const void* index_addr, uint32_t flags, uint32_t type,
|
||||
uint32_t instance_count, uint32_t render_target_slice_offset,
|
||||
int32_t vertex_offset_add, uint32_t first_instance) {
|
||||
uint32_t index_type_and_size, uint32_t index_count,
|
||||
const void* index_addr, uint32_t flags, uint32_t type,
|
||||
uint32_t instance_count, uint32_t render_target_slice_offset,
|
||||
int32_t vertex_offset_add, uint32_t first_instance) {
|
||||
KYTY_PROFILER_FUNCTION();
|
||||
|
||||
EXIT_IF(buffer.IsInvalid());
|
||||
@@ -1228,11 +1278,10 @@ void RenderExecutor::DrawIndex(uint64_t submit_id, RenderCommandBuffer& buffer,
|
||||
}
|
||||
|
||||
// NOLINTNEXTLINE(readability-function-cognitive-complexity)
|
||||
void RenderExecutor::DrawAuto(uint64_t submit_id, RenderCommandBuffer& buffer,
|
||||
uint32_t index_count,
|
||||
uint32_t flags, uint32_t render_target_slice_offset,
|
||||
uint32_t instance_count, uint32_t first_vertex,
|
||||
uint32_t first_instance) {
|
||||
void RenderExecutor::DrawAuto(uint64_t submit_id, RenderCommandBuffer& buffer, uint32_t index_count,
|
||||
uint32_t flags, uint32_t render_target_slice_offset,
|
||||
uint32_t instance_count, uint32_t first_vertex,
|
||||
uint32_t first_instance) {
|
||||
KYTY_PROFILER_FUNCTION();
|
||||
|
||||
EXIT_IF(buffer.IsInvalid());
|
||||
@@ -1290,7 +1339,8 @@ void RenderExecutor::DrawAuto(uint64_t submit_id, RenderCommandBuffer& buffer,
|
||||
instance_count, first_instance};
|
||||
|
||||
DrawRenderState state {};
|
||||
if (!PrepareDrawRenderState(submit_id, buffer, draw, render_target_slice_offset, false, state)) {
|
||||
if (!PrepareDrawRenderState(submit_id, buffer, draw, render_target_slice_offset, false,
|
||||
state)) {
|
||||
ResetBindings();
|
||||
return;
|
||||
}
|
||||
@@ -1340,7 +1390,7 @@ void RenderExecutor::DrawAuto(uint64_t submit_id, RenderCommandBuffer& buffer,
|
||||
}
|
||||
|
||||
bool RenderExecutor::ResolveColorTargets(uint64_t submit_id, RenderCommandBuffer& buffer,
|
||||
uint32_t render_target_slice_offset) {
|
||||
uint32_t render_target_slice_offset) {
|
||||
const auto& hw = buffer.GetRegisters();
|
||||
if (hw.GetColorControl().mode != 3) {
|
||||
return false;
|
||||
@@ -1369,8 +1419,7 @@ bool RenderExecutor::ResolveColorTargets(uint64_t submit_id, RenderCommandBuffer
|
||||
cache.MarkGpuWritten(dst.image_id);
|
||||
auto& source = cache.GetImage(src.image_id);
|
||||
auto& destination = cache.GetImage(dst.image_id);
|
||||
destination.Resolve(source,
|
||||
{src.base_mip_level, 1, src.base_array_layer, 1},
|
||||
destination.Resolve(source, {src.base_mip_level, 1, src.base_array_layer, 1},
|
||||
{dst.base_mip_level, 1, dst.base_array_layer, 1});
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -784,10 +784,10 @@ private:
|
||||
if (incoming.empty()) {
|
||||
return ScalarProvenance::Undefined;
|
||||
}
|
||||
if (incoming.size() == 1) {
|
||||
return incoming[0];
|
||||
}
|
||||
if (*phi == ScalarProvenance::Undefined) {
|
||||
if (incoming.size() == 1) {
|
||||
return incoming[0];
|
||||
}
|
||||
*phi = AddValue({ScalarValueOp::Phi, block.start_pc});
|
||||
}
|
||||
m_graph.values[*phi].phi_args = std::move(incoming);
|
||||
|
||||
+14
-14
@@ -238,8 +238,9 @@ static std::filesystem::path ResolvePathIgnoringCase(const std::filesystem::path
|
||||
}
|
||||
|
||||
// Preserve unmatched components for the caller's ENOENT path.
|
||||
std::filesystem::path resolved = path.has_root_path() ? path.root_path() : std::filesystem::path(".");
|
||||
bool matched = true;
|
||||
std::filesystem::path resolved =
|
||||
path.has_root_path() ? path.root_path() : std::filesystem::path(".");
|
||||
bool matched = true;
|
||||
|
||||
for (const auto& component: path.relative_path()) {
|
||||
if (component.empty()) {
|
||||
@@ -568,11 +569,11 @@ int64_t KYTY_SYSV_ABI KernelRead(int d, void* buf, size_t nbytes) {
|
||||
|
||||
file->mutex.Lock();
|
||||
|
||||
bool is_invalid = file->f.IsInvalid();
|
||||
const auto pos = file->f.Tell();
|
||||
const auto file_size = file->f.Size();
|
||||
const auto remaining = pos < file_size ? file_size - pos : 0;
|
||||
Memory::PrepareHostWrite(reinterpret_cast<uint64_t>(buf),
|
||||
bool is_invalid = file->f.IsInvalid();
|
||||
const auto pos = file->f.Tell();
|
||||
const auto file_size = file->f.Size();
|
||||
const auto remaining = pos < file_size ? file_size - pos : 0;
|
||||
Memory::InvalidateMemory(reinterpret_cast<uint64_t>(buf),
|
||||
std::min<uint64_t>(nbytes, remaining));
|
||||
uint32_t bytes_read = 0;
|
||||
file->f.Read(buf, static_cast<uint32_t>(nbytes), &bytes_read);
|
||||
@@ -692,13 +693,12 @@ int64_t KYTY_SYSV_ABI KernelPread(int d, void* buf, size_t nbytes, int64_t offse
|
||||
|
||||
file->mutex.Lock();
|
||||
|
||||
bool is_invalid = file->f.IsInvalid();
|
||||
auto pos = file->f.Tell();
|
||||
const auto file_size = file->f.Size();
|
||||
const auto remaining = static_cast<uint64_t>(offset) < file_size
|
||||
? file_size - static_cast<uint64_t>(offset)
|
||||
: 0;
|
||||
Memory::PrepareHostWrite(reinterpret_cast<uint64_t>(buf),
|
||||
bool is_invalid = file->f.IsInvalid();
|
||||
auto pos = file->f.Tell();
|
||||
const auto file_size = file->f.Size();
|
||||
const auto remaining =
|
||||
static_cast<uint64_t>(offset) < file_size ? file_size - static_cast<uint64_t>(offset) : 0;
|
||||
Memory::InvalidateMemory(reinterpret_cast<uint64_t>(buf),
|
||||
std::min<uint64_t>(nbytes, remaining));
|
||||
uint32_t bytes_read = 0;
|
||||
file->f.Seek(offset);
|
||||
|
||||
+76
-23
@@ -66,8 +66,8 @@ constexpr int PAGE_TABLE_POOL_ENTRIES =
|
||||
static_cast<int>(PAGE_TABLE_POOL_SIZE / PAGE_TABLE_GRANULARITY);
|
||||
constexpr uint64_t DEFAULT_FLEXIBLE_MEMORY_SIZE = 4ull * 1024ull * 1024ull * 1024ull;
|
||||
|
||||
static uint64_t g_flexible_memory_size = DEFAULT_FLEXIBLE_MEMORY_SIZE;
|
||||
static Graphics::GpuResourceManager* g_gpu_resources = nullptr;
|
||||
static uint64_t g_flexible_memory_size = DEFAULT_FLEXIBLE_MEMORY_SIZE;
|
||||
static Graphics::GpuResourceManager* g_gpu_resources = nullptr;
|
||||
|
||||
static Graphics::GpuResourceManager& GetGpuResources() {
|
||||
EXIT_IF(g_gpu_resources == nullptr);
|
||||
@@ -528,6 +528,42 @@ public:
|
||||
return false;
|
||||
}
|
||||
|
||||
uint64_t ClampRangeSize(uint64_t virtual_addr, uint64_t size) {
|
||||
Common::LockGuard lock(m_mutex);
|
||||
|
||||
if (virtual_addr == 0 || size == 0 || size > UINT64_MAX - virtual_addr) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
auto vma = std::upper_bound(
|
||||
m_ranges.begin(), m_ranges.end(), virtual_addr,
|
||||
[](uint64_t value, const Range& range) { return value < range.start; });
|
||||
if (vma == m_ranges.begin()) {
|
||||
return 0;
|
||||
}
|
||||
--vma;
|
||||
|
||||
const auto vma_end = End(vma->start, vma->size);
|
||||
if (virtual_addr < vma->start || virtual_addr >= vma_end ||
|
||||
!IsCommittedRangeType(vma->type)) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
uint64_t clamped_size = std::min(size, vma_end - virtual_addr);
|
||||
uint64_t expected = virtual_addr + clamped_size;
|
||||
++vma;
|
||||
|
||||
while (vma != m_ranges.end() && vma->start == expected && IsCommittedRangeType(vma->type) &&
|
||||
clamped_size < size) {
|
||||
const auto chunk = std::min(size - clamped_size, vma->size);
|
||||
clamped_size += chunk;
|
||||
expected += chunk;
|
||||
++vma;
|
||||
}
|
||||
|
||||
return clamped_size;
|
||||
}
|
||||
|
||||
uint64_t CountPageTableEntries(bool gpu) {
|
||||
Common::LockGuard lock(m_mutex);
|
||||
|
||||
@@ -884,6 +920,23 @@ bool TryReadBacking(uint64_t vaddr, void* data, uint64_t size) {
|
||||
g_direct_memory_backing->TryReadBacking(vaddr, data, size);
|
||||
}
|
||||
|
||||
uint64_t ClampRangeSize(uint64_t vaddr, uint64_t size) {
|
||||
EXIT_IF(g_virtual_ranges == nullptr);
|
||||
|
||||
const auto clamped_size = g_virtual_ranges->ClampRangeSize(vaddr, size);
|
||||
if (clamped_size == 0) {
|
||||
EXIT("Memory: attempted to access invalid address 0x%016" PRIx64 " with size 0x%016" PRIx64
|
||||
"\n",
|
||||
vaddr, size);
|
||||
}
|
||||
if (clamped_size != size) {
|
||||
LOGF("Memory: clamped buffer range addr=0x%016" PRIx64 " size=0x%016" PRIx64
|
||||
" to 0x%016" PRIx64 "\n",
|
||||
vaddr, size, clamped_size);
|
||||
}
|
||||
return clamped_size;
|
||||
}
|
||||
|
||||
void WriteBacking(uint64_t vaddr, const void* data, uint64_t size) noexcept {
|
||||
if (!TryWriteBacking(vaddr, data, size)) {
|
||||
EXIT("Memory: required direct-backing write failed, addr=0x%016" PRIx64
|
||||
@@ -892,11 +945,11 @@ void WriteBacking(uint64_t vaddr, const void* data, uint64_t size) noexcept {
|
||||
}
|
||||
}
|
||||
|
||||
void PrepareHostWrite(uint64_t vaddr, uint64_t size) {
|
||||
void InvalidateMemory(uint64_t vaddr, uint64_t size) {
|
||||
if (size == 0) {
|
||||
return;
|
||||
}
|
||||
GetGpuResources().PrepareHostWrite(vaddr, size);
|
||||
(void)GetGpuResources().InvalidateMemory(vaddr, size);
|
||||
}
|
||||
|
||||
void InstallGpuResources(Graphics::GpuResourceManager* resources) noexcept {
|
||||
@@ -1915,24 +1968,24 @@ int32_t KYTY_SYSV_ABI KernelMapNamedFlexibleMemory(void** addr_in_out, size_t le
|
||||
|
||||
EXIT_NOT_IMPLEMENTED(addr_in_out == nullptr);
|
||||
|
||||
constexpr size_t PAGE_SIZE = 0x4000;
|
||||
constexpr size_t MAXIMUM_NAME_SIZE = 32;
|
||||
constexpr uint64_t DEFAULT_PS5_BASE = 0x200000000;
|
||||
constexpr int GUEST_MAP_FIXED = 0x10;
|
||||
constexpr int GUEST_MAP_SHARED = 0x01;
|
||||
constexpr int GUEST_MAP_PRIVATE = 0x02;
|
||||
constexpr int GUEST_MAP_NO_OVERWRITE = 0x80;
|
||||
constexpr int GUEST_MAP_VOID = 0x100;
|
||||
constexpr int GUEST_MAP_STACK = 0x400;
|
||||
constexpr int GUEST_MAP_NO_SYNC = 0x800;
|
||||
constexpr int GUEST_MAP_ANON = 0x1000;
|
||||
constexpr int GUEST_MAP_UNKNOWN_8000 = 0x8000;
|
||||
constexpr int GUEST_MAP_NO_CORE = 0x20000;
|
||||
constexpr int GUEST_MAP_NO_COALESCE = 0x400000;
|
||||
constexpr int SUPPORTED_MAP_BITS =
|
||||
GUEST_MAP_SHARED | GUEST_MAP_PRIVATE | GUEST_MAP_FIXED | GUEST_MAP_NO_OVERWRITE |
|
||||
GUEST_MAP_VOID | GUEST_MAP_STACK | GUEST_MAP_NO_SYNC | GUEST_MAP_ANON |
|
||||
GUEST_MAP_UNKNOWN_8000 | GUEST_MAP_NO_CORE | GUEST_MAP_NO_COALESCE;
|
||||
constexpr size_t PAGE_SIZE = 0x4000;
|
||||
constexpr size_t MAXIMUM_NAME_SIZE = 32;
|
||||
constexpr uint64_t DEFAULT_PS5_BASE = 0x200000000;
|
||||
constexpr int GUEST_MAP_FIXED = 0x10;
|
||||
constexpr int GUEST_MAP_SHARED = 0x01;
|
||||
constexpr int GUEST_MAP_PRIVATE = 0x02;
|
||||
constexpr int GUEST_MAP_NO_OVERWRITE = 0x80;
|
||||
constexpr int GUEST_MAP_VOID = 0x100;
|
||||
constexpr int GUEST_MAP_STACK = 0x400;
|
||||
constexpr int GUEST_MAP_NO_SYNC = 0x800;
|
||||
constexpr int GUEST_MAP_ANON = 0x1000;
|
||||
constexpr int GUEST_MAP_UNKNOWN_8000 = 0x8000;
|
||||
constexpr int GUEST_MAP_NO_CORE = 0x20000;
|
||||
constexpr int GUEST_MAP_NO_COALESCE = 0x400000;
|
||||
constexpr int SUPPORTED_MAP_BITS = GUEST_MAP_SHARED | GUEST_MAP_PRIVATE | GUEST_MAP_FIXED |
|
||||
GUEST_MAP_NO_OVERWRITE | GUEST_MAP_VOID | GUEST_MAP_STACK |
|
||||
GUEST_MAP_NO_SYNC | GUEST_MAP_ANON | GUEST_MAP_UNKNOWN_8000 |
|
||||
GUEST_MAP_NO_CORE | GUEST_MAP_NO_COALESCE;
|
||||
|
||||
if (len == 0 || (len & (PAGE_SIZE - 1)) != 0) {
|
||||
return KERNEL_ERROR_EINVAL;
|
||||
@@ -3294,7 +3347,7 @@ int KYTY_SYSV_ABI KernelReserveVirtualRange(void** addr, size_t len, int flags,
|
||||
"\t alignment = 0x%016" PRIx64 "\n",
|
||||
in_addr, len, flags, alignment);
|
||||
|
||||
constexpr size_t PAGE_SIZE = 0x4000;
|
||||
constexpr size_t PAGE_SIZE = 0x4000;
|
||||
constexpr int GUEST_MAP_FIXED = 0x10;
|
||||
constexpr int GUEST_MAP_NO_OVERWRITE = 0x80;
|
||||
|
||||
|
||||
+8
-7
@@ -99,13 +99,14 @@ struct KernelMemoryPoolBlockStats {
|
||||
static_assert(sizeof(KernelMemoryPoolBlockStats) == 16,
|
||||
"KernelMemoryPoolBlockStats struct size is incorrect");
|
||||
|
||||
void RegisterCallbacks(callback_func_t alloc_func, callback_func_t free_func);
|
||||
void SetFlexibleMemorySize(uint64_t size);
|
||||
bool TryWriteBacking(uint64_t vaddr, const void* data, uint64_t size);
|
||||
bool TryReadBacking(uint64_t vaddr, void* data, uint64_t size);
|
||||
void WriteBacking(uint64_t vaddr, const void* data, uint64_t size) noexcept;
|
||||
void PrepareHostWrite(uint64_t vaddr, uint64_t size);
|
||||
void InstallGpuResources(Graphics::GpuResourceManager* resources) noexcept;
|
||||
void RegisterCallbacks(callback_func_t alloc_func, callback_func_t free_func);
|
||||
void SetFlexibleMemorySize(uint64_t size);
|
||||
bool TryWriteBacking(uint64_t vaddr, const void* data, uint64_t size);
|
||||
bool TryReadBacking(uint64_t vaddr, void* data, uint64_t size);
|
||||
[[nodiscard]] uint64_t ClampRangeSize(uint64_t vaddr, uint64_t size);
|
||||
void WriteBacking(uint64_t vaddr, const void* data, uint64_t size) noexcept;
|
||||
void InvalidateMemory(uint64_t vaddr, uint64_t size);
|
||||
void InstallGpuResources(Graphics::GpuResourceManager* resources) noexcept;
|
||||
[[nodiscard]] bool HandleGpuFault(Graphics::PageFaultAccess access, uint64_t fault_vaddr) noexcept;
|
||||
|
||||
int KYTY_SYSV_ABI KernelMapNamedFlexibleMemory(void** addr_in_out, size_t len, int prot, int flags,
|
||||
|
||||
+2
-1
@@ -34,7 +34,7 @@ namespace LibNet {
|
||||
|
||||
LIB_VERSION("Net", 1, "Net", 1, 1);
|
||||
|
||||
static thread_local int g_net_errno = 0;
|
||||
static thread_local int g_net_errno = 0;
|
||||
static constexpr uint32_t g_in6addr_any[4] {};
|
||||
|
||||
namespace Net = Network::Net;
|
||||
@@ -1398,6 +1398,7 @@ LIB_DEFINE(InitNet_1_NpManager) {
|
||||
LIB_FUNC("O80NrhUOPGY", NpManager::NpCheckPremium);
|
||||
LIB_FUNC("eQH7nWPcAgc", NpManager::NpGetState);
|
||||
LIB_FUNC("e-ZuhGEoeC4", NpManager::NpGetNpReachabilityState);
|
||||
LIB_FUNC("Oad3rvY-NJQ", NpManager::NpHasSignedUp);
|
||||
}
|
||||
|
||||
} // namespace LibNpManager
|
||||
|
||||
+19
-6
@@ -19,7 +19,7 @@
|
||||
|
||||
// POSIX uses plain int file descriptors for sockets; provide the Winsock spellings
|
||||
// the shared (non-guarded) code paths reference.
|
||||
using SOCKET = int;
|
||||
using SOCKET = int;
|
||||
static constexpr SOCKET INVALID_SOCKET = -1;
|
||||
#endif
|
||||
|
||||
@@ -820,10 +820,10 @@ struct NetEtherAddr {
|
||||
};
|
||||
|
||||
#if defined(_WIN32)
|
||||
using NativeSocket = SOCKET;
|
||||
using NativeSocket = SOCKET;
|
||||
static constexpr NativeSocket INVALID_NATIVE_SOCKET = INVALID_SOCKET;
|
||||
#else
|
||||
using NativeSocket = int;
|
||||
using NativeSocket = int;
|
||||
static constexpr NativeSocket INVALID_NATIVE_SOCKET = -1;
|
||||
#endif
|
||||
|
||||
@@ -1738,7 +1738,8 @@ int KYTY_SYSV_ABI Accept(int s, void* addr, uint32_t* addrlen) {
|
||||
#if defined(_WIN32)
|
||||
sockaddr_storage host_addr {};
|
||||
int host_addrlen = sizeof(host_addr);
|
||||
NativeSocket accepted = ::accept(socket, reinterpret_cast<sockaddr*>(&host_addr), &host_addrlen);
|
||||
NativeSocket accepted =
|
||||
::accept(socket, reinterpret_cast<sockaddr*>(&host_addr), &host_addrlen);
|
||||
if (accepted == INVALID_NATIVE_SOCKET) {
|
||||
return SetPosixSocketError();
|
||||
}
|
||||
@@ -3753,8 +3754,6 @@ int KYTY_SYSV_ABI NpGetState(int user_id, uint32_t* state) {
|
||||
int KYTY_SYSV_ABI NpGetNpReachabilityState(int user_id, uint32_t* state) {
|
||||
PRINT_NAME();
|
||||
|
||||
constexpr int np_error_invalid_argument = -2141913085; /* 0x80550003 */
|
||||
|
||||
if (state == nullptr) {
|
||||
return np_error_invalid_argument;
|
||||
}
|
||||
@@ -3767,6 +3766,20 @@ int KYTY_SYSV_ABI NpGetNpReachabilityState(int user_id, uint32_t* state) {
|
||||
return OK;
|
||||
}
|
||||
|
||||
int KYTY_SYSV_ABI NpHasSignedUp(int user_id, bool* has_signed_up) {
|
||||
PRINT_NAME();
|
||||
|
||||
if (has_signed_up == nullptr) {
|
||||
return np_error_invalid_argument;
|
||||
}
|
||||
|
||||
LOGF("\t user_id = %d\n", user_id);
|
||||
|
||||
*has_signed_up = false;
|
||||
|
||||
return OK;
|
||||
}
|
||||
|
||||
} // namespace NpManager
|
||||
|
||||
} // namespace Libs::Network
|
||||
|
||||
@@ -184,6 +184,7 @@ int KYTY_SYSV_ABI NpCheckPremium(int req_id, const NpCheckPremiumParameter* par
|
||||
NpCheckPremiumResult* result);
|
||||
int KYTY_SYSV_ABI NpGetState(int user_id, uint32_t* state);
|
||||
int KYTY_SYSV_ABI NpGetNpReachabilityState(int user_id, uint32_t* state);
|
||||
int KYTY_SYSV_ABI NpHasSignedUp(int user_id, bool* has_signed_up);
|
||||
|
||||
} // namespace NpManager
|
||||
|
||||
|
||||
@@ -669,6 +669,8 @@ void TestRangeSet() {
|
||||
ranges.Add(0x1000, 0x80);
|
||||
ranges.Add(0x1080, 0x80);
|
||||
ranges.Add(0x1200, 0x40);
|
||||
Check(ranges.Contains(0x1010, 0xe0) && !ranges.Contains(0x1010, 0x200),
|
||||
"range set containment did not require full coverage");
|
||||
auto intersections = ranges.Intersections(0x1070, 0x1b0);
|
||||
Check(intersections.size() == 2 && intersections[0].address == 0x1070 &&
|
||||
intersections[0].size == 0x90 && intersections[1].address == 0x1200 &&
|
||||
@@ -682,6 +684,46 @@ void TestRangeSet() {
|
||||
"range set subtraction did not preserve both exact tails");
|
||||
}
|
||||
|
||||
void TestRangeInvalidation() {
|
||||
constexpr uintptr_t base = 0x0000000201000000ull;
|
||||
TrackerHarness harness;
|
||||
auto &tracker = harness.tracker;
|
||||
auto &page_manager = harness.page_manager;
|
||||
constexpr uint64_t size = Libs::Graphics::TRACKER_REGION_SIZE * 2;
|
||||
auto *memory = static_cast<uint8_t *>(
|
||||
VirtualAlloc(reinterpret_cast<void *>(base), size, MEM_RESERVE | MEM_COMMIT,
|
||||
PAGE_READWRITE));
|
||||
Check(memory == reinterpret_cast<void *>(base),
|
||||
"range invalidation allocation failed");
|
||||
const auto address = reinterpret_cast<uint64_t>(memory);
|
||||
page_manager.OnGpuMap(address, size);
|
||||
|
||||
tracker.ForEachUploadRange(
|
||||
address, size, true, [](uint64_t, uint64_t) noexcept {},
|
||||
[]() noexcept {});
|
||||
Check(tracker.IsRegionGpuModified(address, size) && !IsWritable(memory),
|
||||
"range invalidation setup did not establish GPU ownership");
|
||||
|
||||
uint32_t flushes = 0;
|
||||
tracker.InvalidateRegion(address + 16, size - 32, [&] {
|
||||
flushes++;
|
||||
tracker.ForEachDownloadRange<true>(
|
||||
address + 16, size - 32, [](uint64_t, uint64_t) noexcept {});
|
||||
});
|
||||
Check(flushes == 1 && !tracker.IsRegionGpuModified(address, size) &&
|
||||
tracker.IsRegionCpuModified(address, size) && IsWritable(memory) &&
|
||||
IsWritable(memory + size - 1),
|
||||
"range invalidation did not batch ownership transfer across regions");
|
||||
|
||||
tracker.InvalidateRegion(address + 16, size - 32, [&] { flushes++; });
|
||||
Check(flushes == 1,
|
||||
"clean range invalidation unnecessarily requested a GPU flush");
|
||||
tracker.UntrackMemory(address, size);
|
||||
page_manager.OnGpuUnmap(address, size);
|
||||
Check(VirtualFree(memory, 0, MEM_RELEASE) != 0,
|
||||
"range invalidation VirtualFree failed");
|
||||
}
|
||||
|
||||
void TestCpuDirtyUploadAndFault() {
|
||||
constexpr uintptr_t base = 0x0000000200010000ull;
|
||||
TrackerHarness harness;
|
||||
@@ -1207,6 +1249,7 @@ int main(int argc, char **argv) {
|
||||
TestSameSlabTrackerArbitration();
|
||||
TestSharedMetadataAndImagePageFault();
|
||||
TestRangeSet();
|
||||
TestRangeInvalidation();
|
||||
TestGpuDirtyBits();
|
||||
TestCrossRegionUpload();
|
||||
TestFaultDuringUploadRemainsDirty();
|
||||
|
||||
@@ -301,36 +301,6 @@ void TestSharedWatcherFault() {
|
||||
Check(VirtualFree(memory, 0, MEM_RELEASE) != 0, "VirtualFree failed");
|
||||
}
|
||||
|
||||
void TestMappedHostWriteRange() {
|
||||
FaultContext context;
|
||||
PageManager manager(InvalidateFault, &context);
|
||||
context.manager = &manager;
|
||||
const auto page_size = manager.GetPageSize();
|
||||
auto *memory = Allocate(page_size * 3);
|
||||
const auto address = reinterpret_cast<uint64_t>(memory);
|
||||
|
||||
manager.OnGpuMap(address, page_size);
|
||||
manager.OnGpuMap(address + page_size * 2, page_size);
|
||||
manager.UpdatePageWatchers(true, address, page_size);
|
||||
manager.UpdatePageWatchers(true, address + page_size * 2, page_size);
|
||||
Check(manager.HasAnyMapping(address + 16, page_size * 3 - 32),
|
||||
"host-write range did not find partial GPU mappings");
|
||||
Check(!manager.IsMapped(address, page_size * 3),
|
||||
"partial GPU mappings were reported as a full mapping");
|
||||
Check(manager.HandleWriteRange(address + 16, page_size * 3 - 32),
|
||||
"mapped host-write range was not handled");
|
||||
Check(context.calls.load(std::memory_order_relaxed) == 2,
|
||||
"host-write range did not invalidate each mapped watched page");
|
||||
Check(IsWritable(memory) && IsWritable(memory + page_size * 2),
|
||||
"host-write range did not restore writable protection");
|
||||
|
||||
manager.OnGpuUnmap(address, page_size);
|
||||
manager.OnGpuUnmap(address + page_size * 2, page_size);
|
||||
Check(!manager.HasAnyMapping(address, page_size * 3),
|
||||
"host-write range retained stale GPU mappings");
|
||||
Check(VirtualFree(memory, 0, MEM_RELEASE) != 0, "VirtualFree failed");
|
||||
}
|
||||
|
||||
void TestReadWriteWatcherFault() {
|
||||
FaultContext context;
|
||||
PageManager manager(InvalidateFault, &context);
|
||||
@@ -855,7 +825,6 @@ int main(int argc, char **argv) {
|
||||
}
|
||||
TestWatchFaultAndUnwatch();
|
||||
TestSharedWatcherFault();
|
||||
TestMappedHostWriteRange();
|
||||
TestReadWriteWatcherFault();
|
||||
TestPermittedMappedLateFaultsResume();
|
||||
TestPartialMappingUnmapPreservesTokens();
|
||||
|
||||
@@ -186,6 +186,39 @@ void TestCfgPhi() {
|
||||
"acyclic control-flow descriptor phi was not classified dynamic");
|
||||
}
|
||||
|
||||
void TestNestedLoopPhiConvergence() {
|
||||
Program program;
|
||||
program.blocks.resize(4);
|
||||
program.blocks[0].predecessors = {1};
|
||||
program.blocks[0].successors = {1};
|
||||
program.blocks[1].predecessors = {0, 3};
|
||||
program.blocks[1].successors = {0, 2};
|
||||
program.blocks[2].predecessors = {1};
|
||||
program.blocks[2].successors = {3};
|
||||
program.blocks[3].predecessors = {2};
|
||||
program.blocks[3].successors = {1};
|
||||
Instruction increment;
|
||||
increment.op = Opcode::IAddU32;
|
||||
increment.dst = Sgpr(0);
|
||||
increment.src[0] = Sgpr(0);
|
||||
increment.src[1] = Imm(1);
|
||||
increment.src_count = 2;
|
||||
program.blocks[0].instructions = {increment};
|
||||
program.blocks[2].instructions = {BufferUse(4, 0)};
|
||||
|
||||
std::string error;
|
||||
Check(BuildScalarProvenance(program, &error), error.c_str());
|
||||
const auto* source =
|
||||
GetDescriptorSource(program, program.blocks[2].instructions[0].memory.resource_source);
|
||||
Check(source != nullptr, "nested-loop descriptor source was not attached");
|
||||
const auto value_id = source->dwords[0];
|
||||
const auto& phi = Value(program, value_id);
|
||||
Check(phi.op == ScalarValueOp::Phi && phi.phi_args.size() == 2 &&
|
||||
((phi.phi_args[0] == value_id && phi.phi_args[1] != value_id) ||
|
||||
(phi.phi_args[1] == value_id && phi.phi_args[0] != value_id)),
|
||||
"nested loop did not retain its recursive scalar provenance phi");
|
||||
}
|
||||
|
||||
void TestDiamondReadPathsAreDynamic() {
|
||||
std::array<uint32_t, 1> left = {0x11111111u};
|
||||
std::array<uint32_t, 1> right = {0x22222222u};
|
||||
@@ -1045,6 +1078,7 @@ int main() {
|
||||
try {
|
||||
TestPerUseDescriptorDefinitions();
|
||||
TestCfgPhi();
|
||||
TestNestedLoopPhiConvergence();
|
||||
TestDiamondReadPathsAreDynamic();
|
||||
TestEquivalentConstantPhiIsStatic();
|
||||
TestWideMoveInvalidatesAndCopiesBothDwords();
|
||||
|
||||
@@ -1679,7 +1679,9 @@ public:
|
||||
Require("GpuCommandLane", "processor fault context",
|
||||
Gpu::CurrentCommandProcessor() == &processor,
|
||||
"processor resource test lost its command context");
|
||||
resources.PrepareHostWrite(fault_base, sizeof(uint32_t));
|
||||
Require("GpuCommandLane", "processor memory invalidation",
|
||||
resources.InvalidateMemory(fault_base, sizeof(uint32_t)),
|
||||
"processor memory invalidation did not find its mapped range");
|
||||
});
|
||||
resources.UnmapMemory(fault_base, fault_size, GpuAccess::ReadWrite);
|
||||
Require("GpuCommandLane", "processor fault unmap",
|
||||
@@ -15218,7 +15220,7 @@ void CheckRenderTargetFormatContract() {
|
||||
resource.kind = ShaderRecompiler::IR::ResourceKind::Image;
|
||||
} else if (std::strcmp(kind, "storage-no-write") == 0) {
|
||||
resource.written = false;
|
||||
} else if (std::strcmp(kind, "storage-atomic") == 0) {
|
||||
} else if (std::strcmp(kind, "storage-nonuint-atomic") == 0) {
|
||||
resource.atomic = true;
|
||||
} else if (std::strcmp(kind, "storage-compare") == 0) {
|
||||
resource.depth_compare = true;
|
||||
@@ -15435,6 +15437,12 @@ void CheckSampledColorViews() {
|
||||
Require("SampledColorViews", "write-only uint 2D-array storage resource",
|
||||
IsSupportedStorageImageResource(storage_resource),
|
||||
"basic write-only uint 2D-array storage resource was rejected");
|
||||
storage_resource.dimension = ShaderRecompiler::Decoder::ImageDimension::Dim2D;
|
||||
storage_resource.read = true;
|
||||
storage_resource.atomic = true;
|
||||
Require("SampledColorViews", "atomic uint 2D storage resource",
|
||||
IsSupportedStorageImageResource(storage_resource),
|
||||
"atomic uint storage resource was rejected");
|
||||
|
||||
char path[MAX_PATH]{};
|
||||
Require("SampledColorViews", "host",
|
||||
@@ -15444,7 +15452,7 @@ void CheckSampledColorViews() {
|
||||
{"sampled-invalid-selector", "sampled-incompatible-format",
|
||||
"sampled-invalid-high", "sampled-depth-format", "sampled-depth-swizzle",
|
||||
"storage-incompatible-format", "storage-kind", "storage-no-write",
|
||||
"storage-atomic", "storage-compare", "storage-mip", "storage-dimension",
|
||||
"storage-nonuint-atomic", "storage-compare", "storage-mip", "storage-dimension",
|
||||
"volume-mip-count", "volume-slice-range"}) {
|
||||
std::string command =
|
||||
std::string("\"") + path + "\" --image-view-death " + kind;
|
||||
@@ -16153,6 +16161,19 @@ ShaderTextureResource BasicUintVolumeStorageTextureDescriptor() {
|
||||
0x00700000u, 0x00000000u, 0x00000000u}};
|
||||
}
|
||||
|
||||
ShaderRecompiler::IR::ImageResource AtomicStorageTextureResource() {
|
||||
auto resource = BasicLinearStorageTextureResource();
|
||||
resource.kind = ShaderRecompiler::IR::ResourceKind::StorageImageUint;
|
||||
resource.read = true;
|
||||
resource.atomic = true;
|
||||
return resource;
|
||||
}
|
||||
|
||||
ShaderTextureResource AtomicStorageTextureDescriptor() {
|
||||
return {{0x304bb700u, 0xc1400000u, 0x0000001fu, 0x91b00204u, 0x00000000u,
|
||||
0x00700000u, 0x00000000u, 0x00000000u}};
|
||||
}
|
||||
|
||||
[[noreturn]] void RunStorageTextureDescriptorDeathCase(const char *kind) {
|
||||
auto resource = BasicStorageTextureResource();
|
||||
auto descriptor = BasicStorageTextureDescriptor();
|
||||
@@ -16214,6 +16235,12 @@ ShaderTextureResource BasicUintVolumeStorageTextureDescriptor() {
|
||||
} else if (std::strcmp(kind, "uint-resource-float-format") == 0) {
|
||||
resource = BasicUintArrayStorageTextureResource();
|
||||
descriptor = BasicArrayStorageTextureDescriptor();
|
||||
} else if (std::strcmp(kind, "atomic-format") == 0) {
|
||||
resource = AtomicStorageTextureResource();
|
||||
descriptor = AtomicStorageTextureDescriptor();
|
||||
descriptor.fields[1] =
|
||||
(descriptor.fields[1] & ~0x1ff00000u) |
|
||||
(Prospero::GpuEnumValue(Prospero::BufferFormat::k8UInt) << 20u);
|
||||
} else if (std::strcmp(kind, "depth-tile-read") == 0) {
|
||||
resource = Ppsa14053DepthTileStorageTextureResource();
|
||||
descriptor = Ppsa14053DepthTileStorageTextureDescriptor();
|
||||
@@ -16297,7 +16324,7 @@ void CheckBasicStorageTextureDescriptor() {
|
||||
"PPSA06228 R11G11B10 storage descriptor fixture is malformed");
|
||||
ValidateStorageTexture(BasicBgraStorageTextureResource(), r11g11b10,
|
||||
0x870000);
|
||||
ValidateStorageColorView(vk::Format::eB10G11R11UfloatPack32,
|
||||
ValidateStorageColorView(vk::Format::eB8G8R8A8Unorm,
|
||||
vk::Format::eB10G11R11UfloatPack32,
|
||||
r11g11b10.DstSelXYZW());
|
||||
|
||||
@@ -16588,6 +16615,18 @@ void CheckBasicStorageTextureDescriptor() {
|
||||
IsValidImageSwizzle(DstSel(4, 4, 4, 4)),
|
||||
"single-channel replicated destination selection was rejected");
|
||||
|
||||
const auto atomic = AtomicStorageTextureDescriptor();
|
||||
Require("BasicStorageTexture", "atomic R32_UINT descriptor",
|
||||
atomic.Width5() + 1u == 128 && atomic.Height5() + 1u == 1 &&
|
||||
atomic.Depth() + 1u == 1 &&
|
||||
atomic.Type() ==
|
||||
Prospero::GpuEnumValue(Prospero::ImageType::kColor2D) &&
|
||||
atomic.Format() ==
|
||||
Prospero::GpuEnumValue(Prospero::BufferFormat::k32UInt) &&
|
||||
atomic.DstSelXYZW() == DstSel(4, 0, 0, 1),
|
||||
"PPSA22102 image-atomic descriptor fixture is malformed");
|
||||
ValidateStorageTexture(AtomicStorageTextureResource(), atomic, 0x10000);
|
||||
|
||||
char path[MAX_PATH]{};
|
||||
Require("BasicStorageTexture", "host",
|
||||
GetModuleFileNameA(nullptr, path, MAX_PATH) != 0,
|
||||
@@ -16596,7 +16635,7 @@ void CheckBasicStorageTextureDescriptor() {
|
||||
{"resource", "type", "tile", "mip", "swizzle", "linear-rgb1-read",
|
||||
"bgra-read", "r16-float-read", "r8-unorm-read", "yzwx-read",
|
||||
"reserved-swizzle", "array-base-out-of-range", "array-mip-view",
|
||||
"reserved", "uint-format", "uint-resource-float-format",
|
||||
"reserved", "uint-format", "uint-resource-float-format", "atomic-format",
|
||||
"depth-tile-read", "depth-tile-extent", "depth-tile-fmask"}) {
|
||||
std::string command = std::string("\"") + path +
|
||||
"\" --storage-texture-descriptor-death " + kind;
|
||||
|
||||
@@ -405,6 +405,10 @@ void TestDirectMapQueryOffsetAndPartialMunmap() {
|
||||
"TryReadBacking should reject a range crossing an unmapped span");
|
||||
Check(test, rejected_read == transaction_sentinel,
|
||||
"failed backing reads must not modify a destination prefix");
|
||||
Check(test,
|
||||
Libs::LibKernel::Memory::ClampRangeSize(base + SceKernelPageSize - 0xf30, 0x1560) ==
|
||||
0xf30,
|
||||
"ClampRangeSize did not stop at an unmapped span");
|
||||
|
||||
info = Query(test, base + SceKernelPageSize, SceKernelVqFindNext);
|
||||
ExpectRange(test, info, base + SceKernelPageSize * 2, base + SceKernelPageSize * 4,
|
||||
@@ -473,6 +477,11 @@ void TestMunmapAcrossAdjacentFlexibleMappings() {
|
||||
&right, SceKernelPageSize, SceKernelProtCpuRw, SceKernelMapFixed, "adjacent_right"),
|
||||
"KernelMapNamedFlexibleMemory(right)");
|
||||
|
||||
Check(test,
|
||||
Libs::LibKernel::Memory::ClampRangeSize(base + SceKernelPageSize - 0x100, 0x200) ==
|
||||
0x200,
|
||||
"ClampRangeSize did not cross adjacent committed mappings");
|
||||
|
||||
CheckOk(test, Libs::LibKernel::Memory::KernelMunmap(base, SceKernelPageSize * 2),
|
||||
"KernelMunmap(adjacent mappings)");
|
||||
Check(test, AvailableFlexibleMemory(test) == baseline,
|
||||
|
||||
Reference in New Issue
Block a user