From e1e8ae68bf0bd1f503136a360b12ef8614a17b3c Mon Sep 17 00:00:00 2001 From: CamilleLaVey Date: Fri, 3 Jul 2026 03:25:45 -0400 Subject: [PATCH] [vulkan] First intent on unswizzling via gpu shaders --- .../renderer_vulkan/vk_compute_pass.cpp | 323 ++++++++++++++++++ .../renderer_vulkan/vk_compute_pass.h | 51 +++ .../renderer_vulkan/vk_texture_cache.cpp | 73 +++- .../renderer_vulkan/vk_texture_cache.h | 3 + 4 files changed, 444 insertions(+), 6 deletions(-) diff --git a/src/video_core/renderer_vulkan/vk_compute_pass.cpp b/src/video_core/renderer_vulkan/vk_compute_pass.cpp index 0011ec3bf5..0cd8e5c9e7 100644 --- a/src/video_core/renderer_vulkan/vk_compute_pass.cpp +++ b/src/video_core/renderer_vulkan/vk_compute_pass.cpp @@ -25,6 +25,9 @@ #include "video_core/host_shaders/vulkan_quad_indexed_comp_spv.h" #include "video_core/host_shaders/vulkan_uint8_comp_spv.h" #include "video_core/host_shaders/block_linear_unswizzle_3d_bcn_comp_spv.h" +#include "video_core/host_shaders/block_linear_unswizzle_2d_comp_spv.h" +#include "video_core/host_shaders/block_linear_unswizzle_3d_comp_spv.h" +#include "video_core/host_shaders/pitch_unswizzle_comp_spv.h" #include "video_core/renderer_vulkan/vk_compute_pass.h" #include "video_core/renderer_vulkan/vk_descriptor_pool.h" #include "video_core/renderer_vulkan/vk_scheduler.h" @@ -232,6 +235,26 @@ struct QueriesPrefixScanPushConstants { struct ConditionalRenderingResolvePushConstants { u32 compare_to_zero; }; + +struct BlockLinear3DImagePushConstants { + alignas(16) std::array origin; + alignas(16) std::array destination; + u32 bytes_per_block_log2; + u32 slice_size; + u32 block_size; + u32 x_shift; + u32 block_height; + u32 block_height_mask; + u32 block_depth; + u32 block_depth_mask; +}; + +struct PitchUnswizzlePushConstants { + std::array origin; + std::array destination; + u32 bytes_per_block; + u32 pitch; +}; } // Anonymous namespace ComputePass::ComputePass(const Device& device_, Scheduler& scheduler, DescriptorPool& descriptor_pool, @@ -653,6 +676,306 @@ void ASTCDecoderPass::Assemble(Image& image, const StagingBufferRef& map, scheduler.Finish(); } +BlockLinearUnswizzleImage2DPass::BlockLinearUnswizzleImage2DPass( + const Device& device_, Scheduler& scheduler_, DescriptorPool& descriptor_pool_, + StagingBufferPool& staging_buffer_pool_, + ComputePassDescriptorQueue& compute_pass_descriptor_queue_) + : ComputePass(device_, scheduler_, descriptor_pool_, ASTC_DESCRIPTOR_SET_BINDINGS, + ASTC_PASS_DESCRIPTOR_UPDATE_TEMPLATE_ENTRY, ASTC_BANK_INFO, + COMPUTE_PUSH_CONSTANT_RANGE, + BLOCK_LINEAR_UNSWIZZLE_2D_COMP_SPV), + scheduler{scheduler_}, staging_buffer_pool{staging_buffer_pool_}, + compute_pass_descriptor_queue{compute_pass_descriptor_queue_} {} + +BlockLinearUnswizzleImage2DPass::~BlockLinearUnswizzleImage2DPass() = default; + +void BlockLinearUnswizzleImage2DPass::Unswizzle( + Image& image, const StagingBufferRef& map, + std::span swizzles) { + using namespace VideoCommon::Accelerated; + scheduler.RequestOutsideRenderPassOperationContext(); + const VkPipeline vk_pipeline = *pipeline; + const VkImageAspectFlags aspect_mask = image.AspectMask(); + const VkImage vk_image = image.Handle(); + const bool is_initialized = image.ExchangeInitialization(); + scheduler.Record([vk_pipeline, vk_image, aspect_mask, + is_initialized](vk::CommandBuffer cmdbuf) { + const VkImageMemoryBarrier image_barrier{ + .sType = VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER, + .pNext = nullptr, + .srcAccessMask = static_cast(is_initialized ? VK_ACCESS_SHADER_WRITE_BIT + : VK_ACCESS_NONE), + .dstAccessMask = VK_ACCESS_SHADER_READ_BIT | VK_ACCESS_SHADER_WRITE_BIT, + .oldLayout = is_initialized ? VK_IMAGE_LAYOUT_GENERAL : VK_IMAGE_LAYOUT_UNDEFINED, + .newLayout = VK_IMAGE_LAYOUT_GENERAL, + .srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED, + .dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED, + .image = vk_image, + .subresourceRange{ + .aspectMask = aspect_mask, + .baseMipLevel = 0, + .levelCount = VK_REMAINING_MIP_LEVELS, + .baseArrayLayer = 0, + .layerCount = VK_REMAINING_ARRAY_LAYERS, + }, + }; + cmdbuf.PipelineBarrier(is_initialized ? vk::PIPELINE_STAGE_GRAPHICS_COMPUTE_TRANSFER + : VkPipelineStageFlags(VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT), + VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT, 0, image_barrier); + cmdbuf.BindPipeline(VK_PIPELINE_BIND_POINT_COMPUTE, vk_pipeline); + }); + for (const VideoCommon::SwizzleParameters& swizzle : swizzles) { + const size_t input_offset = swizzle.buffer_offset + map.offset; + const u32 num_dispatches_x = Common::DivCeil(swizzle.num_tiles.width, 32U); + const u32 num_dispatches_y = Common::DivCeil(swizzle.num_tiles.height, 32U); + const u32 num_dispatches_z = image.info.resources.layers; + + compute_pass_descriptor_queue.Acquire(scheduler, 2); + compute_pass_descriptor_queue.AddBuffer(map.buffer, input_offset, + image.guest_size_bytes - swizzle.buffer_offset); + compute_pass_descriptor_queue.AddImage(image.StorageImageView(swizzle.level)); + const void* const descriptor_data{compute_pass_descriptor_queue.UpdateData()}; + + const auto params = MakeBlockLinearSwizzle2DParams(swizzle, image.info); + scheduler.Record([this, num_dispatches_x, num_dispatches_y, num_dispatches_z, params, + descriptor_data](vk::CommandBuffer cmdbuf) { + const VkDescriptorSet set = descriptor_allocator.Commit(); + device.GetLogical().UpdateDescriptorSet(set, *descriptor_template, descriptor_data); + cmdbuf.BindDescriptorSets(VK_PIPELINE_BIND_POINT_COMPUTE, *layout, 0, set, {}); + cmdbuf.PushConstants(*layout, VK_SHADER_STAGE_COMPUTE_BIT, params); + cmdbuf.Dispatch(num_dispatches_x, num_dispatches_y, num_dispatches_z); + }); + } + scheduler.Record([vk_image, aspect_mask](vk::CommandBuffer cmdbuf) { + const VkImageMemoryBarrier image_barrier{ + .sType = VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER, + .pNext = nullptr, + .srcAccessMask = VK_ACCESS_SHADER_WRITE_BIT, + .dstAccessMask = VK_ACCESS_SHADER_READ_BIT | VK_ACCESS_SHADER_WRITE_BIT, + .oldLayout = VK_IMAGE_LAYOUT_GENERAL, + .newLayout = VK_IMAGE_LAYOUT_GENERAL, + .srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED, + .dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED, + .image = vk_image, + .subresourceRange{ + .aspectMask = aspect_mask, + .baseMipLevel = 0, + .levelCount = VK_REMAINING_MIP_LEVELS, + .baseArrayLayer = 0, + .layerCount = VK_REMAINING_ARRAY_LAYERS, + }, + }; + cmdbuf.PipelineBarrier(VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT, + vk::PIPELINE_STAGE_GRAPHICS_COMPUTE, 0, image_barrier); + }); + scheduler.Finish(); +} + +BlockLinearUnswizzleImage3DPass::BlockLinearUnswizzleImage3DPass( + const Device& device_, Scheduler& scheduler_, DescriptorPool& descriptor_pool_, + StagingBufferPool& staging_buffer_pool_, + ComputePassDescriptorQueue& compute_pass_descriptor_queue_) + : ComputePass(device_, scheduler_, descriptor_pool_, ASTC_DESCRIPTOR_SET_BINDINGS, + ASTC_PASS_DESCRIPTOR_UPDATE_TEMPLATE_ENTRY, ASTC_BANK_INFO, + COMPUTE_PUSH_CONSTANT_RANGE, + BLOCK_LINEAR_UNSWIZZLE_3D_COMP_SPV), + scheduler{scheduler_}, staging_buffer_pool{staging_buffer_pool_}, + compute_pass_descriptor_queue{compute_pass_descriptor_queue_} {} + +BlockLinearUnswizzleImage3DPass::~BlockLinearUnswizzleImage3DPass() = default; + +void BlockLinearUnswizzleImage3DPass::Unswizzle( + Image& image, const StagingBufferRef& map, + std::span swizzles) { + using namespace VideoCommon::Accelerated; + scheduler.RequestOutsideRenderPassOperationContext(); + const VkPipeline vk_pipeline = *pipeline; + const VkImageAspectFlags aspect_mask = image.AspectMask(); + const VkImage vk_image = image.Handle(); + const bool is_initialized = image.ExchangeInitialization(); + scheduler.Record([vk_pipeline, vk_image, aspect_mask, + is_initialized](vk::CommandBuffer cmdbuf) { + const VkImageMemoryBarrier image_barrier{ + .sType = VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER, + .pNext = nullptr, + .srcAccessMask = static_cast(is_initialized ? VK_ACCESS_SHADER_WRITE_BIT + : VK_ACCESS_NONE), + .dstAccessMask = VK_ACCESS_SHADER_READ_BIT | VK_ACCESS_SHADER_WRITE_BIT, + .oldLayout = is_initialized ? VK_IMAGE_LAYOUT_GENERAL : VK_IMAGE_LAYOUT_UNDEFINED, + .newLayout = VK_IMAGE_LAYOUT_GENERAL, + .srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED, + .dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED, + .image = vk_image, + .subresourceRange{ + .aspectMask = aspect_mask, + .baseMipLevel = 0, + .levelCount = VK_REMAINING_MIP_LEVELS, + .baseArrayLayer = 0, + .layerCount = VK_REMAINING_ARRAY_LAYERS, + }, + }; + cmdbuf.PipelineBarrier(is_initialized ? vk::PIPELINE_STAGE_GRAPHICS_COMPUTE_TRANSFER + : VkPipelineStageFlags(VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT), + VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT, 0, image_barrier); + cmdbuf.BindPipeline(VK_PIPELINE_BIND_POINT_COMPUTE, vk_pipeline); + }); + for (const VideoCommon::SwizzleParameters& swizzle : swizzles) { + const size_t input_offset = swizzle.buffer_offset + map.offset; + const u32 num_dispatches_x = Common::DivCeil(swizzle.num_tiles.width, 16U); + const u32 num_dispatches_y = Common::DivCeil(swizzle.num_tiles.height, 8U); + const u32 num_dispatches_z = Common::DivCeil(swizzle.num_tiles.depth, 8U); + + compute_pass_descriptor_queue.Acquire(scheduler, 2); + compute_pass_descriptor_queue.AddBuffer(map.buffer, input_offset, + image.guest_size_bytes - swizzle.buffer_offset); + compute_pass_descriptor_queue.AddImage(image.StorageImageView(swizzle.level)); + const void* const descriptor_data{compute_pass_descriptor_queue.UpdateData()}; + + const auto p = MakeBlockLinearSwizzle3DParams(swizzle, image.info); + const BlockLinear3DImagePushConstants params{ + .origin = p.origin, + .destination = p.destination, + .bytes_per_block_log2 = p.bytes_per_block_log2, + .slice_size = p.slice_size, + .block_size = p.block_size, + .x_shift = p.x_shift, + .block_height = p.block_height, + .block_height_mask = p.block_height_mask, + .block_depth = p.block_depth, + .block_depth_mask = p.block_depth_mask, + }; + scheduler.Record([this, num_dispatches_x, num_dispatches_y, num_dispatches_z, params, + descriptor_data](vk::CommandBuffer cmdbuf) { + const VkDescriptorSet set = descriptor_allocator.Commit(); + device.GetLogical().UpdateDescriptorSet(set, *descriptor_template, descriptor_data); + cmdbuf.BindDescriptorSets(VK_PIPELINE_BIND_POINT_COMPUTE, *layout, 0, set, {}); + cmdbuf.PushConstants(*layout, VK_SHADER_STAGE_COMPUTE_BIT, params); + cmdbuf.Dispatch(num_dispatches_x, num_dispatches_y, num_dispatches_z); + }); + } + scheduler.Record([vk_image, aspect_mask](vk::CommandBuffer cmdbuf) { + const VkImageMemoryBarrier image_barrier{ + .sType = VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER, + .pNext = nullptr, + .srcAccessMask = VK_ACCESS_SHADER_WRITE_BIT, + .dstAccessMask = VK_ACCESS_SHADER_READ_BIT | VK_ACCESS_SHADER_WRITE_BIT, + .oldLayout = VK_IMAGE_LAYOUT_GENERAL, + .newLayout = VK_IMAGE_LAYOUT_GENERAL, + .srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED, + .dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED, + .image = vk_image, + .subresourceRange{ + .aspectMask = aspect_mask, + .baseMipLevel = 0, + .levelCount = VK_REMAINING_MIP_LEVELS, + .baseArrayLayer = 0, + .layerCount = VK_REMAINING_ARRAY_LAYERS, + }, + }; + cmdbuf.PipelineBarrier(VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT, + vk::PIPELINE_STAGE_GRAPHICS_COMPUTE, 0, image_barrier); + }); + scheduler.Finish(); +} + +PitchUnswizzlePass::PitchUnswizzlePass(const Device& device_, Scheduler& scheduler_, + DescriptorPool& descriptor_pool_, + StagingBufferPool& staging_buffer_pool_, + ComputePassDescriptorQueue& compute_pass_descriptor_queue_) + : ComputePass(device_, scheduler_, descriptor_pool_, ASTC_DESCRIPTOR_SET_BINDINGS, + ASTC_PASS_DESCRIPTOR_UPDATE_TEMPLATE_ENTRY, ASTC_BANK_INFO, + COMPUTE_PUSH_CONSTANT_RANGE, + PITCH_UNSWIZZLE_COMP_SPV), + scheduler{scheduler_}, staging_buffer_pool{staging_buffer_pool_}, + compute_pass_descriptor_queue{compute_pass_descriptor_queue_} {} + +PitchUnswizzlePass::~PitchUnswizzlePass() = default; + +void PitchUnswizzlePass::Unswizzle(Image& image, const StagingBufferRef& map, + std::span swizzles) { + scheduler.RequestOutsideRenderPassOperationContext(); + const VkPipeline vk_pipeline = *pipeline; + const VkImageAspectFlags aspect_mask = image.AspectMask(); + const VkImage vk_image = image.Handle(); + const bool is_initialized = image.ExchangeInitialization(); + scheduler.Record([vk_pipeline, vk_image, aspect_mask, + is_initialized](vk::CommandBuffer cmdbuf) { + const VkImageMemoryBarrier image_barrier{ + .sType = VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER, + .pNext = nullptr, + .srcAccessMask = static_cast(is_initialized ? VK_ACCESS_SHADER_WRITE_BIT + : VK_ACCESS_NONE), + .dstAccessMask = VK_ACCESS_SHADER_READ_BIT | VK_ACCESS_SHADER_WRITE_BIT, + .oldLayout = is_initialized ? VK_IMAGE_LAYOUT_GENERAL : VK_IMAGE_LAYOUT_UNDEFINED, + .newLayout = VK_IMAGE_LAYOUT_GENERAL, + .srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED, + .dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED, + .image = vk_image, + .subresourceRange{ + .aspectMask = aspect_mask, + .baseMipLevel = 0, + .levelCount = VK_REMAINING_MIP_LEVELS, + .baseArrayLayer = 0, + .layerCount = VK_REMAINING_ARRAY_LAYERS, + }, + }; + cmdbuf.PipelineBarrier(is_initialized ? vk::PIPELINE_STAGE_GRAPHICS_COMPUTE_TRANSFER + : VkPipelineStageFlags(VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT), + VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT, 0, image_barrier); + cmdbuf.BindPipeline(VK_PIPELINE_BIND_POINT_COMPUTE, vk_pipeline); + }); + const u32 bytes_per_block = VideoCore::Surface::BytesPerBlock(image.info.format); + for (const VideoCommon::SwizzleParameters& swizzle : swizzles) { + const size_t input_offset = swizzle.buffer_offset + map.offset; + const u32 num_dispatches_x = Common::DivCeil(swizzle.num_tiles.width, 32U); + const u32 num_dispatches_y = Common::DivCeil(swizzle.num_tiles.height, 32U); + + compute_pass_descriptor_queue.Acquire(scheduler, 2); + compute_pass_descriptor_queue.AddBuffer(map.buffer, input_offset, + image.guest_size_bytes - swizzle.buffer_offset); + compute_pass_descriptor_queue.AddImage(image.StorageImageView(swizzle.level)); + const void* const descriptor_data{compute_pass_descriptor_queue.UpdateData()}; + + const PitchUnswizzlePushConstants params{ + .origin = {0, 0}, + .destination = {0, 0}, + .bytes_per_block = bytes_per_block, + .pitch = image.info.pitch, + }; + scheduler.Record([this, num_dispatches_x, num_dispatches_y, params, + descriptor_data](vk::CommandBuffer cmdbuf) { + const VkDescriptorSet set = descriptor_allocator.Commit(); + device.GetLogical().UpdateDescriptorSet(set, *descriptor_template, descriptor_data); + cmdbuf.BindDescriptorSets(VK_PIPELINE_BIND_POINT_COMPUTE, *layout, 0, set, {}); + cmdbuf.PushConstants(*layout, VK_SHADER_STAGE_COMPUTE_BIT, params); + cmdbuf.Dispatch(num_dispatches_x, num_dispatches_y, 1); + }); + } + scheduler.Record([vk_image, aspect_mask](vk::CommandBuffer cmdbuf) { + const VkImageMemoryBarrier image_barrier{ + .sType = VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER, + .pNext = nullptr, + .srcAccessMask = VK_ACCESS_SHADER_WRITE_BIT, + .dstAccessMask = VK_ACCESS_SHADER_READ_BIT | VK_ACCESS_SHADER_WRITE_BIT, + .oldLayout = VK_IMAGE_LAYOUT_GENERAL, + .newLayout = VK_IMAGE_LAYOUT_GENERAL, + .srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED, + .dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED, + .image = vk_image, + .subresourceRange{ + .aspectMask = aspect_mask, + .baseMipLevel = 0, + .levelCount = VK_REMAINING_MIP_LEVELS, + .baseArrayLayer = 0, + .layerCount = VK_REMAINING_ARRAY_LAYERS, + }, + }; + cmdbuf.PipelineBarrier(VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT, + vk::PIPELINE_STAGE_GRAPHICS_COMPUTE, 0, image_barrier); + }); + scheduler.Finish(); +} + constexpr u32 BL3D_BINDING_INPUT_BUFFER = 0; constexpr u32 BL3D_BINDING_OUTPUT_BUFFER = 1; diff --git a/src/video_core/renderer_vulkan/vk_compute_pass.h b/src/video_core/renderer_vulkan/vk_compute_pass.h index cb213dae7e..dcef5ad014 100644 --- a/src/video_core/renderer_vulkan/vk_compute_pass.h +++ b/src/video_core/renderer_vulkan/vk_compute_pass.h @@ -25,6 +25,7 @@ struct SwizzleParameters; namespace Vulkan { +using VideoCommon::Accelerated::BlockLinearSwizzle2DParams; using VideoCommon::Accelerated::BlockLinearSwizzle3DParams; class Device; @@ -164,6 +165,56 @@ private: ComputePassDescriptorQueue& compute_pass_descriptor_queue; }; +class BlockLinearUnswizzleImage2DPass final : public ComputePass { +public: + explicit BlockLinearUnswizzleImage2DPass(const Device& device_, Scheduler& scheduler_, + DescriptorPool& descriptor_pool_, + StagingBufferPool& staging_buffer_pool_, + ComputePassDescriptorQueue& compute_pass_descriptor_queue_); + ~BlockLinearUnswizzleImage2DPass(); + + void Unswizzle(Image& image, const StagingBufferRef& map, + std::span swizzles); + +private: + Scheduler& scheduler; + StagingBufferPool& staging_buffer_pool; + ComputePassDescriptorQueue& compute_pass_descriptor_queue; +}; + +class BlockLinearUnswizzleImage3DPass final : public ComputePass { +public: + explicit BlockLinearUnswizzleImage3DPass(const Device& device_, Scheduler& scheduler_, + DescriptorPool& descriptor_pool_, + StagingBufferPool& staging_buffer_pool_, + ComputePassDescriptorQueue& compute_pass_descriptor_queue_); + ~BlockLinearUnswizzleImage3DPass(); + + void Unswizzle(Image& image, const StagingBufferRef& map, + std::span swizzles); + +private: + Scheduler& scheduler; + StagingBufferPool& staging_buffer_pool; + ComputePassDescriptorQueue& compute_pass_descriptor_queue; +}; + +class PitchUnswizzlePass final : public ComputePass { +public: + explicit PitchUnswizzlePass(const Device& device_, Scheduler& scheduler_, + DescriptorPool& descriptor_pool_, + StagingBufferPool& staging_buffer_pool_, + ComputePassDescriptorQueue& compute_pass_descriptor_queue_); + ~PitchUnswizzlePass(); + + void Unswizzle(Image& image, const StagingBufferRef& map, + std::span swizzles); + +private: + Scheduler& scheduler; + StagingBufferPool& staging_buffer_pool; + ComputePassDescriptorQueue& compute_pass_descriptor_queue; +}; class MSAACopyPass final : public ComputePass { public: diff --git a/src/video_core/renderer_vulkan/vk_texture_cache.cpp b/src/video_core/renderer_vulkan/vk_texture_cache.cpp index 7804e33d46..9d7ca4d251 100644 --- a/src/video_core/renderer_vulkan/vk_texture_cache.cpp +++ b/src/video_core/renderer_vulkan/vk_texture_cache.cpp @@ -195,6 +195,29 @@ constexpr VkBorderColor ConvertBorderColor(const std::array& color) { return allocator.CreateImage(image_ci); } +[[nodiscard]] VkFormat UnswizzleStorageFormat(u32 bytes_per_block) { + switch (bytes_per_block) { + case 1: + return VK_FORMAT_R8_UINT; + case 2: + return VK_FORMAT_R16_UINT; + case 4: + return VK_FORMAT_R32_UINT; + case 8: + return VK_FORMAT_R32G32_UINT; + case 16: + return VK_FORMAT_R32G32B32A32_UINT; + default: + ASSERT_MSG(false, "Invalid bytes_per_block={} for accelerated unswizzle", bytes_per_block); + return VK_FORMAT_R32_UINT; + } +} + +[[nodiscard]] bool IsUnswizzleStorageFormatSupported(u32 bytes_per_block) { + return bytes_per_block == 1 || bytes_per_block == 2 || bytes_per_block == 4 || + bytes_per_block == 8 || bytes_per_block == 16; +} + [[nodiscard]] vk::ImageView MakeStorageView(const vk::Device& device, u32 level, VkImage image, VkFormat format) { static constexpr VkImageViewUsageCreateInfo storage_image_view_usage_create_info{ @@ -887,6 +910,12 @@ TextureCacheRuntime::TextureCacheRuntime(const Device& device_, Scheduler& sched if (device.IsStorageImageMultisampleSupported()) { msaa_copy_pass.emplace(device, scheduler, descriptor_pool, staging_buffer_pool, compute_pass_descriptor_queue); } + bl_unswizzle_2d_pass.emplace(device, scheduler, descriptor_pool, staging_buffer_pool, + compute_pass_descriptor_queue); + bl_unswizzle_3d_pass.emplace(device, scheduler, descriptor_pool, staging_buffer_pool, + compute_pass_descriptor_queue); + pitch_unswizzle_pass.emplace(device, scheduler, descriptor_pool, staging_buffer_pool, + compute_pass_descriptor_queue); if (!device.IsKhrImageFormatListSupported()) { return; } @@ -894,6 +923,14 @@ TextureCacheRuntime::TextureCacheRuntime(const Device& device_, Scheduler& sched const auto image_format = static_cast(index_a); if (IsPixelFormatASTC(image_format) && !device.IsOptimalAstcSupported()) { view_formats[index_a].push_back(VK_FORMAT_A8B8G8R8_UNORM_PACK32); + } else if (!IsPixelFormatASTC(image_format) && !IsPixelFormatBCn(image_format)) { + // Generic block-linear/pitch accelerated unswizzle needs an alternate UINT + // storage view registered up-front so MakeImage() enables mutable format + + // storage usage for these images (see Image::Image / storage_image_views below). + const u32 bpp = VideoCore::Surface::BytesPerBlock(image_format); + if (IsUnswizzleStorageFormatSupported(bpp)) { + view_formats[index_a].push_back(UnswizzleStorageFormat(bpp)); + } } for (size_t index_b = 0; index_b < VideoCore::Surface::MaxPixelFormat; index_b++) { const auto view_format = static_cast(index_b); @@ -1580,6 +1617,15 @@ Image::Image(TextureCacheRuntime& runtime_, const ImageInfo& info_, GPUVAddr gpu flags |= VideoCommon::ImageFlagBits::Converted; flags |= VideoCommon::ImageFlagBits::CostlyLoad; } + if (!IsPixelFormatASTC(info.format) && !IsPixelFormatBCn(info.format)) { + if (info.type == ImageType::e2D || info.type == ImageType::e3D) { + flags |= VideoCommon::ImageFlagBits::AcceleratedUpload; + } else if (info.type == ImageType::Linear) { + if (IsUnswizzleStorageFormatSupported(VideoCore::Surface::BytesPerBlock(info.format))) { + flags |= VideoCommon::ImageFlagBits::AcceleratedUpload; + } + } + } if (runtime->device.HasDebuggingToolAttached()) { original_image.SetObjectNameEXT(VideoCommon::Name(*this).c_str()); } @@ -1593,6 +1639,13 @@ Image::Image(TextureCacheRuntime& runtime_, const ImageInfo& info_, GPUVAddr gpu storage_image_views[level] = MakeStorageView(device, level, *original_image, VK_FORMAT_A8B8G8R8_UNORM_PACK32); } + } else if (True(flags & VideoCommon::ImageFlagBits::AcceleratedUpload)) { + const auto& device = runtime->device.GetLogical(); + const VkFormat storage_format = + UnswizzleStorageFormat(VideoCore::Surface::BytesPerBlock(info.format)); + for (s32 level = 0; level < info.resources.levels; ++level) { + storage_image_views[level] = MakeStorageView(device, level, *original_image, storage_format); + } } } @@ -2448,16 +2501,24 @@ void TextureCacheRuntime::AccelerateImageUpload( return astc_decoder_pass->Assemble(image, map, swizzles); } - if (!Settings::values.gpu_unswizzle_enabled.GetValue() || !bl3d_unswizzle_pass) { - if (IsPixelFormatBCn(image.info.format) && image.info.type == ImageType::e3D) { - ASSERT(false && "GPU unswizzle is disabled for BCn 3D texture"); + if (IsPixelFormatBCn(image.info.format)) { + if (Settings::values.gpu_unswizzle_enabled.GetValue() && bl3d_unswizzle_pass && + image.info.type == ImageType::e3D && image.info.resources.levels == 1 && + image.info.resources.layers == 1) { + return bl3d_unswizzle_pass->Unswizzle(image, map, swizzles, z_start, z_count); } - ASSERT(false); + ASSERT(false && "GPU unswizzle is disabled for BCn 3D texture"); return; } - if (bl3d_unswizzle_pass && IsPixelFormatBCn(image.info.format) && image.info.type == ImageType::e3D && image.info.resources.levels == 1 && image.info.resources.layers == 1) { - return bl3d_unswizzle_pass->Unswizzle(image, map, swizzles, z_start, z_count); + if (image.info.type == ImageType::e2D) { + return bl_unswizzle_2d_pass->Unswizzle(image, map, swizzles); + } + if (image.info.type == ImageType::e3D) { + return bl_unswizzle_3d_pass->Unswizzle(image, map, swizzles); + } + if (image.info.type == ImageType::Linear) { + return pitch_unswizzle_pass->Unswizzle(image, map, swizzles); } ASSERT(false); diff --git a/src/video_core/renderer_vulkan/vk_texture_cache.h b/src/video_core/renderer_vulkan/vk_texture_cache.h index 881135da1b..e817afc448 100644 --- a/src/video_core/renderer_vulkan/vk_texture_cache.h +++ b/src/video_core/renderer_vulkan/vk_texture_cache.h @@ -130,6 +130,9 @@ public: std::optional astc_decoder_pass; std::optional bl3d_unswizzle_pass; + std::optional bl_unswizzle_2d_pass; + std::optional bl_unswizzle_3d_pass; + std::optional pitch_unswizzle_pass; std::optional msaa_copy_pass; const Settings::ResolutionScalingInfo& resolution; std::array, VideoCore::Surface::MaxPixelFormat> view_formats;