diff --git a/Common/GPU/D3D11/thin3d_d3d11.cpp b/Common/GPU/D3D11/thin3d_d3d11.cpp index c0b527fa2b..9b59852c88 100644 --- a/Common/GPU/D3D11/thin3d_d3d11.cpp +++ b/Common/GPU/D3D11/thin3d_d3d11.cpp @@ -271,6 +271,7 @@ D3D11DrawContext::D3D11DrawContext(ID3D11Device *device, ID3D11DeviceContext *de caps_.fragmentShaderDepthWriteSupported = true; caps_.fragmentShaderStencilWriteSupported = false; caps_.blendMinMaxSupported = true; + caps_.multiSampleLevelsMask = 1; // More could be supported with some work. D3D11_FEATURE_DATA_D3D11_OPTIONS options{}; HRESULT result = device_->CheckFeatureSupport(D3D11_FEATURE_D3D11_OPTIONS, &options, sizeof(options)); diff --git a/Common/GPU/D3D9/thin3d_d3d9.cpp b/Common/GPU/D3D9/thin3d_d3d9.cpp index dce9f66335..e50eb3b1c8 100644 --- a/Common/GPU/D3D9/thin3d_d3d9.cpp +++ b/Common/GPU/D3D9/thin3d_d3d9.cpp @@ -764,6 +764,7 @@ D3D9Context::D3D9Context(IDirect3D9 *d3d, IDirect3D9Ex *d3dEx, int adapterId, ID caps_.fragmentShaderStencilWriteSupported = false; caps_.blendMinMaxSupported = true; caps_.isTilingGPU = false; + caps_.multiSampleLevelsMask = 1; // More could be supported with some work. if ((caps.RasterCaps & D3DPRASTERCAPS_ANISOTROPY) != 0 && caps.MaxAnisotropy > 1) { caps_.anisoSupported = true; diff --git a/Common/GPU/OpenGL/thin3d_gl.cpp b/Common/GPU/OpenGL/thin3d_gl.cpp index 61df0bdaee..b107f2f194 100644 --- a/Common/GPU/OpenGL/thin3d_gl.cpp +++ b/Common/GPU/OpenGL/thin3d_gl.cpp @@ -557,6 +557,7 @@ OpenGLContext::OpenGLContext() { caps_.framebufferStencilBlitSupported = caps_.framebufferBlitSupported; caps_.depthClampSupported = gl_extensions.ARB_depth_clamp || gl_extensions.EXT_depth_clamp; caps_.blendMinMaxSupported = gl_extensions.EXT_blend_minmax; + caps_.multiSampleLevelsMask = 1; // More could be supported with some work. if (gl_extensions.IsGLES) { caps_.clipDistanceSupported = gl_extensions.EXT_clip_cull_distance || gl_extensions.APPLE_clip_distance; diff --git a/Common/GPU/Vulkan/VulkanFramebuffer.cpp b/Common/GPU/Vulkan/VulkanFramebuffer.cpp index 0de8f76a3c..c3bdcb4ae5 100644 --- a/Common/GPU/Vulkan/VulkanFramebuffer.cpp +++ b/Common/GPU/Vulkan/VulkanFramebuffer.cpp @@ -2,21 +2,39 @@ #include "Common/GPU/Vulkan/VulkanFramebuffer.h" #include "Common/GPU/Vulkan/VulkanQueueRunner.h" -VkSampleCountFlagBits SampleCountToFlagBits(int count) { +VkSampleCountFlagBits MultiSampleLevelToFlagBits(int count) { // TODO: Check hardware support here, or elsewhere? // Some hardware only supports 4x. switch (count) { - case 1: return VK_SAMPLE_COUNT_1_BIT; - case 2: return VK_SAMPLE_COUNT_2_BIT; - case 4: return VK_SAMPLE_COUNT_4_BIT; - case 8: return VK_SAMPLE_COUNT_8_BIT; - case 16: return VK_SAMPLE_COUNT_16_BIT; // rare + case 0: return VK_SAMPLE_COUNT_1_BIT; + case 1: return VK_SAMPLE_COUNT_2_BIT; + case 2: return VK_SAMPLE_COUNT_4_BIT; // The only non-1 level supported on some mobile chips. + case 3: return VK_SAMPLE_COUNT_8_BIT; + case 4: return VK_SAMPLE_COUNT_16_BIT; // rare but exists, on Intel for example default: _assert_(false); return VK_SAMPLE_COUNT_1_BIT; } } +void VKRImage::Delete(VulkanContext *vulkan) { + // Get rid of the views first, feels cleaner (but in reality doesn't matter). + if (rtView) + vulkan->Delete().QueueDeleteImageView(rtView); + if (texAllLayersView) + vulkan->Delete().QueueDeleteImageView(texAllLayersView); + for (int i = 0; i < 2; i++) { + if (texLayerViews[i]) { + vulkan->Delete().QueueDeleteImageView(texLayerViews[i]); + } + } + + if (image) { + _dbg_assert_(alloc); + vulkan->Delete().QueueDeleteImageAllocation(image, alloc); + } +} + VKRFramebuffer::VKRFramebuffer(VulkanContext *vk, VkCommandBuffer initCmd, VKRRenderPass *compatibleRenderPass, int _width, int _height, int _numLayers, int _numSamples, bool createDepthStencilBuffer, const char *tag) : vulkan_(vk), tag_(tag), width(_width), height(_height), numLayers(_numLayers) { @@ -28,12 +46,12 @@ VKRFramebuffer::VKRFramebuffer(VulkanContext *vk, VkCommandBuffer initCmd, VKRRe } if (_numSamples > 1) { - sampleCount = SampleCountToFlagBits(_numSamples); + sampleCount = MultiSampleLevelToFlagBits(_numSamples); // TODO: Create a different tag for these? CreateImage(vulkan_, initCmd, msaaColor, width, height, numLayers, sampleCount, VK_FORMAT_R8G8B8A8_UNORM, VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL, true, tag); if (createDepthStencilBuffer) { - CreateImage(vulkan_, initCmd, depth, width, height, numLayers, sampleCount, vulkan_->GetDeviceInfo().preferredDepthStencilFormat, VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL, false, tag); + CreateImage(vulkan_, initCmd, msaaDepth, width, height, numLayers, sampleCount, vulkan_->GetDeviceInfo().preferredDepthStencilFormat, VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL, false, tag); } } else { sampleCount = VK_SAMPLE_COUNT_1_BIT; @@ -71,19 +89,27 @@ VkFramebuffer VKRFramebuffer::Get(VKRRenderPass *compatibleRenderPass, RenderPas } VkFramebufferCreateInfo fbci{ VK_STRUCTURE_TYPE_FRAMEBUFFER_CREATE_INFO }; - VkImageView views[2]{}; + VkImageView views[4]{}; bool hasDepth = RenderPassTypeHasDepth(rpType); - views[0] = color.rtView; // 2D array texture if multilayered. + int attachmentCount = 0; + views[attachmentCount++] = color.rtView; // 2D array texture if multilayered. if (hasDepth) { if (!depth.rtView) { WARN_LOG(G3D, "depth render type to non-depth fb: %p %p fmt=%d (%s %dx%d)", depth.image, depth.texAllLayersView, depth.format, tag_.c_str(), width, height); // Will probably crash, depending on driver. } - views[1] = depth.rtView; + views[attachmentCount++] = depth.rtView; } + if (rpType & RenderPassType::MULTISAMPLE) { + views[attachmentCount++] = msaaColor.rtView; + if (hasDepth) { + views[attachmentCount++] = msaaDepth.rtView; + } + } + fbci.renderPass = compatibleRenderPass->Get(vulkan_, rpType, sampleCount); - fbci.attachmentCount = hasDepth ? 2 : 1; + fbci.attachmentCount = attachmentCount; fbci.pAttachments = views; fbci.width = width; fbci.height = height; @@ -100,32 +126,11 @@ VkFramebuffer VKRFramebuffer::Get(VKRRenderPass *compatibleRenderPass, RenderPas } VKRFramebuffer::~VKRFramebuffer() { - // Get rid of the views first, feels cleaner (but in reality doesn't matter). - if (color.rtView) - vulkan_->Delete().QueueDeleteImageView(color.rtView); - if (depth.rtView) - vulkan_->Delete().QueueDeleteImageView(depth.rtView); - if (color.texAllLayersView) - vulkan_->Delete().QueueDeleteImageView(color.texAllLayersView); - if (depth.texAllLayersView) - vulkan_->Delete().QueueDeleteImageView(depth.texAllLayersView); - for (int i = 0; i < 2; i++) { - if (color.texLayerViews[i]) { - vulkan_->Delete().QueueDeleteImageView(color.texLayerViews[i]); - } - if (depth.texLayerViews[i]) { - vulkan_->Delete().QueueDeleteImageView(depth.texLayerViews[i]); - } - } + color.Delete(vulkan_); + depth.Delete(vulkan_); + msaaColor.Delete(vulkan_); + msaaDepth.Delete(vulkan_); - if (color.image) { - _dbg_assert_(color.alloc); - vulkan_->Delete().QueueDeleteImageAllocation(color.image, color.alloc); - } - if (depth.image) { - _dbg_assert_(depth.alloc); - vulkan_->Delete().QueueDeleteImageAllocation(depth.image, depth.alloc); - } for (auto &fb : framebuf) { if (fb) { vulkan_->Delete().QueueDeleteFramebuffer(fb); @@ -229,6 +234,7 @@ void VKRFramebuffer::CreateImage(VulkanContext *vulkan, VkCommandBuffer cmd, VKR 0, dstAccessMask); img.layout = initialLayout; img.format = format; + img.sampleCount = sampleCount; img.tag = tag ? tag : "N/A"; img.numLayers = numLayers; } @@ -258,7 +264,7 @@ VkRenderPass CreateRenderPass(VulkanContext *vulkan, const RPKey &key, RenderPas bool isBackbuffer = rpType == RenderPassType::BACKBUFFER; bool hasDepth = RenderPassTypeHasDepth(rpType); bool multiview = RenderPassTypeHasMultiView(rpType); - bool multisample = rpType & RenderPassType::MULTISAMPLE; + bool multisample = RenderPassTypeHasMultisample(rpType); if (multiview) { // TODO: Assert that the device has multiview support enabled. @@ -306,7 +312,7 @@ VkRenderPass CreateRenderPass(VulkanContext *vulkan, const RPKey &key, RenderPas if (hasDepth) { depthAttachmentIndex = attachmentCount; attachments[attachmentCount].format = vulkan->GetDeviceInfo().preferredDepthStencilFormat; - attachments[attachmentCount].samples = VK_SAMPLE_COUNT_1_BIT; + attachments[attachmentCount].samples = sampleCount; attachments[attachmentCount].loadOp = ConvertLoadAction(key.depthLoadAction); attachments[attachmentCount].storeOp = ConvertStoreAction(key.depthStoreAction); attachments[attachmentCount].stencilLoadOp = ConvertLoadAction(key.stencilLoadAction); @@ -338,9 +344,10 @@ VkRenderPass CreateRenderPass(VulkanContext *vulkan, const RPKey &key, RenderPas subpass.colorAttachmentCount = 1; subpass.pColorAttachments = &color_reference; - VkAttachmentReference resolve_references[2]; + VkAttachmentReference resolve_references[2]{}; if (multisample) { resolve_references[0].attachment = 0; // the non-msaa color buffer. + resolve_references[0].layout = selfDependency ? VK_IMAGE_LAYOUT_GENERAL : VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL; subpass.pResolveAttachments = resolve_references; } else { subpass.pResolveAttachments = nullptr; @@ -415,6 +422,9 @@ VkRenderPass VKRRenderPass::Get(VulkanContext *vulkan, RenderPassType rpType, Vk // WARNING: We don't include sampleCount in the key, there's only the distinction multisampled or not // which comes from the rpType. // So you CAN NOT mix and match different non-one sample counts. + + _dbg_assert_(!((rpType & RenderPassType::MULTISAMPLE) && sampleCount == VK_SAMPLE_COUNT_1_BIT)); + if (!pass[(int)rpType]) { pass[(int)rpType] = CreateRenderPass(vulkan, key_, (RenderPassType)rpType, sampleCount); } diff --git a/Common/GPU/Vulkan/VulkanFramebuffer.h b/Common/GPU/Vulkan/VulkanFramebuffer.h index 0247410df9..8ee9e84235 100644 --- a/Common/GPU/Vulkan/VulkanFramebuffer.h +++ b/Common/GPU/Vulkan/VulkanFramebuffer.h @@ -43,6 +43,7 @@ struct VKRImage { VmaAllocation alloc; VkFormat format; + VkSampleCountFlagBits sampleCount; // This one is used by QueueRunner's Perform functions to keep track. CANNOT be used anywhere else due to sync issues. VkImageLayout layout; @@ -51,6 +52,8 @@ struct VKRImage { // For debugging. std::string tag; + + void Delete(VulkanContext *vulkan); }; class VKRFramebuffer { @@ -82,6 +85,14 @@ public: return depth.image != VK_NULL_HANDLE; } + VkImageView GetRTView() { + if (sampleCount == VK_SAMPLE_COUNT_1_BIT) { + return color.rtView; + } else { + return msaaColor.rtView; + } + } + VulkanContext *Vulkan() const { return vulkan_; } private: static void CreateImage(VulkanContext *vulkan, VkCommandBuffer cmd, VKRImage &img, int width, int height, int numLayers, VkSampleCountFlagBits sampleCount, VkFormat format, VkImageLayout initialLayout, bool color, const char *tag); @@ -104,7 +115,11 @@ inline bool RenderPassTypeHasMultiView(RenderPassType type) { return (type & RenderPassType::MULTIVIEW) != 0; } -VkSampleCountFlagBits SampleCountToFlagBits(int count); +inline bool RenderPassTypeHasMultisample(RenderPassType type) { + return (type & RenderPassType::MULTISAMPLE) != 0; +} + +VkSampleCountFlagBits MultiSampleLevelToFlagBits(int count); // Must be the same order as Draw::RPAction enum class VKRRenderPassLoadAction : uint8_t { diff --git a/Common/GPU/Vulkan/VulkanQueueRunner.cpp b/Common/GPU/Vulkan/VulkanQueueRunner.cpp index 7f7d9e8717..ee5a7ec4ed 100644 --- a/Common/GPU/Vulkan/VulkanQueueRunner.cpp +++ b/Common/GPU/Vulkan/VulkanQueueRunner.cpp @@ -762,6 +762,14 @@ static const char *rpTypeDebugNames[] = { "MV_RENDER_DEPTH", "MV_RENDER_INPUT", "MV_RENDER_DEPTH_INPUT", + "MS_RENDER", + "MS_RENDER_DEPTH", + "MS_RENDER_INPUT", + "MS_RENDER_DEPTH_INPUT", + "MS_MV_RENDER", + "MS_MV_RENDER_DEPTH", + "MS_MV_RENDER_INPUT", + "MS_MV_RENDER_DEPTH_INPUT", "BACKBUF", }; @@ -1323,7 +1331,8 @@ void VulkanQueueRunner::PerformRenderPass(const VKRStep &step, VkCommandBuffer c // Maybe a middle pass. But let's try to just block and compile here for now, this doesn't // happen all that much. graphicsPipeline->pipeline[(size_t)rpType] = Promise::CreateEmpty(); - graphicsPipeline->Create(vulkan_, renderPass->Get(vulkan_, rpType, step.render.framebuffer ? step.render.framebuffer->sampleCount : VK_SAMPLE_COUNT_1_BIT), rpType); + VkSampleCountFlagBits sampleCount = step.render.framebuffer ? step.render.framebuffer->sampleCount : VK_SAMPLE_COUNT_1_BIT; + graphicsPipeline->Create(vulkan_, renderPass->Get(vulkan_, rpType, sampleCount), rpType, sampleCount); } VkPipeline pipeline = graphicsPipeline->pipeline[(size_t)rpType]->BlockUntilReady(); @@ -1405,7 +1414,12 @@ void VulkanQueueRunner::PerformRenderPass(const VKRStep &step, VkCommandBuffer c { _assert_(step.render.pipelineFlags & PipelineFlags::USES_INPUT_ATTACHMENT); VulkanBarrier barrier; - SelfDependencyBarrier(step.render.framebuffer->color, VK_IMAGE_ASPECT_COLOR_BIT, &barrier); + if (step.render.framebuffer->sampleCount != VK_SAMPLE_COUNT_1_BIT) { + // Rendering is happening to the multisample buffer, not the color buffer. + SelfDependencyBarrier(step.render.framebuffer->msaaColor, VK_IMAGE_ASPECT_COLOR_BIT, &barrier); + } else { + SelfDependencyBarrier(step.render.framebuffer->color, VK_IMAGE_ASPECT_COLOR_BIT, &barrier); + } barrier.Flush(cmd); break; } @@ -1516,7 +1530,7 @@ void VulkanQueueRunner::PerformRenderPass(const VKRStep &step, VkCommandBuffer c VKRRenderPass *VulkanQueueRunner::PerformBindFramebufferAsRenderTarget(const VKRStep &step, VkCommandBuffer cmd) { VKRRenderPass *renderPass; int numClearVals = 0; - VkClearValue clearVal[2]{}; + VkClearValue clearVal[4]{}; VkFramebuffer framebuf; int w; int h; @@ -1563,15 +1577,22 @@ VKRRenderPass *VulkanQueueRunner::PerformBindFramebufferAsRenderTarget(const VKR // The transition from the optimal format happens after EndRenderPass, now that we don't // do it as part of the renderpass itself anymore. + if (sampleCount != VK_SAMPLE_COUNT_1_BIT) { + // We don't initialize values for these. + numClearVals = hasDepth ? 2 : 1; // Skip the resolve buffers, don't need to clear those. + } if (step.render.colorLoad == VKRRenderPassLoadAction::CLEAR) { - Uint8x4ToFloat4(clearVal[0].color.float32, step.render.clearColor); - numClearVals = 1; + Uint8x4ToFloat4(clearVal[numClearVals].color.float32, step.render.clearColor); } - if (hasDepth && (step.render.depthLoad == VKRRenderPassLoadAction::CLEAR || step.render.stencilLoad == VKRRenderPassLoadAction::CLEAR)) { - clearVal[1].depthStencil.depth = step.render.clearDepth; - clearVal[1].depthStencil.stencil = step.render.clearStencil; - numClearVals = 2; + numClearVals++; + if (hasDepth) { + if (step.render.depthLoad == VKRRenderPassLoadAction::CLEAR || step.render.stencilLoad == VKRRenderPassLoadAction::CLEAR) { + clearVal[numClearVals].depthStencil.depth = step.render.clearDepth; + clearVal[numClearVals].depthStencil.stencil = step.render.clearStencil; + } + numClearVals++; } + _dbg_assert_(numClearVals != 3); } else { RPKey key{ VKRRenderPassLoadAction::CLEAR, VKRRenderPassLoadAction::CLEAR, VKRRenderPassLoadAction::CLEAR, diff --git a/Common/GPU/Vulkan/VulkanRenderManager.cpp b/Common/GPU/Vulkan/VulkanRenderManager.cpp index 294f8d40d5..4a8c8c7227 100644 --- a/Common/GPU/Vulkan/VulkanRenderManager.cpp +++ b/Common/GPU/Vulkan/VulkanRenderManager.cpp @@ -27,7 +27,7 @@ using namespace PPSSPP_VK; // renderPass is an example of the "compatibility class" or RenderPassType type. -bool VKRGraphicsPipeline::Create(VulkanContext *vulkan, VkRenderPass compatibleRenderPass, RenderPassType rpType) { +bool VKRGraphicsPipeline::Create(VulkanContext *vulkan, VkRenderPass compatibleRenderPass, RenderPassType rpType, VkSampleCountFlagBits sampleCount) { // Fill in the last part of the desc since now it's time to block. VkShaderModule vs = desc->vertexShader->BlockUntilReady(); VkShaderModule fs = desc->fragmentShader->BlockUntilReady(); @@ -69,13 +69,16 @@ bool VKRGraphicsPipeline::Create(VulkanContext *vulkan, VkRenderPass compatibleR pipe.pDepthStencilState = &desc->dss; pipe.pRasterizationState = &desc->rs; + VkPipelineMultisampleStateCreateInfo ms{ VK_STRUCTURE_TYPE_PIPELINE_MULTISAMPLE_STATE_CREATE_INFO }; + ms.rasterizationSamples = sampleCount; + // We will use dynamic viewport state. pipe.pVertexInputState = &desc->vis; pipe.pViewportState = &desc->views; pipe.pTessellationState = nullptr; pipe.pDynamicState = &desc->ds; pipe.pInputAssemblyState = &desc->inputAssembly; - pipe.pMultisampleState = &desc->ms; + pipe.pMultisampleState = &ms; pipe.layout = desc->pipelineLayout; pipe.basePipelineHandle = VK_NULL_HANDLE; pipe.basePipelineIndex = 0; @@ -332,7 +335,7 @@ void VulkanRenderManager::CompileThreadFunc() { for (auto &entry : toCompile) { switch (entry.type) { case CompileQueueEntry::Type::GRAPHICS: - entry.graphics->Create(vulkan_, entry.compatibleRenderPass, entry.renderPassType); + entry.graphics->Create(vulkan_, entry.compatibleRenderPass, entry.renderPassType, entry.sampleCount); break; case CompileQueueEntry::Type::COMPUTE: entry.compute->Create(vulkan_); @@ -523,7 +526,7 @@ VKRGraphicsPipeline *VulkanRenderManager::CreateGraphicsPipeline(VKRGraphicsPipe } pipeline->pipeline[i] = Promise::CreateEmpty(); - compileQueue_.push_back(CompileQueueEntry(pipeline, compatibleRenderPass->Get(vulkan_, rpType, sampleCount), rpType)); + compileQueue_.push_back(CompileQueueEntry(pipeline, compatibleRenderPass->Get(vulkan_, rpType, sampleCount), rpType, sampleCount)); needsCompile = true; } if (needsCompile) @@ -591,7 +594,7 @@ void VulkanRenderManager::EndCurRenderStep() { for (VKRGraphicsPipeline *pipeline : pipelinesToCheck_) { if (!pipeline->pipeline[(size_t)rpType]) { pipeline->pipeline[(size_t)rpType] = Promise::CreateEmpty(); - compileQueue_.push_back(CompileQueueEntry(pipeline, renderPass->Get(vulkan_, rpType, sampleCount), rpType)); + compileQueue_.push_back(CompileQueueEntry(pipeline, renderPass->Get(vulkan_, rpType, sampleCount), rpType, sampleCount)); needsCompile = true; } } diff --git a/Common/GPU/Vulkan/VulkanRenderManager.h b/Common/GPU/Vulkan/VulkanRenderManager.h index 2e5cc1a1a5..a2c9abf9b4 100644 --- a/Common/GPU/Vulkan/VulkanRenderManager.h +++ b/Common/GPU/Vulkan/VulkanRenderManager.h @@ -82,7 +82,6 @@ struct VKRGraphicsPipelineDesc { VkDynamicState dynamicStates[6]{}; VkPipelineDynamicStateCreateInfo ds{ VK_STRUCTURE_TYPE_PIPELINE_DYNAMIC_STATE_CREATE_INFO }; VkPipelineRasterizationStateCreateInfo rs{ VK_STRUCTURE_TYPE_PIPELINE_RASTERIZATION_STATE_CREATE_INFO }; - VkPipelineMultisampleStateCreateInfo ms{ VK_STRUCTURE_TYPE_PIPELINE_MULTISAMPLE_STATE_CREATE_INFO }; // Replaced the ShaderStageInfo with promises here so we can wait for compiles to finish. Promise *vertexShader = nullptr; @@ -122,7 +121,7 @@ struct VKRGraphicsPipeline { } } - bool Create(VulkanContext *vulkan, VkRenderPass compatibleRenderPass, RenderPassType rpType); + bool Create(VulkanContext *vulkan, VkRenderPass compatibleRenderPass, RenderPassType rpType, VkSampleCountFlagBits sampleCount); // This deletes the whole VKRGraphicsPipeline, you must remove your last pointer to it when doing this. void QueueForDeletion(VulkanContext *vulkan); @@ -151,9 +150,9 @@ struct VKRComputePipeline { }; struct CompileQueueEntry { - CompileQueueEntry(VKRGraphicsPipeline *p, VkRenderPass _compatibleRenderPass, RenderPassType _renderPassType) - : type(Type::GRAPHICS), graphics(p), compatibleRenderPass(_compatibleRenderPass), renderPassType(_renderPassType) {} - CompileQueueEntry(VKRComputePipeline *p) : type(Type::COMPUTE), compute(p), renderPassType(RenderPassType::HAS_DEPTH) {} // renderpasstype here shouldn't matter + CompileQueueEntry(VKRGraphicsPipeline *p, VkRenderPass _compatibleRenderPass, RenderPassType _renderPassType, VkSampleCountFlagBits _sampleCount) + : type(Type::GRAPHICS), graphics(p), compatibleRenderPass(_compatibleRenderPass), renderPassType(_renderPassType), sampleCount(_sampleCount) {} + CompileQueueEntry(VKRComputePipeline *p) : type(Type::COMPUTE), compute(p), renderPassType(RenderPassType::HAS_DEPTH), sampleCount(VK_SAMPLE_COUNT_1_BIT) {} // renderpasstype here shouldn't matter enum class Type { GRAPHICS, COMPUTE, @@ -163,6 +162,7 @@ struct CompileQueueEntry { RenderPassType renderPassType; VKRGraphicsPipeline *graphics = nullptr; VKRComputePipeline *compute = nullptr; + VkSampleCountFlagBits sampleCount; }; class VulkanRenderManager { diff --git a/Common/GPU/Vulkan/thin3d_vulkan.cpp b/Common/GPU/Vulkan/thin3d_vulkan.cpp index b944a1962c..b7a0bb2ed1 100644 --- a/Common/GPU/Vulkan/thin3d_vulkan.cpp +++ b/Common/GPU/Vulkan/thin3d_vulkan.cpp @@ -832,6 +832,7 @@ VKContext::VKContext(VulkanContext *vulkan) caps_.blendMinMaxSupported = true; caps_.logicOpSupported = vulkan->GetDeviceFeatures().enabled.standard.logicOp != 0; caps_.multiViewSupported = vulkan->GetDeviceFeatures().enabled.multiview.multiview != 0; + const auto &limits = vulkan->GetPhysicalDeviceProperties().properties.limits; auto deviceProps = vulkan->GetPhysicalDeviceProperties(vulkan_->GetCurrentPhysicalDeviceIndex()).properties; @@ -853,6 +854,14 @@ VKContext::VKContext(VulkanContext *vulkan) } caps_.isTilingGPU = hasLazyMemory; + // VkSampleCountFlagBits is arranged correctly for our purposes. + // Only support MSAA levels that have support for all three of color, depth, stencil. + if (!caps_.isTilingGPU) { + caps_.multiSampleLevelsMask = (limits.framebufferColorSampleCounts & limits.framebufferDepthSampleCounts & limits.framebufferStencilSampleCounts); + } else { + caps_.multiSampleLevelsMask = 1; + } + if (caps_.vendor == GPUVendor::VENDOR_QUALCOMM) { // Adreno 5xx devices, all known driver versions, fail to discard stencil when depth write is off. // See: https://github.com/hrydgard/ppsspp/pull/11684 @@ -1197,9 +1206,6 @@ Pipeline *VKContext::CreateGraphicsPipeline(const PipelineDesc &desc, const char } gDesc.ds.pDynamicStates = gDesc.dynamicStates; - gDesc.ms.pSampleMask = nullptr; - gDesc.ms.rasterizationSamples = VK_SAMPLE_COUNT_1_BIT; - gDesc.views.viewportCount = 1; gDesc.views.scissorCount = 1; gDesc.views.pViewports = nullptr; // dynamic @@ -1741,7 +1747,7 @@ uint64_t VKContext::GetNativeObject(NativeObject obj, void *srcObject) { case NativeObject::BOUND_FRAMEBUFFER_COLOR_IMAGEVIEW_ALL_LAYERS: return (uint64_t)curFramebuffer_->GetFB()->color.texAllLayersView; case NativeObject::BOUND_FRAMEBUFFER_COLOR_IMAGEVIEW_RT: - return (uint64_t)curFramebuffer_->GetFB()->color.rtView; + return (uint64_t)curFramebuffer_->GetFB()->GetRTView(); case NativeObject::FRAME_DATA_DESC_SET_LAYOUT: return (uint64_t)frameDescSetLayout_; case NativeObject::THIN3D_PIPELINE_LAYOUT: diff --git a/Common/GPU/thin3d.h b/Common/GPU/thin3d.h index 9cc732b86a..cfe4af4b34 100644 --- a/Common/GPU/thin3d.h +++ b/Common/GPU/thin3d.h @@ -579,6 +579,7 @@ struct DeviceCaps { bool multiViewSupported; bool isTilingGPU; // This means that it benefits from correct store-ops, msaa without backing memory, etc. + u32 multiSampleLevelsMask; // Bit n is set if (1 << n) is a valid multisample level. Bit 0 is always set. std::string deviceName; // The device name to use when creating the thin3d context, to get the same one. }; diff --git a/Core/Config.cpp b/Core/Config.cpp index e7b10013af..bd3e2dccf2 100644 --- a/Core/Config.cpp +++ b/Core/Config.cpp @@ -901,7 +901,8 @@ static ConfigSetting graphicsSettings[] = { // Most low-performance (and many high performance) mobile GPUs do not support aniso anyway so defaulting to 4 is fine. ConfigSetting("AnisotropyLevel", &g_Config.iAnisotropyLevel, 4, true, true), - ConfigSetting("MultiSampleLevel", &g_Config.iMultiSampleLevel, 1, true, true), + ConfigSetting("MultiSampleLevel", &g_Config.iMultiSampleLevel, 0, true, true), // Number of samples is 1 << iMultiSampleLevel + ConfigSetting("MultiSampleQuality", &g_Config.iMultiSampleQuality, 0, true, true), // 0 = multisampling, 1 = halfway to SGSSAA, 2 = SGSSAA. ReportedConfigSetting("VertexDecCache", &g_Config.bVertexCache, false, true, true), ReportedConfigSetting("TextureBackoffCache", &g_Config.bTextureBackoffCache, false, true, true), diff --git a/Core/Config.h b/Core/Config.h index 342b5c3450..05244592be 100644 --- a/Core/Config.h +++ b/Core/Config.h @@ -211,6 +211,7 @@ public: int iInternalResolution; // 0 = Auto (native), 1 = 1x (480x272), 2 = 2x, 3 = 3x, 4 = 4x and so on. int iAnisotropyLevel; // 0 - 5, powers of 2: 0 = 1x = no aniso int iMultiSampleLevel; + int iMultiSampleQuality; int bHighQualityDepth; bool bReplaceTextures; bool bSaveNewTextures; diff --git a/GPU/Common/FramebufferManagerCommon.cpp b/GPU/Common/FramebufferManagerCommon.cpp index e8f3139a10..4fc69de515 100644 --- a/GPU/Common/FramebufferManagerCommon.cpp +++ b/GPU/Common/FramebufferManagerCommon.cpp @@ -1652,7 +1652,10 @@ void FramebufferManagerCommon::ResizeFramebufFBO(VirtualFramebuffer *vfb, int w, shaderManager_->DirtyLastShader(); char tag[128]; size_t len = FormatFramebufferName(vfb, tag, sizeof(tag)); - vfb->fbo = draw_->CreateFramebuffer({ vfb->renderWidth, vfb->renderHeight, 1, GetFramebufferLayers(), 1, true, tag }); + + int msaaLevel = g_Config.iMultiSampleLevel; + + vfb->fbo = draw_->CreateFramebuffer({ vfb->renderWidth, vfb->renderHeight, 1, GetFramebufferLayers(), msaaLevel, true, tag }); if (Memory::IsVRAMAddress(vfb->fb_address) && vfb->fb_stride != 0) { NotifyMemInfo(MemBlockFlags::ALLOC, vfb->fb_address, ColorBufferByteSize(vfb), tag, len); } diff --git a/GPU/Vulkan/GPU_Vulkan.cpp b/GPU/Vulkan/GPU_Vulkan.cpp index 0c50c361bd..f735f3267f 100644 --- a/GPU/Vulkan/GPU_Vulkan.cpp +++ b/GPU/Vulkan/GPU_Vulkan.cpp @@ -269,6 +269,12 @@ u32 GPU_Vulkan::CheckGPUFeatures() const { } } + // We need to turn off framebuffer fetch through input attachments if MSAA is on for now. + // This is fixable, just needs some shader generator work (subpassInputMS). + if (g_Config.iMultiSampleLevel != 0) { + features &= ~GPU_USE_FRAMEBUFFER_FETCH; + } + return CheckGPUFeaturesLate(features); } diff --git a/GPU/Vulkan/PipelineManagerVulkan.cpp b/GPU/Vulkan/PipelineManagerVulkan.cpp index 61ff24fb2c..e52df37028 100644 --- a/GPU/Vulkan/PipelineManagerVulkan.cpp +++ b/GPU/Vulkan/PipelineManagerVulkan.cpp @@ -250,17 +250,13 @@ static VulkanPipeline *CreateVulkanPipeline(VulkanRenderManager *renderManager, rs.polygonMode = VK_POLYGON_MODE_FILL; rs.depthClampEnable = key.depthClampEnable; - VkPipelineMultisampleStateCreateInfo &ms = desc->ms; - ms.pSampleMask = nullptr; - ms.rasterizationSamples = VK_SAMPLE_COUNT_1_BIT; - desc->fragmentShader = fs->GetModule(); desc->vertexShader = vs->GetModule(); desc->geometryShader = gs ? gs->GetModule() : nullptr; desc->fragmentShaderSource = fs->GetShaderString(SHADER_STRING_SOURCE_CODE); desc->vertexShaderSource = vs->GetShaderString(SHADER_STRING_SOURCE_CODE); if (gs) { - desc->geometryShaderSource = gs->GetShaderString(SHADER_STRING_SOURCE_CODE); + desc->geometryShaderSource = gs->GetShaderString(SHADER_STRING_SOURCE_CODE); } VkPipelineInputAssemblyStateCreateInfo &inputAssembly = desc->inputAssembly; @@ -352,7 +348,7 @@ VulkanPipeline *PipelineManagerVulkan::GetOrCreatePipeline(VulkanRenderManager * pipelineFlags |= PipelineFlags::USES_MULTIVIEW; } - VkSampleCountFlagBits sampleCount = SampleCountToFlagBits(g_Config.iMultiSampleLevel); + VkSampleCountFlagBits sampleCount = MultiSampleLevelToFlagBits(g_Config.iMultiSampleLevel); VulkanPipeline *pipeline = CreateVulkanPipeline( renderManager, pipelineCache_, layout, pipelineFlags, sampleCount, diff --git a/UI/GameSettingsScreen.cpp b/UI/GameSettingsScreen.cpp index b2ad14f16e..89ff1b49d7 100644 --- a/UI/GameSettingsScreen.cpp +++ b/UI/GameSettingsScreen.cpp @@ -305,6 +305,21 @@ void GameSettingsScreen::CreateViews() { return !g_Config.bSoftwareRendering && !g_Config.bSkipBufferEffects; }); + if (draw->GetDeviceCaps().multiSampleLevelsMask != 1) { + static const char *msaaModes[] = { "Off", "2xMSAA", "4xMSAA", "8xMSAA", "16xMSAA" }; + auto msaaChoice = graphicsSettings->Add(new PopupMultiChoice(&g_Config.iMultiSampleLevel, gr->T("Antialiasing (MSAA)"), msaaModes, 0, ARRAY_SIZE(msaaModes), gr->GetName(), screenManager())); + msaaChoice->OnChoice.Add([&](UI::EventParams &) -> UI::EventReturn { + NativeMessageReceived("gpu_renderResized", ""); + return UI::EVENT_DONE; + }); + // Hide unsupported levels. + for (int i = 1; i < 5; i++) { + if ((draw->GetDeviceCaps().multiSampleLevelsMask & (1 << i)) == 0) { + msaaChoice->HideChoice(i); + } + } + } + #if PPSSPP_PLATFORM(ANDROID) if ((deviceType != DEVICE_TYPE_TV) && (deviceType != DEVICE_TYPE_VR)) { static const char *deviceResolutions[] = { "Native device resolution", "Auto (same as Rendering)", "1x PSP", "2x PSP", "3x PSP", "4x PSP", "5x PSP" };