diff --git a/.gitignore b/.gitignore
index b7653209ae..fd1d2d372f 100644
--- a/.gitignore
+++ b/.gitignore
@@ -129,3 +129,5 @@ CMakeFiles
# Clangd
.cache/
+build
+libretro/obj/local
diff --git a/.gitlab-ci.yml b/.gitlab-ci.yml
index 84881497b9..ec27838610 100644
--- a/.gitlab-ci.yml
+++ b/.gitlab-ci.yml
@@ -49,6 +49,14 @@ include:
- project: 'libretro-infrastructure/ci-templates'
file: '/android-cmake.yml'
+ # iOS
+ - project: 'libretro-infrastructure/ci-templates'
+ file: '/ios-cmake.yml'
+
+ # tvOS
+ - project: 'libretro-infrastructure/ci-templates'
+ file: '/tvos-cmake.yml'
+
################################## CONSOLES ################################
#################################### MISC ##################################
@@ -98,7 +106,7 @@ libretro-build-osx-x64:
extends:
- .libretro-osx-cmake-x86_64
- .core-defs
- - .make-defs
+ - .cmake-defs
# MacOS 64-bit
libretro-build-osx-arm64:
@@ -107,7 +115,7 @@ libretro-build-osx-arm64:
extends:
- .libretro-osx-cmake-arm64
- .core-defs
- - .make-defs
+ - .cmake-defs
################################### CELLULAR #################################
# Android ARMv7a
@@ -133,3 +141,21 @@ libretro-build-android-x86:
extends:
- .libretro-android-cmake-x86
- .core-defs
+
+# iOS arm64
+libretro-build-ios-arm64:
+ extends:
+ - .libretro-ios-cmake-arm64
+ - .core-defs
+ - .cmake-defs
+ variables:
+ CORE_ARGS: -DCMAKE_TOOLCHAIN_FILE=cmake/Toolchains/ios.cmake -DLIBRETRO=ON
+
+# tvOS arm64
+libretro-build-tvos-arm64:
+ extends:
+ - .libretro-tvos-cmake-arm64
+ - .core-defs
+ - .cmake-defs
+ variables:
+ CORE_ARGS: -DIOS_PLATFORM=TVOS -DUSE_FFMPEG=NO -DCMAKE_TOOLCHAIN_FILE=cmake/Toolchains/ios.cmake -DLIBRETRO=ON
diff --git a/.gitmodules b/.gitmodules
index da3169b8eb..a3c92a9929 100644
--- a/.gitmodules
+++ b/.gitmodules
@@ -1,3 +1,6 @@
+[submodule "libretro/libretro-common"]
+ path = libretro/libretro-common
+ url = https://github.com/libretro/libretro-common.git
[submodule "pspautotests"]
path = pspautotests
url = https://github.com/hrydgard/pspautotests.git
diff --git a/CMakeLists.txt b/CMakeLists.txt
index 27dfcf93ff..a8ac6c29b2 100644
--- a/CMakeLists.txt
+++ b/CMakeLists.txt
@@ -119,16 +119,8 @@ add_definitions(-DASSETS_DIR="${CMAKE_INSTALL_FULL_DATADIR}/ppsspp/assets/")
if(OPENXR)
add_definitions(-DOPENXR)
add_library(openxr SHARED IMPORTED)
- if(OPENXR_PLATFORM_PICO)
- add_definitions(-DOPENXR_PLATFORM_PICO)
- set_property(TARGET openxr PROPERTY IMPORTED_LOCATION "${CMAKE_SOURCE_DIR}/ext/openxr/libs/pico/arm64-v8a/libopenxr_loader.so")
- message("OpenXR for Pico enabled")
- endif()
- if(OPENXR_PLATFORM_QUEST)
- add_definitions(-DOPENXR_PLATFORM_QUEST)
- set_property(TARGET openxr PROPERTY IMPORTED_LOCATION "${CMAKE_SOURCE_DIR}/ext/openxr/libs/quest/arm64-v8a/libopenxr_loader.so")
- message("OpenXR for Quest enabled")
- endif()
+ set_property(TARGET openxr PROPERTY IMPORTED_LOCATION "${CMAKE_SOURCE_DIR}/ext/openxr/stub/arm64-v8a/libopenxr_loader.so")
+ message("OpenXR enabled")
endif()
if(GOLD)
@@ -376,12 +368,20 @@ if(NOT MSVC)
if(X86 OR X86_64)
# enable sse2 code generation
add_definitions(-msse2)
+ if(NOT X86_64 AND NOT CLANG)
+ add_definitions(-mfpmath=sse)
+ # add_definitions(-mstackrealign)
+ endif()
endif()
if(IOS)
set(CMAKE_OSX_DEPLOYMENT_TARGET "11.0")
elseif(APPLE AND NOT CMAKE_CROSSCOMPILING)
- set(CMAKE_OSX_DEPLOYMENT_TARGET "10.13")
+ if(LIBRETRO AND ARM64)
+ set(CMAKE_OSX_DEPLOYMENT_TARGET "10.14")
+ else()
+ set(CMAKE_OSX_DEPLOYMENT_TARGET "10.13")
+ endif()
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -stdlib=libc++ -U__STRICT_ANSI__")
set(CMAKE_XCODE_ATTRIBUTE_CLANG_CXX_LIBRARY "libc++")
elseif(NOT ANDROID)
@@ -575,6 +575,8 @@ add_library(Common STATIC
Common/Data/Format/JSONReader.cpp
Common/Data/Format/JSONWriter.h
Common/Data/Format/JSONWriter.cpp
+ Common/Data/Format/DDSLoad.cpp
+ Common/Data/Format/DDSLoad.h
Common/Data/Format/PNGLoad.cpp
Common/Data/Format/PNGLoad.h
Common/Data/Format/ZIMLoad.cpp
@@ -592,8 +594,10 @@ add_library(Common STATIC
Common/Data/Random/Rng.h
Common/File/VFS/VFS.h
Common/File/VFS/VFS.cpp
- Common/File/VFS/AssetReader.cpp
- Common/File/VFS/AssetReader.h
+ Common/File/VFS/ZipFileReader.cpp
+ Common/File/VFS/ZipFileReader.h
+ Common/File/VFS/DirectoryReader.cpp
+ Common/File/VFS/DirectoryReader.h
Common/File/AndroidStorage.h
Common/File/AndroidStorage.cpp
Common/File/DiskFree.h
@@ -628,6 +632,8 @@ add_library(Common STATIC
Common/GPU/OpenGL/gl3stub.h
Common/GPU/OpenGL/GLFeatures.cpp
Common/GPU/OpenGL/GLFeatures.h
+ Common/GPU/OpenGL/GLFrameData.cpp
+ Common/GPU/OpenGL/GLFrameData.h
Common/GPU/OpenGL/thin3d_gl.cpp
Common/GPU/OpenGL/GLRenderManager.cpp
Common/GPU/OpenGL/GLRenderManager.h
@@ -710,6 +716,10 @@ add_library(Common STATIC
Common/Render/Text/draw_text_uwp.h
Common/System/Display.cpp
Common/System/Display.h
+ Common/System/System.h
+ Common/System/NativeApp.h
+ Common/System/Request.cpp
+ Common/System/Request.h
Common/Thread/Channel.h
Common/Thread/ParallelLoop.cpp
Common/Thread/ParallelLoop.h
@@ -1097,6 +1107,28 @@ else()
include_directories(ext/libpng17)
endif()
+add_library(basis_universal STATIC
+ ext/basis_universal/basisu.h
+ ext/basis_universal/basisu_containers.h
+ ext/basis_universal/basisu_containers_impl.h
+ ext/basis_universal/basisu_file_headers.h
+ ext/basis_universal/basisu_transcoder.cpp
+ ext/basis_universal/basisu_transcoder.h
+ ext/basis_universal/basisu_transcoder_internal.h
+ ext/basis_universal/basisu_transcoder_tables_astc.inc
+ ext/basis_universal/basisu_transcoder_tables_astc_0_255.inc
+ ext/basis_universal/basisu_transcoder_tables_atc_55.inc
+ ext/basis_universal/basisu_transcoder_tables_atc_56.inc
+ ext/basis_universal/basisu_transcoder_tables_bc7_m5_alpha.inc
+ ext/basis_universal/basisu_transcoder_tables_bc7_m5_color.inc
+ ext/basis_universal/basisu_transcoder_tables_dxt1_5.inc
+ ext/basis_universal/basisu_transcoder_tables_dxt1_6.inc
+ ext/basis_universal/basisu_transcoder_tables_pvrtc2_45.inc
+ ext/basis_universal/basisu_transcoder_tables_pvrtc2_alpha_33.inc
+ ext/basis_universal/basisu_transcoder_uastc.h
+)
+set(BASISU_LIBRARIES basis_universal)
+
set(nativeExtra)
set(nativeExtraLibs)
@@ -1126,7 +1158,7 @@ if(ANDROID)
set(nativeExtraLibs ${nativeExtraLibs} openxr)
endif()
# No target
-elseif(IOS)
+elseif(IOS AND NOT LIBRETRO)
set(nativeExtra ${nativeExtra}
ios/main.mm
ios/AppDelegate.mm
@@ -1145,8 +1177,6 @@ elseif(IOS)
ios/PPSSPPUIApplication.mm
ios/SmartKeyboardMap.cpp
ios/SmartKeyboardMap.hpp
- ios/SubtleVolume.h
- ios/SubtleVolume.mm
ios/iCade/iCadeReaderView.h
ios/iCade/iCadeReaderView.m
ios/iCade/iCadeState.h
@@ -1154,7 +1184,7 @@ elseif(IOS)
UI/DarwinFileSystemServices.h
Common/Battery/AppleBatteryClient.m
)
-
+
set(nativeExtraLibs ${nativeExtraLibs} "-framework Foundation -framework MediaPlayer -framework AudioToolbox -framework CoreGraphics -framework QuartzCore -framework UIKit -framework GLKit -framework OpenAL -framework AVFoundation -framework CoreLocation -framework CoreVideo -framework CoreMedia -framework CoreServices" )
if(EXISTS "${CMAKE_IOS_SDK_ROOT}/System/Library/Frameworks/GameController.framework")
set(nativeExtraLibs ${nativeExtraLibs} "-weak_framework GameController")
@@ -1175,6 +1205,8 @@ elseif(IOS)
set_source_files_properties(UI/DarwinFileSystemServices.mm PROPERTIES COMPILE_FLAGS -fobjc-arc)
set_source_files_properties(Common/Battery/AppleBatteryClient.m PROPERTIES COMPILE_FLAGS -fobjc-arc)
set(TargetBin PPSSPP)
+elseif(IOS AND LIBRETRO)
+ set(nativeExtraLibs ${nativeExtraLibs} "-framework GLKit")
elseif(USING_QT_UI)
set(CMAKE_AUTOMOC ON)
find_package(Qt5 COMPONENTS OpenGL Gui Core Multimedia)
@@ -1190,7 +1222,15 @@ elseif(USING_QT_UI)
if(USING_GLES2)
add_definitions(-DQT_OPENGL_ES -DQT_OPENGL_ES_2)
endif()
-
+ if(APPLE)
+ list(APPEND NativeAppSource
+ UI/DarwinFileSystemServices.mm
+ UI/DarwinFileSystemServices.h
+ Common/Battery/AppleBatteryClient.m)
+ set_source_files_properties(Common/Battery/AppleBatteryClient.m PROPERTIES COMPILE_FLAGS -fobjc-arc)
+ set_source_files_properties(UI/DarwinFileSystemServices.mm PROPERTIES COMPILE_FLAGS -fobjc-arc)
+ set(nativeExtraLibs ${nativeExtraLibs} ${COCOA_LIBRARY} ${QUARTZ_CORE_LIBRARY} ${IOKIT_LIBRARY})
+ endif()
include_directories(Qt)
include_directories(${CMAKE_CURRENT_BINARY_DIR})
set(nativeExtraLibs ${nativeExtraLibs} Qt5::OpenGL Qt5::Gui Qt5::Core Qt5::Multimedia)
@@ -1349,11 +1389,11 @@ if(LINUX AND NOT ANDROID)
endif()
set(ATOMIC_LIB)
-if(ANDROID OR (LINUX AND ARM_DEVICE))
+if(ANDROID OR (LINUX AND ARM_DEVICE) OR (LINUX AND RISCV64))
set(ATOMIC_LIB atomic)
endif()
-target_link_libraries(native ${LIBZIP_LIBRARY} ${PNG_LIBRARIES} ${ZLIB_LIBRARY} vma gason udis86 ${RT_LIB} ${nativeExtraLibs} ${ATOMIC_LIB} Common)
+target_link_libraries(native ${LIBZIP_LIBRARY} ${PNG_LIBRARIES} ${BASISU_LIBRARIES} ${ZLIB_LIBRARY} vma gason udis86 ${RT_LIB} ${nativeExtraLibs} ${ATOMIC_LIB} Common)
if(TARGET Ext::GLEW)
target_link_libraries(native Ext::GLEW)
endif()
@@ -1488,6 +1528,10 @@ list(APPEND CoreExtra
Core/MIPS/MIPS/MipsJit.h
)
+list(APPEND CoreExtra
+ GPU/Common/VertexDecoderRiscV.cpp
+)
+
if(NOT MOBILE_DEVICE)
set(CoreExtra ${CoreExtra}
Core/AVIDump.cpp
@@ -1498,7 +1542,7 @@ if(NOT MOBILE_DEVICE)
endif()
set(GPU_GLES
- GPU/GLES/DepthBufferGLES.cpp
+ GPU/GLES/StencilBufferGLES.cpp
GPU/GLES/GPU_GLES.cpp
GPU/GLES/GPU_GLES.h
GPU/GLES/FragmentTestCacheGLES.cpp
@@ -1579,6 +1623,7 @@ set(GPU_SOURCES
${GPU_NEON}
GPU/Common/Draw2D.cpp
GPU/Common/Draw2D.h
+ GPU/Common/DepthBufferCommon.cpp
GPU/Common/TextureShaderCommon.cpp
GPU/Common/TextureShaderCommon.h
GPU/Common/DepalettizeShaderCommon.cpp
@@ -1627,6 +1672,10 @@ set(GPU_SOURCES
GPU/Common/TextureScalerCommon.h
GPU/Common/PostShader.cpp
GPU/Common/PostShader.h
+ GPU/Common/TextureReplacer.cpp
+ GPU/Common/TextureReplacer.h
+ GPU/Common/ReplacedTexture.cpp
+ GPU/Common/ReplacedTexture.h
GPU/Debugger/Breakpoints.cpp
GPU/Debugger/Breakpoints.h
GPU/Debugger/Debugger.cpp
@@ -1649,6 +1698,8 @@ set(GPU_SOURCES
GPU/GPU.h
GPU/GPUCommon.cpp
GPU/GPUCommon.h
+ GPU/GPUCommonHW.cpp
+ GPU/GPUCommonHW.h
GPU/GPUState.cpp
GPU/GPUState.h
GPU/Math3D.cpp
@@ -2033,8 +2084,6 @@ add_library(${CoreLibName} ${CoreLinkType}
Core/Screenshot.h
Core/System.cpp
Core/System.h
- Core/TextureReplacer.cpp
- Core/TextureReplacer.h
Core/ThreadPools.cpp
Core/ThreadPools.h
Core/Util/AudioFormat.cpp
@@ -2491,7 +2540,7 @@ if(NOT ANDROID)
file(INSTALL assets/flash0 DESTINATION assets)
endif()
# packaging and code signing
-if(IOS)
+if(IOS AND NOT LIBRETRO)
set(DEPLOYMENT_TARGET 11.0)
file(GLOB IOSAssets ios/assets/*.png)
list(REMOVE_ITEM IOSAssets ${CMAKE_CURRENT_SOURCE_DIR}/ios/assets/Default-568h@2x.png)
diff --git a/Common/CPUDetect.h b/Common/CPUDetect.h
index 13e1170acc..1feb20483f 100644
--- a/Common/CPUDetect.h
+++ b/Common/CPUDetect.h
@@ -111,6 +111,10 @@ struct CPUInfo {
bool RiscV_V;
bool RiscV_B;
bool RiscV_Zicsr;
+ bool RiscV_Zba;
+ bool RiscV_Zbb;
+ bool RiscV_Zbc;
+ bool RiscV_Zbs;
// Quirks
struct {
diff --git a/Common/Common.vcxproj b/Common/Common.vcxproj
index c3c1ebebaf..3fabaafa64 100644
--- a/Common/Common.vcxproj
+++ b/Common/Common.vcxproj
@@ -377,6 +377,13 @@
+
+
+
+
+
+
+
@@ -405,6 +412,7 @@
+
@@ -425,8 +433,9 @@
+
-
+
@@ -436,6 +445,7 @@
+
@@ -540,6 +550,7 @@
+
@@ -574,6 +585,7 @@
+
NotUsing
NotUsing
@@ -842,6 +854,7 @@
+
@@ -860,8 +873,9 @@
+
-
+
@@ -871,6 +885,7 @@
+
@@ -991,6 +1006,7 @@
+
@@ -1028,6 +1044,18 @@
{f761046e-6c38-4428-a5f1-38391a37bb34}
+
+
+
+
+
+
+
+
+
+
+
+
diff --git a/Common/Common.vcxproj.filters b/Common/Common.vcxproj.filters
index 6dfa552f1b..879e3495b8 100644
--- a/Common/Common.vcxproj.filters
+++ b/Common/Common.vcxproj.filters
@@ -176,9 +176,6 @@
File\VFS
-
- File\VFS
-
Data\Format
@@ -464,6 +461,42 @@
UI
+
+ GPU\OpenGL
+
+
+ File\VFS
+
+
+ File\VFS
+
+
+ Data\Format
+
+
+ ext\basis_universal
+
+
+ ext\basis_universal
+
+
+ ext\basis_universal
+
+
+ ext\basis_universal
+
+
+ ext\basis_universal
+
+
+ ext\basis_universal
+
+
+ ext\basis_universal
+
+
+ System
+
@@ -625,9 +658,6 @@
File\VFS
-
- File\VFS
-
Data\Format
@@ -878,6 +908,24 @@
UI
+
+ GPU\OpenGL
+
+
+ File\VFS
+
+
+ File\VFS
+
+
+ Data\Format
+
+
+ ext\basis_universal
+
+
+ System
+
@@ -985,10 +1033,45 @@
{9d1c29fd-8ac7-4475-8ea6-c8c759b695fe}
+
+ {d6d5f6e0-1c72-496b-af11-6d52d5123033}
+
ext\libpng17
+
+
+ ext\basis_universal
+
+
+ ext\basis_universal
+
+
+ ext\basis_universal
+
+
+ ext\basis_universal
+
+
+ ext\basis_universal
+
+
+ ext\basis_universal
+
+
+ ext\basis_universal
+
+
+ ext\basis_universal
+
+
+ ext\basis_universal
+
+
+ ext\basis_universal
+
+
\ No newline at end of file
diff --git a/Common/CommonFuncs.h b/Common/CommonFuncs.h
index 848a3eae83..2b2d46f91a 100644
--- a/Common/CommonFuncs.h
+++ b/Common/CommonFuncs.h
@@ -36,6 +36,8 @@
#define Crash() {asm ("bkpt #0");}
#elif PPSSPP_ARCH(ARM64)
#define Crash() {asm ("brk #0");}
+#elif PPSSPP_ARCH(RISCV64)
+#define Crash() {asm ("ebreak");}
#else
#include
#define Crash() {kill(getpid(), SIGINT);}
diff --git a/Common/Data/Format/DDSLoad.cpp b/Common/Data/Format/DDSLoad.cpp
new file mode 100644
index 0000000000..1d3bdb4d1d
--- /dev/null
+++ b/Common/Data/Format/DDSLoad.cpp
@@ -0,0 +1,5 @@
+#include "Common/Data/Format/DDSLoad.h"
+
+bool DetectDDSParams(const DDSHeader *header, DDSLoadInfo *info) {
+ return false;
+}
diff --git a/Common/Data/Format/DDSLoad.h b/Common/Data/Format/DDSLoad.h
new file mode 100644
index 0000000000..3f7d36cf27
--- /dev/null
+++ b/Common/Data/Format/DDSLoad.h
@@ -0,0 +1,83 @@
+#pragma once
+
+#include
+
+// DDSPixelFormat.dwFlags bits
+enum {
+ DDPF_ALPHAPIXELS = 1, // Texture contains alpha data; dwRGBAlphaBitMask contains valid data.
+ DDPF_ALPHA = 2, // Used in some older DDS files for alpha channel only uncompressed data(dwRGBBitCount contains the alpha channel bitcount; dwABitMask contains valid data)
+ DDPF_FOURCC = 4, // Texture contains compressed RGB data; dwFourCC contains valid data. 0x4
+ DDPF_RGB = 8, // Texture contains uncompressed RGB data; dwRGBBitCount and the RGB masks(dwRBitMask, dwGBitMask, dwBBitMask) contain valid data. 0x40
+ DDPF_YUV = 16, //Used in some older DDS files for YUV uncompressed data(dwRGBBitCount contains the YUV bit count; dwRBitMask contains the Y mask, dwGBitMask contains the U mask, dwBBitMask contains the V mask)
+ DDPF_LUMINANCE = 32, // Used in some older DDS files for single channel color uncompressed data (dwRGBBitCount contains the luminance channel bit count; dwRBitMask contains the channel mask). Can be combined with DDPF_ALPHAPIXELS for a two channel DDS file.
+};
+
+// dwCaps members
+enum {
+ DDSCAPS_COMPLEX = 8,
+ DDSCAPS_MIPMAP = 0x400000,
+ DDSCAPS_TEXTURE = 0x1000, // Required
+};
+
+// Not using any D3D headers here, this is cross platform and minimal.
+// Boiled down from the public documentation.
+struct DDSPixelFormat {
+ uint32_t dwSize; // must be 32
+ uint32_t dwFlags;
+ uint32_t dwFourCC;
+ uint32_t dwRGBBitCount;
+ uint32_t dwRBitMask;
+ uint32_t dwGBitMask;
+ uint32_t dwBBitMask;
+ uint32_t dwABitMask;
+};
+
+struct DDSHeader {
+ uint32_t dwMagic; // Magic is not technically part of the header struct but convenient to have here when reading the files.
+ uint32_t dwSize; // must be 124
+ uint32_t dwFlags;
+ uint32_t dwHeight;
+ uint32_t dwWidth;
+ uint32_t dwPitchOrLinearSize; // The pitch or number of bytes per scan line in an uncompressed texture; the total number of bytes in the top level texture for a compressed texture
+ uint32_t dwDepth; // we'll always use 1 here
+ uint32_t dwMipMapCount;
+ uint32_t dwReserved1[11];
+ DDSPixelFormat ddspf;
+ uint32_t dwCaps;
+ uint32_t dwCaps2; // nothing we care about
+ uint32_t dwCaps3; // unused
+ uint32_t dwCaps4; // unused
+ uint32_t dwReserved2;
+};
+
+// DDS header extension to handle resource arrays, DXGI pixel formats that don't map to the legacy Microsoft DirectDraw pixel format structures, and additional metadata.
+struct DDSHeaderDXT10 {
+ uint32_t dxgiFormat;
+ uint32_t resourceDimension; // 1d = 2, 2d = 3, 3d = 4. very intuitive
+ uint32_t miscFlag;
+ uint32_t arraySize; // we only support 1 here
+ uint32_t miscFlags2; // sets alpha interpretation, let's not bother
+};
+
+// Simple DDS parser, suitable for texture replacement packs.
+// Doesn't actually load, only does some logic to fill out DDSLoadInfo so the caller can then
+// do the actual load with a simple series of memcpys or whatever is appropriate.
+struct DDSLoadInfo {
+ uint32_t bytesToCopy;
+};
+
+bool DetectDDSParams(const DDSHeader *header, DDSLoadInfo *info);
+
+// We use the Basis library for the actual reading, but before we do, we pre-scan using this, similarly to the png trick.
+struct KTXHeader {
+ uint8_t identifier[12];
+ uint32_t vkFormat;
+ uint32_t typeSize;
+ uint32_t pixelWidth;
+ uint32_t pixelHeight;
+ uint32_t pixelDepth;
+ uint32_t layerCount;
+ uint32_t faceCount;
+ uint32_t levelCount;
+ uint32_t supercompressionScheme;
+};
diff --git a/Common/Data/Format/IniFile.cpp b/Common/Data/Format/IniFile.cpp
index 72aeee6f89..91c042d5b4 100644
--- a/Common/Data/Format/IniFile.cpp
+++ b/Common/Data/Format/IniFile.cpp
@@ -567,9 +567,9 @@ bool IniFile::Load(const Path &path)
return success;
}
-bool IniFile::LoadFromVFS(const std::string &filename) {
+bool IniFile::LoadFromVFS(VFSInterface &vfs, const std::string &filename) {
size_t size;
- uint8_t *data = VFSReadFile(filename.c_str(), &size);
+ uint8_t *data = vfs.ReadFile(filename.c_str(), &size);
if (!data)
return false;
std::string str((const char*)data, size);
@@ -582,10 +582,10 @@ bool IniFile::LoadFromVFS(const std::string &filename) {
bool IniFile::Load(std::istream &in) {
// Maximum number of letters in a line
static const int MAX_BYTES = 1024*32;
+ char *templine = new char[MAX_BYTES]; // avoid using up massive stack space
while (!(in.eof() || in.fail()))
{
- char templine[MAX_BYTES];
in.getline(templine, MAX_BYTES);
std::string line = templine;
@@ -624,6 +624,7 @@ bool IniFile::Load(std::istream &in) {
}
}
+ delete[] templine;
return true;
}
diff --git a/Common/Data/Format/IniFile.h b/Common/Data/Format/IniFile.h
index 707f0ab3eb..1008fff5df 100644
--- a/Common/Data/Format/IniFile.h
+++ b/Common/Data/Format/IniFile.h
@@ -11,6 +11,8 @@
#include "Common/File/Path.h"
+class VFSInterface;
+
class Section {
friend class IniFile;
@@ -86,7 +88,7 @@ public:
bool Load(const Path &path);
bool Load(const std::string &filename) { return Load(Path(filename)); }
bool Load(std::istream &istream);
- bool LoadFromVFS(const std::string &filename);
+ bool LoadFromVFS(VFSInterface &vfs, const std::string &filename);
bool Save(const Path &path);
bool Save(const std::string &filename) { return Save(Path(filename)); }
diff --git a/Common/Data/Format/JSONReader.cpp b/Common/Data/Format/JSONReader.cpp
index 9c2128e6ef..b65381d986 100644
--- a/Common/Data/Format/JSONReader.cpp
+++ b/Common/Data/Format/JSONReader.cpp
@@ -8,7 +8,7 @@ namespace json {
JsonReader::JsonReader(const std::string &filename) {
size_t buf_size;
- buffer_ = (char *)VFSReadFile(filename.c_str(), &buf_size);
+ buffer_ = (char *)g_VFS.ReadFile(filename.c_str(), &buf_size);
if (buffer_) {
parse();
} else {
diff --git a/Common/Data/Format/PNGLoad.cpp b/Common/Data/Format/PNGLoad.cpp
index 0174b156fe..b2ed18f99b 100644
--- a/Common/Data/Format/PNGLoad.cpp
+++ b/Common/Data/Format/PNGLoad.cpp
@@ -59,3 +59,14 @@ int pngLoadPtr(const unsigned char *input_ptr, size_t input_len, int *pwidth, in
png_image_finish_read(&png, NULL, *image_data_ptr, stride, NULL);
return 1;
}
+
+bool PNGHeaderPeek::IsValidPNGHeader() const {
+ if (magic != 0x474e5089 || ihdrTag != 0x52444849) {
+ return false;
+ }
+ // Reject crazy sized images, too.
+ if (Width() > 32768 && Height() > 32768) {
+ return false;
+ }
+ return true;
+}
diff --git a/Common/Data/Format/PNGLoad.h b/Common/Data/Format/PNGLoad.h
index 48eaf6a1db..76ab04ee9e 100644
--- a/Common/Data/Format/PNGLoad.h
+++ b/Common/Data/Format/PNGLoad.h
@@ -1,5 +1,8 @@
-#ifndef _PNG_LOAD_H
-#define _PNG_LOAD_H
+#pragma once
+
+#include
+
+#include "Common/BitSet.h"
// *image_data_ptr should be deleted with free()
// return value of 1 == success.
@@ -9,4 +12,22 @@ int pngLoad(const char *file, int *pwidth,
int pngLoadPtr(const unsigned char *input_ptr, size_t input_len, int *pwidth,
int *pheight, unsigned char **image_data_ptr);
-#endif // _PNG_LOAD_H
+// PNG peeker - just read the start of a PNG straight into this struct, in order to
+// look at basic parameters like width and height. Note that while PNG is a chunk-based
+// format, the IHDR chunk is REQUIRED to be the first one, so this will work.
+// Does not handle Apple's weirdo extension CgBI. http://iphonedevwiki.net/index.php/CgBI_file_format
+// That should not be an issue.
+struct PNGHeaderPeek {
+ uint32_t magic;
+ uint32_t ignore0;
+ uint32_t ignore1;
+ uint32_t ihdrTag;
+ uint32_t be_width; // big endian
+ uint32_t be_height;
+ uint8_t bitDepth; // bits per channel, can be 1, 2, 4, 8, 16
+ uint8_t colorType; // really, pixel format. 0 = grayscale, 2 = rgb, 3 = palette index, 4 = gray+alpha, 6 = rgba
+
+ bool IsValidPNGHeader() const;
+ int Width() const { return swap32(be_width); }
+ int Height() const { return swap32(be_height); }
+};
diff --git a/Common/Data/Format/ZIMLoad.cpp b/Common/Data/Format/ZIMLoad.cpp
index 3fd759e92c..9a145afbe1 100644
--- a/Common/Data/Format/ZIMLoad.cpp
+++ b/Common/Data/Format/ZIMLoad.cpp
@@ -126,7 +126,7 @@ int LoadZIMPtr(const uint8_t *zim, size_t datasize, int *width, int *height, int
int LoadZIM(const char *filename, int *width, int *height, int *format, uint8_t **image) {
size_t size;
- uint8_t *buffer = VFSReadFile(filename, &size);
+ uint8_t *buffer = g_VFS.ReadFile(filename, &size);
if (!buffer) {
ERROR_LOG(IO, "Couldn't read data for '%s'", filename);
return 0;
diff --git a/Common/Data/Text/I18n.cpp b/Common/Data/Text/I18n.cpp
index 9c222da8b8..ab5509dcbd 100644
--- a/Common/Data/Text/I18n.cpp
+++ b/Common/Data/Text/I18n.cpp
@@ -73,7 +73,7 @@ Path I18NRepo::GetIniPath(const std::string &languageID) const {
bool I18NRepo::IniExists(const std::string &languageID) const {
File::FileInfo info;
- if (!VFSGetFileInfo(GetIniPath(languageID).ToString().c_str(), &info))
+ if (!g_VFS.GetFileInfo(GetIniPath(languageID).ToString().c_str(), &info))
return false;
if (!info.exists)
return false;
@@ -91,7 +91,7 @@ bool I18NRepo::LoadIni(const std::string &languageID, const Path &overridePath)
iniPath = GetIniPath(languageID);
}
- if (!ini.LoadFromVFS(iniPath.ToString()))
+ if (!ini.LoadFromVFS(g_VFS, iniPath.ToString()))
return false;
Clear();
@@ -127,7 +127,7 @@ I18NCategory *I18NRepo::LoadSection(const Section *section, const char *name) {
return cat;
}
-// This is a very light touched save variant - it won't overwrite
+// This is a very light touched save variant - it won't overwrite
// anything, only create new entries.
void I18NRepo::SaveIni(const std::string &languageID) {
IniFile ini;
diff --git a/Common/File/AndroidStorage.cpp b/Common/File/AndroidStorage.cpp
index ec8248947b..c4ddcc73c1 100644
--- a/Common/File/AndroidStorage.cpp
+++ b/Common/File/AndroidStorage.cpp
@@ -1,3 +1,5 @@
+#include
+
#include "Common/File/AndroidStorage.h"
#include "Common/StringUtils.h"
#include "Common/Log.h"
diff --git a/Common/File/DirListing.h b/Common/File/DirListing.h
index c53d420588..4ad9ce0c3e 100644
--- a/Common/File/DirListing.h
+++ b/Common/File/DirListing.h
@@ -2,10 +2,8 @@
#include
#include
-
#include
-
-#include
+#include
#include "Common/File/Path.h"
diff --git a/Common/File/Path.cpp b/Common/File/Path.cpp
index 489ab692b4..71a795d683 100644
--- a/Common/File/Path.cpp
+++ b/Common/File/Path.cpp
@@ -255,12 +255,34 @@ std::wstring Path::ToWString() const {
}
#endif
-std::string Path::ToVisualString() const {
+std::string Path::ToVisualString(const char *relativeRoot) const {
if (type_ == PathType::CONTENT_URI) {
return AndroidContentURI(path_).ToVisualString();
#if PPSSPP_PLATFORM(WINDOWS)
} else if (type_ == PathType::NATIVE) {
- return ReplaceAll(path_, "/", "\\");
+ // It can be useful to show the path as relative to the memstick
+ if (relativeRoot) {
+ std::string root = ReplaceAll(relativeRoot, "/", "\\");
+ std::string path = ReplaceAll(path_, "/", "\\");
+ if (startsWithNoCase(path, root)) {
+ return path.substr(root.size());
+ } else {
+ return path;
+ }
+ } else {
+ return ReplaceAll(path_, "/", "\\");
+ }
+#else
+ if (relativeRoot) {
+ std::string root = relativeRoot;
+ if (startsWithNoCase(path_, root)) {
+ return path_.substr(root.size());
+ } else {
+ return path_;
+ }
+ } else {
+ return path_;
+ }
#endif
} else {
return path_;
diff --git a/Common/File/Path.h b/Common/File/Path.h
index aded7b5bde..ac93f9a9b8 100644
--- a/Common/File/Path.h
+++ b/Common/File/Path.h
@@ -81,9 +81,9 @@ public:
Path WithReplacedExtension(const std::string &oldExtension, const std::string &newExtension) const;
Path WithReplacedExtension(const std::string &newExtension) const;
- // Removes the last component.
std::string GetFilename() const; // Really, GetLastComponent. Could be a file or directory. Includes the extension.
std::string GetFileExtension() const; // Always lowercase return. Includes the dot.
+ // Removes the last component.
std::string GetDirectory() const;
const std::string &ToString() const;
@@ -92,7 +92,8 @@ public:
std::wstring ToWString() const;
#endif
- std::string ToVisualString() const;
+ // Pass in a relative root to turn the path into a relative path - if it is one!
+ std::string ToVisualString(const char *relativeRoot = nullptr) const;
bool CanNavigateUp() const;
Path NavigateUp() const;
diff --git a/Common/File/VFS/AssetReader.cpp b/Common/File/VFS/AssetReader.cpp
deleted file mode 100644
index 306cd0c5e0..0000000000
--- a/Common/File/VFS/AssetReader.cpp
+++ /dev/null
@@ -1,204 +0,0 @@
-#include
-#include
-#include
-#include
-
-#ifdef __ANDROID__
-#include
-#endif
-
-#include "Common/Common.h"
-#include "Common/Log.h"
-#include "Common/File/VFS/AssetReader.h"
-
-#ifdef __ANDROID__
-uint8_t *ReadFromZip(zip *archive, const char* filename, size_t *size) {
- // Figure out the file size first.
- struct zip_stat zstat;
- zip_file *file = zip_fopen(archive, filename, ZIP_FL_NOCASE|ZIP_FL_UNCHANGED);
- if (!file) {
- ERROR_LOG(IO, "Error opening %s from ZIP", filename);
- return 0;
- }
- zip_stat(archive, filename, ZIP_FL_NOCASE|ZIP_FL_UNCHANGED, &zstat);
-
- uint8_t *contents = new uint8_t[zstat.size + 1];
- zip_fread(file, contents, zstat.size);
- zip_fclose(file);
- contents[zstat.size] = 0;
-
- *size = zstat.size;
- return contents;
-}
-
-#endif
-
-#ifdef __ANDROID__
-
-ZipAssetReader::ZipAssetReader(const char *zip_file, const char *in_zip_path) {
- zip_file_ = zip_open(zip_file, 0, NULL);
- strcpy(in_zip_path_, in_zip_path);
- if (!zip_file_) {
- ERROR_LOG(IO, "Failed to open %s as a zip file", zip_file);
- }
-
- std::vector info;
- GetFileListing("assets", &info, 0);
- for (size_t i = 0; i < info.size(); i++) {
- if (info[i].isDirectory) {
- DEBUG_LOG(IO, "Directory: %s", info[i].name.c_str());
- } else {
- DEBUG_LOG(IO, "File: %s", info[i].name.c_str());
- }
- }
-}
-
-ZipAssetReader::~ZipAssetReader() {
- std::lock_guard guard(lock_);
- zip_close(zip_file_);
-}
-
-uint8_t *ZipAssetReader::ReadAsset(const char *path, size_t *size) {
- char temp_path[1024];
- strcpy(temp_path, in_zip_path_);
- strcat(temp_path, path);
-
- std::lock_guard guard(lock_);
- return ReadFromZip(zip_file_, temp_path, size);
-}
-
-bool ZipAssetReader::GetFileListing(const char *orig_path, std::vector *listing, const char *filter = 0) {
- char path[1024];
- strcpy(path, in_zip_path_);
- strcat(path, orig_path);
-
- std::set filters;
- std::string tmp;
- if (filter) {
- while (*filter) {
- if (*filter == ':') {
- filters.insert("." + tmp);
- tmp.clear();
- } else {
- tmp.push_back(*filter);
- }
- filter++;
- }
- }
- if (tmp.size())
- filters.insert("." + tmp);
-
- // We just loop through the whole ZIP file and deduce what files are in this directory, and what subdirectories there are.
- std::set files;
- std::set directories;
- GetZipListings(path, files, directories);
-
- for (auto diter = directories.begin(); diter != directories.end(); ++diter) {
- File::FileInfo info;
- info.name = *diter;
-
- // Remove the "inzip" part of the fullname.
- info.fullName = Path(std::string(path).substr(strlen(in_zip_path_))) / *diter;
- info.exists = true;
- info.isWritable = false;
- info.isDirectory = true;
- listing->push_back(info);
- }
-
- for (auto fiter = files.begin(); fiter != files.end(); ++fiter) {
- std::string fpath = path;
- File::FileInfo info;
- info.name = *fiter;
- info.fullName = Path(std::string(path).substr(strlen(in_zip_path_))) / *fiter;
- info.exists = true;
- info.isWritable = false;
- info.isDirectory = false;
- std::string ext = info.fullName.GetFileExtension();
- if (filter) {
- if (filters.find(ext) == filters.end()) {
- continue;
- }
- }
- listing->push_back(info);
- }
-
- std::sort(listing->begin(), listing->end());
- return true;
-}
-
-void ZipAssetReader::GetZipListings(const char *path, std::set &files, std::set &directories) {
- size_t pathlen = strlen(path);
- if (path[pathlen - 1] == '/')
- pathlen--;
-
- std::lock_guard guard(lock_);
- int numFiles = zip_get_num_files(zip_file_);
- for (int i = 0; i < numFiles; i++) {
- const char* name = zip_get_name(zip_file_, i, 0);
- if (!name)
- continue;
- if (!memcmp(name, path, pathlen)) {
- // The prefix is right. Let's see if this is a file or path.
- const char *slashPos = strchr(name + pathlen + 1, '/');
- if (slashPos != 0) {
- // A directory.
- std::string dirName = std::string(name + pathlen + 1, slashPos - (name + pathlen + 1));
- directories.insert(dirName);
- } else if (name[pathlen] == '/') {
- const char *fn = name + pathlen + 1;
- files.insert(std::string(fn));
- } // else, it was a file with the same prefix as the path. like langregion.ini next to lang/.
- }
- }
-}
-
-bool ZipAssetReader::GetFileInfo(const char *path, File::FileInfo *info) {
- struct zip_stat zstat;
- char temp_path[1024];
- strcpy(temp_path, in_zip_path_);
- strcat(temp_path, path);
- if (0 != zip_stat(zip_file_, temp_path, ZIP_FL_NOCASE|ZIP_FL_UNCHANGED, &zstat)) {
- // ZIP files do not have real directories, so we'll end up here if we
- // try to stat one. For now that's fine.
- info->exists = false;
- info->size = 0;
- return false;
- }
-
- info->fullName = Path(path);
- info->exists = true; // TODO
- info->isWritable = false;
- info->isDirectory = false; // TODO
- info->size = zstat.size;
- return true;
-}
-
-#endif
-
-DirectoryAssetReader::DirectoryAssetReader(const Path &path) {
- path_ = path;
-}
-
-uint8_t *DirectoryAssetReader::ReadAsset(const char *path, size_t *size) {
- Path new_path = Path(path).StartsWith(path_) ? Path(path) : path_ / path;
- return File::ReadLocalFile(new_path, size);
-}
-
-bool DirectoryAssetReader::GetFileListing(const char *path, std::vector *listing, const char *filter = nullptr) {
- Path new_path = Path(path).StartsWith(path_) ? Path(path) : path_ / path;
-
- File::FileInfo info;
- if (!File::GetFileInfo(new_path, &info))
- return false;
-
- if (info.isDirectory) {
- File::GetFilesInDir(new_path, listing, filter);
- return true;
- }
- return false;
-}
-
-bool DirectoryAssetReader::GetFileInfo(const char *path, File::FileInfo *info) {
- Path new_path = Path(path).StartsWith(path_) ? Path(path) : path_ / path;
- return File::GetFileInfo(new_path, info);
-}
diff --git a/Common/File/VFS/AssetReader.h b/Common/File/VFS/AssetReader.h
deleted file mode 100644
index 8352f8eea8..0000000000
--- a/Common/File/VFS/AssetReader.h
+++ /dev/null
@@ -1,65 +0,0 @@
-// TODO: Move much of this code to vfs.cpp
-#pragma once
-
-#ifdef __ANDROID__
-#include
-#endif
-
-#include
-#include
-#include
-#include
-
-#include "Common/File/VFS/VFS.h"
-#include "Common/File/FileUtil.h"
-#include "Common/File/Path.h"
-
-class AssetReader {
-public:
- virtual ~AssetReader() {}
- // use delete[]
- virtual uint8_t *ReadAsset(const char *path, size_t *size) = 0;
- // Filter support is optional but nice to have
- virtual bool GetFileListing(const char *path, std::vector *listing, const char *filter = 0) = 0;
- virtual bool GetFileInfo(const char *path, File::FileInfo *info) = 0;
- virtual std::string toString() const = 0;
-};
-
-#ifdef __ANDROID__
-uint8_t *ReadFromZip(zip *archive, const char* filename, size_t *size);
-class ZipAssetReader : public AssetReader {
-public:
- ZipAssetReader(const char *zip_file, const char *in_zip_path);
- ~ZipAssetReader();
- // use delete[]
- uint8_t *ReadAsset(const char *path, size_t *size) override;
- bool GetFileListing(const char *path, std::vector *listing, const char *filter) override;
- bool GetFileInfo(const char *path, File::FileInfo *info) override;
- std::string toString() const override {
- return in_zip_path_;
- }
-
-private:
- void GetZipListings(const char *path, std::set &files, std::set &directories);
-
- zip *zip_file_;
- std::mutex lock_;
- char in_zip_path_[256];
-};
-#endif
-
-class DirectoryAssetReader : public AssetReader {
-public:
- explicit DirectoryAssetReader(const Path &path);
- // use delete[]
- uint8_t *ReadAsset(const char *path, size_t *size) override;
- bool GetFileListing(const char *path, std::vector *listing, const char *filter) override;
- bool GetFileInfo(const char *path, File::FileInfo *info) override;
- std::string toString() const override {
- return path_.ToString();
- }
-
-private:
- Path path_;
-};
-
diff --git a/Common/File/VFS/DirectoryReader.cpp b/Common/File/VFS/DirectoryReader.cpp
new file mode 100644
index 0000000000..981dff905b
--- /dev/null
+++ b/Common/File/VFS/DirectoryReader.cpp
@@ -0,0 +1,99 @@
+#include
+
+#include "Common/Common.h"
+#include "Common/Log.h"
+#include "Common/File/VFS/DirectoryReader.h"
+
+DirectoryReader::DirectoryReader(const Path &path) {
+ path_ = path;
+}
+
+uint8_t *DirectoryReader::ReadFile(const char *path, size_t *size) {
+ Path new_path = Path(path).StartsWith(path_) ? Path(path) : path_ / path;
+ return File::ReadLocalFile(new_path, size);
+}
+
+bool DirectoryReader::GetFileListing(const char *path, std::vector *listing, const char *filter = nullptr) {
+ Path new_path = Path(path).StartsWith(path_) ? Path(path) : path_ / path;
+
+ File::FileInfo info;
+ if (!File::GetFileInfo(new_path, &info))
+ return false;
+
+ if (info.isDirectory) {
+ File::GetFilesInDir(new_path, listing, filter);
+ return true;
+ }
+ return false;
+}
+
+bool DirectoryReader::GetFileInfo(const char *path, File::FileInfo *info) {
+ Path new_path = Path(path).StartsWith(path_) ? Path(path) : path_ / path;
+ return File::GetFileInfo(new_path, info);
+}
+
+class DirectoryReaderFileReference : public VFSFileReference {
+public:
+ Path path;
+};
+
+class DirectoryReaderOpenFile : public VFSOpenFile {
+public:
+ ~DirectoryReaderOpenFile() {
+ _dbg_assert_(file == nullptr);
+ }
+ FILE *file = nullptr;
+};
+
+VFSFileReference *DirectoryReader::GetFile(const char *path) {
+ Path filePath = path_ / path;
+ if (!File::Exists(filePath)) {
+ return nullptr;
+ }
+
+ DirectoryReaderFileReference *reference = new DirectoryReaderFileReference();
+ reference->path = filePath;
+ return reference;
+}
+
+bool DirectoryReader::GetFileInfo(VFSFileReference *vfsReference, File::FileInfo *fileInfo) {
+ DirectoryReaderFileReference *reference = (DirectoryReaderFileReference *)vfsReference;
+ return File::GetFileInfo(reference->path, fileInfo);
+}
+
+void DirectoryReader::ReleaseFile(VFSFileReference *vfsReference) {
+ DirectoryReaderFileReference *reference = (DirectoryReaderFileReference *)vfsReference;
+ delete reference;
+}
+
+VFSOpenFile *DirectoryReader::OpenFileForRead(VFSFileReference *vfsReference, size_t *size) {
+ DirectoryReaderFileReference *reference = (DirectoryReaderFileReference *)vfsReference;
+ FILE *file = File::OpenCFile(reference->path, "rb");
+ if (!file) {
+ return nullptr;
+ }
+ fseek(file, 0, SEEK_END);
+ *size = ftell(file);
+ fseek(file, 0, SEEK_SET);
+ DirectoryReaderOpenFile *openFile = new DirectoryReaderOpenFile();
+ openFile->file = file;
+ return openFile;
+}
+
+void DirectoryReader::Rewind(VFSOpenFile *vfsOpenFile) {
+ DirectoryReaderOpenFile *openFile = (DirectoryReaderOpenFile *)vfsOpenFile;
+ fseek(openFile->file, 0, SEEK_SET);
+}
+
+size_t DirectoryReader::Read(VFSOpenFile *vfsOpenFile, void *buffer, size_t length) {
+ DirectoryReaderOpenFile *openFile = (DirectoryReaderOpenFile *)vfsOpenFile;
+ return fread(buffer, 1, length, openFile->file);
+}
+
+void DirectoryReader::CloseFile(VFSOpenFile *vfsOpenFile) {
+ DirectoryReaderOpenFile *openFile = (DirectoryReaderOpenFile *)vfsOpenFile;
+ _dbg_assert_(openFile->file != nullptr);
+ fclose(openFile->file);
+ openFile->file = nullptr;
+ delete openFile;
+}
diff --git a/Common/File/VFS/DirectoryReader.h b/Common/File/VFS/DirectoryReader.h
new file mode 100644
index 0000000000..f742c6b3da
--- /dev/null
+++ b/Common/File/VFS/DirectoryReader.h
@@ -0,0 +1,30 @@
+#pragma once
+
+#include "Common/File/VFS/VFS.h"
+#include "Common/File/FileUtil.h"
+#include "Common/File/Path.h"
+
+class DirectoryReader : public VFSBackend {
+public:
+ explicit DirectoryReader(const Path &path);
+ // use delete[] on the returned value.
+ uint8_t *ReadFile(const char *path, size_t *size) override;
+
+ VFSFileReference *GetFile(const char *path) override;
+ bool GetFileInfo(VFSFileReference *vfsReference, File::FileInfo *fileInfo) override;
+ void ReleaseFile(VFSFileReference *vfsReference) override;
+
+ VFSOpenFile *OpenFileForRead(VFSFileReference *vfsReference, size_t *size) override;
+ void Rewind(VFSOpenFile *vfsOpenFile) override;
+ size_t Read(VFSOpenFile *vfsOpenFile, void *buffer, size_t length) override;
+ void CloseFile(VFSOpenFile *vfsOpenFile) override;
+
+ bool GetFileListing(const char *path, std::vector *listing, const char *filter) override;
+ bool GetFileInfo(const char *path, File::FileInfo *info) override;
+ std::string toString() const override {
+ return path_.ToString();
+ }
+
+private:
+ Path path_;
+};
diff --git a/Common/File/VFS/VFS.cpp b/Common/File/VFS/VFS.cpp
index cf347dcf44..260a6b007a 100644
--- a/Common/File/VFS/VFS.cpp
+++ b/Common/File/VFS/VFS.cpp
@@ -1,28 +1,26 @@
+#include
+
#include "Common/Log.h"
#include "Common/File/VFS/VFS.h"
-#include "Common/File/VFS/AssetReader.h"
+#include "Common/File/FileUtil.h"
#include "Common/File/AndroidStorage.h"
-struct VFSEntry {
- const char *prefix;
- AssetReader *reader;
-};
+VFS g_VFS;
-static VFSEntry entries[16];
-static int num_entries = 0;
-
-void VFSRegister(const char *prefix, AssetReader *reader) {
- entries[num_entries].prefix = prefix;
- entries[num_entries].reader = reader;
- DEBUG_LOG(IO, "Registered VFS for prefix %s: %s", prefix, reader->toString().c_str());
- num_entries++;
+void VFS::Register(const char *prefix, VFSBackend *reader) {
+ if (reader) {
+ entries_.push_back(VFSEntry{ prefix, reader });
+ DEBUG_LOG(IO, "Registered VFS for prefix %s: %s", prefix, reader->toString().c_str());
+ } else {
+ ERROR_LOG(IO, "Trying to register null VFS backend for prefix %s", prefix);
+ }
}
-void VFSShutdown() {
- for (int i = 0; i < num_entries; i++) {
- delete entries[i].reader;
+void VFS::Clear() {
+ for (auto &entry : entries_) {
+ delete entry.reader;
}
- num_entries = 0;
+ entries_.clear();
}
// TODO: Use Path more.
@@ -38,7 +36,7 @@ static bool IsLocalAbsolutePath(const char *path) {
}
// The returned data should be free'd with delete[].
-uint8_t *VFSReadFile(const char *filename, size_t *size) {
+uint8_t *VFS::ReadFile(const char *filename, size_t *size) {
if (IsLocalAbsolutePath(filename)) {
// Local path, not VFS.
// INFO_LOG(IO, "Not a VFS path: %s . Reading local file.", filename);
@@ -47,13 +45,13 @@ uint8_t *VFSReadFile(const char *filename, size_t *size) {
int fn_len = (int)strlen(filename);
bool fileSystemFound = false;
- for (int i = 0; i < num_entries; i++) {
- int prefix_len = (int)strlen(entries[i].prefix);
+ for (const auto &entry : entries_) {
+ int prefix_len = (int)strlen(entry.prefix);
if (prefix_len >= fn_len) continue;
- if (0 == memcmp(filename, entries[i].prefix, prefix_len)) {
+ if (0 == memcmp(filename, entry.prefix, prefix_len)) {
fileSystemFound = true;
// INFO_LOG(IO, "Prefix match: %s (%s) -> %s", entries[i].prefix, filename, filename + prefix_len);
- uint8_t *data = entries[i].reader->ReadAsset(filename + prefix_len, size);
+ uint8_t *data = entry.reader->ReadFile(filename + prefix_len, size);
if (data)
return data;
else
@@ -67,7 +65,7 @@ uint8_t *VFSReadFile(const char *filename, size_t *size) {
return 0;
}
-bool VFSGetFileListing(const char *path, std::vector *listing, const char *filter) {
+bool VFS::GetFileListing(const char *path, std::vector *listing, const char *filter) {
if (IsLocalAbsolutePath(path)) {
// Local path, not VFS.
// INFO_LOG(IO, "Not a VFS path: %s . Reading local directory.", path);
@@ -77,12 +75,12 @@ bool VFSGetFileListing(const char *path, std::vector *listing, c
int fn_len = (int)strlen(path);
bool fileSystemFound = false;
- for (int i = 0; i < num_entries; i++) {
- int prefix_len = (int)strlen(entries[i].prefix);
+ for (const auto &entry : entries_) {
+ int prefix_len = (int)strlen(entry.prefix);
if (prefix_len >= fn_len) continue;
- if (0 == memcmp(path, entries[i].prefix, prefix_len)) {
+ if (0 == memcmp(path, entry.prefix, prefix_len)) {
fileSystemFound = true;
- if (entries[i].reader->GetFileListing(path + prefix_len, listing, filter)) {
+ if (entry.reader->GetFileListing(path + prefix_len, listing, filter)) {
return true;
}
}
@@ -94,7 +92,7 @@ bool VFSGetFileListing(const char *path, std::vector *listing, c
return false;
}
-bool VFSGetFileInfo(const char *path, File::FileInfo *info) {
+bool VFS::GetFileInfo(const char *path, File::FileInfo *info) {
if (IsLocalAbsolutePath(path)) {
// Local path, not VFS.
// INFO_LOG(IO, "Not a VFS path: %s . Getting local file info.", path);
@@ -103,19 +101,19 @@ bool VFSGetFileInfo(const char *path, File::FileInfo *info) {
bool fileSystemFound = false;
int fn_len = (int)strlen(path);
- for (int i = 0; i < num_entries; i++) {
- int prefix_len = (int)strlen(entries[i].prefix);
+ for (const auto &entry : entries_) {
+ int prefix_len = (int)strlen(entry.prefix);
if (prefix_len >= fn_len) continue;
- if (0 == memcmp(path, entries[i].prefix, prefix_len)) {
+ if (0 == memcmp(path, entry.prefix, prefix_len)) {
fileSystemFound = true;
- if (entries[i].reader->GetFileInfo(path + prefix_len, info))
+ if (entry.reader->GetFileInfo(path + prefix_len, info))
return true;
else
continue;
}
}
if (!fileSystemFound) {
- ERROR_LOG(IO, "Missing filesystem for %s", path);
+ ERROR_LOG(IO, "Missing filesystem for '%s'", path);
} // Otherwise, the file was just missing. No need to log.
return false;
}
diff --git a/Common/File/VFS/VFS.h b/Common/File/VFS/VFS.h
index 0f7e03c3fa..5d585e309e 100644
--- a/Common/File/VFS/VFS.h
+++ b/Common/File/VFS/VFS.h
@@ -1,21 +1,82 @@
#pragma once
#include
+#include
#include "Common/File/DirListing.h"
-// Basic virtual file system. Used to manage assets on Android, where we have to
+// Basic read-only virtual file system. Used to manage assets on Android, where we have to
// read them manually out of the APK zipfile, while being able to run on other
// platforms as well with the appropriate directory set-up.
-class AssetReader;
+// Note that this is kinda similar in concept to Core/MetaFileSystem.h, but that one
+// is specifically for operations done by the emulated PSP, while this is for operations
+// on the system level, like loading assets, and maybe texture packs. Also, as mentioned,
+// this one is read-only, so a bit smaller and simpler.
-void VFSRegister(const char *prefix, AssetReader *reader);
-void VFSShutdown();
+// VFSBackend instances can be used on their own, without the VFS, to serve as an abstraction of
+// a single directory or ZIP file.
-// Use delete [] to release the returned memory.
-// Always allocates an extra zero byte at the end, so that it
-// can be used for text like shader sources.
-uint8_t *VFSReadFile(const char *filename, size_t *size);
-bool VFSGetFileListing(const char *path, std::vector *listing, const char *filter = 0);
-bool VFSGetFileInfo(const char *filename, File::FileInfo *fileInfo);
+// The VFSFileReference level of abstraction is there to hold things like zip file indices,
+// for fast re-open etc.
+
+class VFSFileReference {
+public:
+ virtual ~VFSFileReference() {}
+};
+
+class VFSOpenFile {
+public:
+ virtual ~VFSOpenFile() {}
+};
+
+// Common interface parts between VFSBackend and VFS.
+// Sometimes you don't need the VFS multiplexing and only have a VFSBackend *, sometimes you do need it,
+// and it would be cool to be able to use the same interface, like when loading INI files.
+class VFSInterface {
+public:
+ virtual ~VFSInterface() {}
+ virtual uint8_t *ReadFile(const char *path, size_t *size) = 0;
+ virtual bool GetFileListing(const char *path, std::vector *listing, const char *filter = nullptr) = 0;
+};
+
+class VFSBackend : public VFSInterface {
+public:
+ virtual VFSFileReference *GetFile(const char *path) = 0;
+ virtual bool GetFileInfo(VFSFileReference *vfsReference, File::FileInfo *fileInfo) = 0;
+ virtual void ReleaseFile(VFSFileReference *vfsReference) = 0;
+
+ // Must write the size of the file to *size. Both backends can do this efficiently here,
+ // avoiding a call to GetFileInfo.
+ virtual VFSOpenFile *OpenFileForRead(VFSFileReference *vfsReference, size_t *size) = 0;
+ virtual void Rewind(VFSOpenFile *vfsOpenFile) = 0;
+ virtual size_t Read(VFSOpenFile *vfsOpenFile, void *buffer, size_t length) = 0;
+ virtual void CloseFile(VFSOpenFile *vfsOpenFile) = 0;
+
+ // Filter support is optional but nice to have
+ virtual bool GetFileInfo(const char *path, File::FileInfo *info) = 0;
+ virtual std::string toString() const = 0;
+};
+
+class VFS : public VFSInterface {
+public:
+ ~VFS() { Clear(); }
+ void Register(const char *prefix, VFSBackend *reader);
+ void Clear();
+
+ // Use delete [] to release the returned memory.
+ // Always allocates an extra zero byte at the end, so that it
+ // can be used for text like shader sources.
+ uint8_t *ReadFile(const char *filename, size_t *size) override;
+ bool GetFileInfo(const char *filename, File::FileInfo *fileInfo);
+ bool GetFileListing(const char *path, std::vector *listing, const char *filter = nullptr) override;
+
+private:
+ struct VFSEntry {
+ const char *prefix;
+ VFSBackend *reader;
+ };
+ std::vector entries_;
+};
+
+extern VFS g_VFS;
diff --git a/Common/File/VFS/ZipFileReader.cpp b/Common/File/VFS/ZipFileReader.cpp
new file mode 100644
index 0000000000..8935fafe55
--- /dev/null
+++ b/Common/File/VFS/ZipFileReader.cpp
@@ -0,0 +1,285 @@
+#include
+#include
+#include
+#include
+#include
+
+#ifdef SHARED_LIBZIP
+#include
+#else
+#include "ext/libzip/zip.h"
+#endif
+
+#include "Common/Common.h"
+#include "Common/Log.h"
+#include "Common/File/VFS/ZipFileReader.h"
+#include "Common/StringUtils.h"
+
+ZipFileReader *ZipFileReader::Create(const Path &zipFile, const char *inZipPath, bool logErrors) {
+ int error = 0;
+ zip *zip_file;
+ if (zipFile.Type() == PathType::CONTENT_URI) {
+ int fd = File::OpenFD(zipFile, File::OPEN_READ);
+ if (!fd) {
+ if (logErrors) {
+ ERROR_LOG(IO, "Failed to open FD for '%s' as zip file", zipFile.c_str());
+ }
+ return nullptr;
+ }
+ zip_file = zip_fdopen(fd, 0, &error);
+ } else {
+ zip_file = zip_open(zipFile.c_str(), 0, &error);
+ }
+
+ if (!zip_file) {
+ if (logErrors) {
+ ERROR_LOG(IO, "Failed to open %s as a zip file", zipFile.c_str());
+ }
+ return nullptr;
+ }
+
+ ZipFileReader *reader = new ZipFileReader();
+ reader->zip_file_ = zip_file;
+ truncate_cpy(reader->inZipPath_, inZipPath);
+ return reader;
+}
+
+ZipFileReader::~ZipFileReader() {
+ std::lock_guard guard(lock_);
+ zip_close(zip_file_);
+}
+
+uint8_t *ZipFileReader::ReadFile(const char *path, size_t *size) {
+ char temp_path[2048];
+ snprintf(temp_path, sizeof(temp_path), "%s%s", inZipPath_, path);
+
+ std::lock_guard guard(lock_);
+ // Figure out the file size first.
+ struct zip_stat zstat;
+ zip_stat(zip_file_, temp_path, ZIP_FL_NOCASE | ZIP_FL_UNCHANGED, &zstat);
+ zip_file *file = zip_fopen(zip_file_, temp_path, ZIP_FL_NOCASE | ZIP_FL_UNCHANGED);
+ if (!file) {
+ ERROR_LOG(IO, "Error opening %s from ZIP", temp_path);
+ return 0;
+ }
+ uint8_t *contents = new uint8_t[zstat.size + 1];
+ zip_fread(file, contents, zstat.size);
+ zip_fclose(file);
+ contents[zstat.size] = 0;
+
+ *size = zstat.size;
+ return contents;
+}
+
+bool ZipFileReader::GetFileListing(const char *orig_path, std::vector *listing, const char *filter = 0) {
+ char path[2048];
+ snprintf(path, sizeof(path), "%s%s", inZipPath_, orig_path);
+
+ std::set filters;
+ std::string tmp;
+ if (filter) {
+ while (*filter) {
+ if (*filter == ':') {
+ filters.insert("." + tmp);
+ tmp.clear();
+ } else {
+ tmp.push_back(*filter);
+ }
+ filter++;
+ }
+ }
+
+ if (tmp.size())
+ filters.insert("." + tmp);
+
+ // We just loop through the whole ZIP file and deduce what files are in this directory, and what subdirectories there are.
+ std::set files;
+ std::set directories;
+ GetZipListings(path, files, directories);
+
+ for (auto diter = directories.begin(); diter != directories.end(); ++diter) {
+ File::FileInfo info;
+ info.name = *diter;
+
+ // Remove the "inzip" part of the fullname.
+ info.fullName = Path(std::string(path).substr(strlen(inZipPath_))) / *diter;
+ info.exists = true;
+ info.isWritable = false;
+ info.isDirectory = true;
+ listing->push_back(info);
+ }
+
+ for (auto fiter = files.begin(); fiter != files.end(); ++fiter) {
+ std::string fpath = path;
+ File::FileInfo info;
+ info.name = *fiter;
+ info.fullName = Path(std::string(path).substr(strlen(inZipPath_))) / *fiter;
+ info.exists = true;
+ info.isWritable = false;
+ info.isDirectory = false;
+ std::string ext = info.fullName.GetFileExtension();
+ if (filter) {
+ if (filters.find(ext) == filters.end()) {
+ continue;
+ }
+ }
+ listing->push_back(info);
+ }
+
+ std::sort(listing->begin(), listing->end());
+ return true;
+}
+
+void ZipFileReader::GetZipListings(const char *path, std::set &files, std::set &directories) {
+ size_t pathlen = strlen(path);
+ if (path[pathlen - 1] == '/')
+ pathlen--;
+
+ std::lock_guard guard(lock_);
+ int numFiles = zip_get_num_files(zip_file_);
+ for (int i = 0; i < numFiles; i++) {
+ const char* name = zip_get_name(zip_file_, i, 0);
+ if (!name)
+ continue;
+ if (!memcmp(name, path, pathlen)) {
+ // The prefix is right. Let's see if this is a file or path.
+ const char *slashPos = strchr(name + pathlen + 1, '/');
+ if (slashPos != 0) {
+ // A directory.
+ std::string dirName = std::string(name + pathlen + 1, slashPos - (name + pathlen + 1));
+ directories.insert(dirName);
+ } else if (name[pathlen] == '/') {
+ const char *fn = name + pathlen + 1;
+ files.insert(std::string(fn));
+ } // else, it was a file with the same prefix as the path. like langregion.ini next to lang/.
+ }
+ }
+}
+
+bool ZipFileReader::GetFileInfo(const char *path, File::FileInfo *info) {
+ struct zip_stat zstat;
+ char temp_path[1024];
+ snprintf(temp_path, sizeof(temp_path), "%s%s", inZipPath_, path);
+
+ // Clear some things to start.
+ info->isDirectory = false;
+ info->isWritable = false;
+ info->size = 0;
+
+ {
+ std::lock_guard guard(lock_);
+ if (0 != zip_stat(zip_file_, temp_path, ZIP_FL_NOCASE | ZIP_FL_UNCHANGED, &zstat)) {
+ // ZIP files do not have real directories, so we'll end up here if we
+ // try to stat one. For now that's fine.
+ info->exists = false;
+ return false;
+ }
+ }
+
+ // Zips usually don't contain directory entries, but they may.
+ if ((zstat.valid & ZIP_STAT_NAME) != 0 && zstat.name) {
+ info->isDirectory = zstat.name[strlen(zstat.name) - 1] == '/';
+ }
+ if ((zstat.valid & ZIP_STAT_SIZE) != 0) {
+ info->size = zstat.size;
+ }
+
+ info->fullName = Path(path);
+ info->exists = true;
+ return true;
+}
+
+class ZipFileReaderFileReference : public VFSFileReference {
+public:
+ int zi;
+};
+
+class ZipFileReaderOpenFile : public VFSOpenFile {
+public:
+ ~ZipFileReaderOpenFile() {
+ // Needs to be closed properly and unlocked.
+ _dbg_assert_(zf == nullptr);
+ }
+ ZipFileReaderFileReference *reference;
+ zip_file_t *zf = nullptr;
+};
+
+VFSFileReference *ZipFileReader::GetFile(const char *path) {
+ std::lock_guard guard(lock_);
+ int zi = zip_name_locate(zip_file_, path, ZIP_FL_NOCASE);
+ if (zi < 0) {
+ // Not found.
+ return nullptr;
+ }
+ ZipFileReaderFileReference *ref = new ZipFileReaderFileReference();
+ ref->zi = zi;
+ return ref;
+}
+
+bool ZipFileReader::GetFileInfo(VFSFileReference *vfsReference, File::FileInfo *fileInfo) {
+ ZipFileReaderFileReference *reference = (ZipFileReaderFileReference *)vfsReference;
+ // If you crash here, you called this while having the lock held by having the file open.
+ // Don't do that, check the info before you open the file.
+ std::lock_guard guard(lock_);
+ zip_stat_t zstat;
+ if (zip_stat_index(zip_file_, reference->zi, 0, &zstat) != 0)
+ return false;
+ *fileInfo = File::FileInfo{};
+ fileInfo->size = 0;
+ if (zstat.valid & ZIP_STAT_SIZE)
+ fileInfo->size = zstat.size;
+ return zstat.size;
+}
+
+void ZipFileReader::ReleaseFile(VFSFileReference *vfsReference) {
+ ZipFileReaderFileReference *reference = (ZipFileReaderFileReference *)vfsReference;
+ // Don't do anything other than deleting it.
+ delete reference;
+}
+
+VFSOpenFile *ZipFileReader::OpenFileForRead(VFSFileReference *vfsReference, size_t *size) {
+ ZipFileReaderFileReference *reference = (ZipFileReaderFileReference *)vfsReference;
+ ZipFileReaderOpenFile *openFile = new ZipFileReaderOpenFile();
+ openFile->reference = reference;
+ *size = 0;
+ // We only allow one file to be open for read concurrently. It's possible that this can be improved,
+ // especially if we only access by index like this.
+ lock_.lock();
+ zip_stat_t zstat;
+ if (zip_stat_index(zip_file_, reference->zi, 0, &zstat) != 0) {
+ lock_.unlock();
+ return nullptr;
+ }
+
+ openFile->zf = zip_fopen_index(zip_file_, reference->zi, 0);
+ if (!openFile->zf) {
+ WARN_LOG(G3D, "File with index %d not found in zip", reference->zi);
+ lock_.unlock();
+ return nullptr;
+ }
+
+ *size = zstat.size;
+ // Intentionally leaving the mutex locked, will be closed in CloseFile.
+ return openFile;
+}
+
+void ZipFileReader::Rewind(VFSOpenFile *vfsOpenFile) {
+ ZipFileReaderOpenFile *openFile = (ZipFileReaderOpenFile *)vfsOpenFile;
+ // Close and re-open.
+ zip_fclose(openFile->zf);
+ openFile->zf = zip_fopen_index(zip_file_, openFile->reference->zi, 0);
+}
+
+size_t ZipFileReader::Read(VFSOpenFile *vfsOpenFile, void *buffer, size_t length) {
+ ZipFileReaderOpenFile *file = (ZipFileReaderOpenFile *)vfsOpenFile;
+ return zip_fread(file->zf, buffer, length);
+}
+
+void ZipFileReader::CloseFile(VFSOpenFile *vfsOpenFile) {
+ ZipFileReaderOpenFile *file = (ZipFileReaderOpenFile *)vfsOpenFile;
+ _dbg_assert_(file->zf != nullptr);
+ zip_fclose(file->zf);
+ file->zf = nullptr;
+ lock_.unlock();
+ delete file;
+}
diff --git a/Common/File/VFS/ZipFileReader.h b/Common/File/VFS/ZipFileReader.h
new file mode 100644
index 0000000000..44a3292ede
--- /dev/null
+++ b/Common/File/VFS/ZipFileReader.h
@@ -0,0 +1,48 @@
+#pragma once
+
+#ifdef SHARED_LIBZIP
+#include
+#else
+#include "ext/libzip/zip.h"
+#endif
+
+#include
+#include
+#include
+
+#include "Common/File/VFS/VFS.h"
+#include "Common/File/FileUtil.h"
+#include "Common/File/Path.h"
+
+class ZipFileReader : public VFSBackend {
+public:
+ static ZipFileReader *Create(const Path &zipFile, const char *inZipPath, bool logErrors = true);
+ ~ZipFileReader();
+
+ bool IsValid() const { return zip_file_ != nullptr; }
+
+ // use delete[] on the returned value.
+ uint8_t *ReadFile(const char *path, size_t *size) override;
+
+ VFSFileReference *GetFile(const char *path) override;
+ bool GetFileInfo(VFSFileReference *vfsReference, File::FileInfo *fileInfo) override;
+ void ReleaseFile(VFSFileReference *vfsReference) override;
+
+ VFSOpenFile *OpenFileForRead(VFSFileReference *vfsReference, size_t *size) override;
+ void Rewind(VFSOpenFile *vfsOpenFile) override;
+ size_t Read(VFSOpenFile *vfsOpenFile, void *buffer, size_t length) override;
+ void CloseFile(VFSOpenFile *vfsOpenFile) override;
+
+ bool GetFileListing(const char *path, std::vector *listing, const char *filter) override;
+ bool GetFileInfo(const char *path, File::FileInfo *info) override;
+ std::string toString() const override {
+ return inZipPath_;
+ }
+
+private:
+ void GetZipListings(const char *path, std::set &files, std::set &directories);
+
+ zip *zip_file_ = nullptr;
+ std::mutex lock_;
+ char inZipPath_[256];
+};
diff --git a/Common/GPU/D3D11/thin3d_d3d11.cpp b/Common/GPU/D3D11/thin3d_d3d11.cpp
index 29bea57c32..58d5a59397 100644
--- a/Common/GPU/D3D11/thin3d_d3d11.cpp
+++ b/Common/GPU/D3D11/thin3d_d3d11.cpp
@@ -91,7 +91,7 @@ public:
void CopyFramebufferImage(Framebuffer *src, int level, int x, int y, int z, Framebuffer *dst, int dstLevel, int dstX, int dstY, int dstZ, int width, int height, int depth, int channelBits, const char *tag) override;
bool BlitFramebuffer(Framebuffer *src, int srcX1, int srcY1, int srcX2, int srcY2, Framebuffer *dst, int dstX1, int dstY1, int dstX2, int dstY2, int channelBits, FBBlitFilter filter, const char *tag) override;
- bool CopyFramebufferToMemorySync(Framebuffer *src, int channelBits, int x, int y, int w, int h, Draw::DataFormat format, void *pixels, int pixelStride, const char *tag) override;
+ bool CopyFramebufferToMemory(Framebuffer *src, int channelBits, int x, int y, int w, int h, Draw::DataFormat format, void *pixels, int pixelStride, ReadbackMode mode, const char *tag) override;
// These functions should be self explanatory.
void BindFramebufferAsRenderTarget(Framebuffer *fbo, const RenderPassInfo &rp, const char *tag) override;
@@ -112,7 +112,7 @@ public:
// Raster state
void SetScissorRect(int left, int top, int width, int height) override;
- void SetViewports(int count, Viewport *viewports) override;
+ void SetViewport(const Viewport &viewport) override;
void SetBlendFactor(float color[4]) override {
if (memcmp(blendFactor_, color, sizeof(float) * 4)) {
memcpy(blendFactor_, color, sizeof(float) * 4);
@@ -420,20 +420,18 @@ void D3D11DrawContext::EndFrame() {
curPipeline_ = nullptr;
}
-void D3D11DrawContext::SetViewports(int count, Viewport *viewports) {
- D3D11_VIEWPORT vp[4];
- for (int i = 0; i < count; i++) {
- DisplayRect rc{ viewports[i].TopLeftX , viewports[i].TopLeftY, viewports[i].Width, viewports[i].Height };
- if (curRenderTargetView_ == bbRenderTargetView_) // Only the backbuffer is actually rotated wrong!
- RotateRectToDisplay(rc, curRTWidth_, curRTHeight_);
- vp[i].TopLeftX = rc.x;
- vp[i].TopLeftY = rc.y;
- vp[i].Width = rc.w;
- vp[i].Height = rc.h;
- vp[i].MinDepth = viewports[i].MinDepth;
- vp[i].MaxDepth = viewports[i].MaxDepth;
- }
- context_->RSSetViewports(count, vp);
+void D3D11DrawContext::SetViewport(const Viewport &viewport) {
+ DisplayRect rc{ viewport.TopLeftX , viewport.TopLeftY, viewport.Width, viewport.Height };
+ if (curRenderTargetView_ == bbRenderTargetView_) // Only the backbuffer is actually rotated wrong!
+ RotateRectToDisplay(rc, curRTWidth_, curRTHeight_);
+ D3D11_VIEWPORT vp;
+ vp.TopLeftX = rc.x;
+ vp.TopLeftY = rc.y;
+ vp.Width = rc.w;
+ vp.Height = rc.h;
+ vp.MinDepth = viewport.MinDepth;
+ vp.MaxDepth = viewport.MaxDepth;
+ context_->RSSetViewports(1, &vp);
}
void D3D11DrawContext::SetScissorRect(int left, int top, int width, int height) {
@@ -493,7 +491,12 @@ static DXGI_FORMAT dataFormatToD3D11(DataFormat format) {
case DataFormat::D16: return DXGI_FORMAT_D16_UNORM;
case DataFormat::D32F: return DXGI_FORMAT_D32_FLOAT;
case DataFormat::D32F_S8: return DXGI_FORMAT_D32_FLOAT_S8X24_UINT;
- case DataFormat::ETC1:
+ case DataFormat::BC1_RGBA_UNORM_BLOCK: return DXGI_FORMAT_BC1_UNORM;
+ case DataFormat::BC2_UNORM_BLOCK: return DXGI_FORMAT_BC2_UNORM;
+ case DataFormat::BC3_UNORM_BLOCK: return DXGI_FORMAT_BC3_UNORM;
+ case DataFormat::BC4_UNORM_BLOCK: return DXGI_FORMAT_BC4_UNORM;
+ case DataFormat::BC5_UNORM_BLOCK: return DXGI_FORMAT_BC5_UNORM;
+ case DataFormat::BC7_UNORM_BLOCK: return DXGI_FORMAT_BC7_UNORM;
default:
return DXGI_FORMAT_UNKNOWN;
}
@@ -1525,7 +1528,7 @@ bool D3D11DrawContext::BlitFramebuffer(Framebuffer *srcfb, int srcX1, int srcY1,
return false;
}
-bool D3D11DrawContext::CopyFramebufferToMemorySync(Framebuffer *src, int channelBits, int bx, int by, int bw, int bh, Draw::DataFormat destFormat, void *pixels, int pixelStride, const char *tag) {
+bool D3D11DrawContext::CopyFramebufferToMemory(Framebuffer *src, int channelBits, int bx, int by, int bw, int bh, Draw::DataFormat destFormat, void *pixels, int pixelStride, ReadbackMode mode, const char *tag) {
D3D11Framebuffer *fb = (D3D11Framebuffer *)src;
if (fb) {
diff --git a/Common/GPU/D3D9/thin3d_d3d9.cpp b/Common/GPU/D3D9/thin3d_d3d9.cpp
index b7e920d604..a560a78fe3 100644
--- a/Common/GPU/D3D9/thin3d_d3d9.cpp
+++ b/Common/GPU/D3D9/thin3d_d3d9.cpp
@@ -113,7 +113,7 @@ static const D3DSTENCILOP stencilOpToD3D9[] = {
D3DSTENCILOP_DECR,
};
-D3DFORMAT FormatToD3DFMT(DataFormat fmt) {
+static D3DFORMAT FormatToD3DFMT(DataFormat fmt) {
switch (fmt) {
case DataFormat::R16_UNORM: return D3DFMT_L16; // closest match, should be a fine substitution if we ignore channels except R.
case DataFormat::R8G8B8A8_UNORM: return D3DFMT_A8R8G8B8;
@@ -125,6 +125,9 @@ D3DFORMAT FormatToD3DFMT(DataFormat fmt) {
case DataFormat::A1R5G5B5_UNORM_PACK16: return D3DFMT_A1R5G5B5;
case DataFormat::D24_S8: return D3DFMT_D24S8;
case DataFormat::D16: return D3DFMT_D16;
+ case DataFormat::BC1_RGBA_UNORM_BLOCK: return D3DFMT_DXT1;
+ case DataFormat::BC2_UNORM_BLOCK: return D3DFMT_DXT3; // DXT3 is indeed BC2.
+ case DataFormat::BC3_UNORM_BLOCK: return D3DFMT_DXT5; // DXT5 is indeed BC3
default: return D3DFMT_UNKNOWN;
}
}
@@ -530,7 +533,7 @@ public:
// Not implemented
}
bool BlitFramebuffer(Framebuffer *src, int srcX1, int srcY1, int srcX2, int srcY2, Framebuffer *dst, int dstX1, int dstY1, int dstX2, int dstY2, int channelBits, FBBlitFilter filter, const char *tag) override;
- bool CopyFramebufferToMemorySync(Framebuffer *src, int channelBits, int x, int y, int w, int h, Draw::DataFormat format, void *pixels, int pixelStride, const char *tag) override;
+ bool CopyFramebufferToMemory(Framebuffer *src, int channelBits, int x, int y, int w, int h, Draw::DataFormat format, void *pixels, int pixelStride, ReadbackMode mode, const char *tag) override;
// These functions should be self explanatory.
void BindFramebufferAsRenderTarget(Framebuffer *fbo, const RenderPassInfo &rp, const char *tag) override;
@@ -573,7 +576,7 @@ public:
// Raster state
void SetScissorRect(int left, int top, int width, int height) override;
- void SetViewports(int count, Viewport *viewports) override;
+ void SetViewport(const Viewport &viewport) override;
void SetBlendFactor(float color[4]) override;
void SetStencilParams(uint8_t refValue, uint8_t writeMask, uint8_t compareMask) override;
@@ -1173,12 +1176,12 @@ void D3D9Context::SetScissorRect(int left, int top, int width, int height) {
dxstate.scissorTest.set(true);
}
-void D3D9Context::SetViewports(int count, Viewport *viewports) {
- int x = (int)viewports[0].TopLeftX;
- int y = (int)viewports[0].TopLeftY;
- int w = (int)viewports[0].Width;
- int h = (int)viewports[0].Height;
- dxstate.viewport.set(x, y, w, h, viewports[0].MinDepth, viewports[0].MaxDepth);
+void D3D9Context::SetViewport(const Viewport &viewport) {
+ int x = (int)viewport.TopLeftX;
+ int y = (int)viewport.TopLeftY;
+ int w = (int)viewport.Width;
+ int h = (int)viewport.Height;
+ dxstate.viewport.set(x, y, w, h, viewport.MinDepth, viewport.MaxDepth);
}
void D3D9Context::SetBlendFactor(float color[4]) {
@@ -1426,7 +1429,7 @@ bool D3D9Context::BlitFramebuffer(Framebuffer *srcfb, int srcX1, int srcY1, int
return SUCCEEDED(device_->StretchRect(srcSurf, &srcRect, dstSurf, &dstRect, (filter == FB_BLIT_LINEAR && channelBits == FB_COLOR_BIT) ? D3DTEXF_LINEAR : D3DTEXF_POINT));
}
-bool D3D9Context::CopyFramebufferToMemorySync(Framebuffer *src, int channelBits, int bx, int by, int bw, int bh, Draw::DataFormat destFormat, void *pixels, int pixelStride, const char *tag) {
+bool D3D9Context::CopyFramebufferToMemory(Framebuffer *src, int channelBits, int bx, int by, int bw, int bh, Draw::DataFormat destFormat, void *pixels, int pixelStride, ReadbackMode mode, const char *tag) {
D3D9Framebuffer *fb = (D3D9Framebuffer *)src;
if (fb) {
@@ -1589,6 +1592,7 @@ uint32_t D3D9Context::GetDataFormatSupport(DataFormat fmt) const {
case DataFormat::BC1_RGBA_UNORM_BLOCK:
case DataFormat::BC2_UNORM_BLOCK:
case DataFormat::BC3_UNORM_BLOCK:
+ // DXT1, DXT3, DXT5.
return FMT_TEXTURE;
default:
return 0;
diff --git a/Common/GPU/DataFormat.h b/Common/GPU/DataFormat.h
index 32463d3a16..6f5cd2bc2e 100644
--- a/Common/GPU/DataFormat.h
+++ b/Common/GPU/DataFormat.h
@@ -46,22 +46,20 @@ enum class DataFormat : uint8_t {
// Block compression formats.
// These are modern names for DXT and friends, now patent free.
// https://msdn.microsoft.com/en-us/library/bb694531.aspx
- BC1_RGBA_UNORM_BLOCK,
- BC1_RGBA_SRGB_BLOCK,
- BC2_UNORM_BLOCK, // 4-bit straight alpha + DXT1 color. Usually not worth using
- BC2_SRGB_BLOCK,
- BC3_UNORM_BLOCK, // 3-bit alpha with 2 ref values (+ magic) + DXT1 color
- BC3_SRGB_BLOCK,
- BC4_UNORM_BLOCK, // 1-channel, same storage as BC3 alpha
- BC4_SNORM_BLOCK,
- BC5_UNORM_BLOCK, // 2-channel RG, each has same storage as BC3 alpha
- BC5_SNORM_BLOCK,
- BC6H_UFLOAT_BLOCK, // TODO
- BC6H_SFLOAT_BLOCK,
- BC7_UNORM_BLOCK, // Highly advanced, very expensive to compress, very good quality.
- BC7_SRGB_BLOCK,
+ BC1_RGBA_UNORM_BLOCK, // 64 bits per 4x4 block. Used by Basis, along with ETC2_R8G8B8_UNORM_BLOCK.
+ BC2_UNORM_BLOCK, // 4-bit straight alpha + DXT1 color. 128 bits per block. Usually not worth using
+ BC3_UNORM_BLOCK, // 3-bit alpha with 2 ref values (+ magic) + DXT1 color. 128 bits per block.
+ BC4_UNORM_BLOCK, // 1-channel, same storage as BC3 alpha. 64 bits per block.
+ BC5_UNORM_BLOCK, // 2-channel RG, each has same storage as BC3 alpha. 128 bits per block.
+ BC7_UNORM_BLOCK, // Highly advanced RGBA, very expensive to compress, very good quality. 128 bits per block.
- ETC1,
+ // Ericsson texture compression.
+ ETC2_R8G8B8_UNORM_BLOCK, // Color-only, 64 bits per 4x4 block.
+ ETC2_R8G8B8A1_UNORM_BLOCK, // Color + alpha, 128 bits per 4x4 block.
+ ETC2_R8G8B8A8_UNORM_BLOCK, // Color + alpha, 128 bits per 4x4 block.
+
+ // This is the one ASTC format used by UASTC / basis Universal.
+ ASTC_4x4_UNORM_BLOCK,
S8,
D16,
@@ -76,6 +74,7 @@ bool DataFormatIsDepthStencil(DataFormat fmt);
inline bool DataFormatIsColor(DataFormat fmt) {
return !DataFormatIsDepthStencil(fmt);
}
+bool DataFormatIsBlockCompressed(DataFormat fmt, int *blockSize);
// Limited format support for now.
const char *DataFormatToString(DataFormat fmt);
diff --git a/Common/GPU/OpenGL/DataFormatGL.cpp b/Common/GPU/OpenGL/DataFormatGL.cpp
index 1b2af548d8..be3b8d8aa2 100644
--- a/Common/GPU/OpenGL/DataFormatGL.cpp
+++ b/Common/GPU/OpenGL/DataFormatGL.cpp
@@ -86,6 +86,73 @@ bool Thin3DFormatToGLFormatAndType(DataFormat fmt, GLuint &internalFormat, GLuin
alignment = 16;
break;
+#ifndef USING_GLES2
+ case DataFormat::BC1_RGBA_UNORM_BLOCK:
+ internalFormat = GL_COMPRESSED_RGB_S3TC_DXT1_EXT;
+ format = GL_RGB;
+ type = GL_FLOAT;
+ alignment = 8;
+ break;
+ case DataFormat::BC2_UNORM_BLOCK:
+ internalFormat = GL_COMPRESSED_RGBA_S3TC_DXT3_EXT;
+ format = GL_RGBA;
+ type = GL_FLOAT;
+ alignment = 16;
+ break;
+ case DataFormat::BC3_UNORM_BLOCK:
+ internalFormat = GL_COMPRESSED_RGBA_S3TC_DXT5_EXT;
+ format = GL_RGBA;
+ type = GL_FLOAT;
+ alignment = 16;
+ break;
+ case DataFormat::BC4_UNORM_BLOCK:
+ internalFormat = GL_COMPRESSED_RED_RGTC1;
+ format = GL_R;
+ type = GL_FLOAT;
+ alignment = 16;
+ break;
+ case DataFormat::BC5_UNORM_BLOCK:
+ internalFormat = GL_COMPRESSED_RG_RGTC2;
+ format = GL_RG;
+ type = GL_FLOAT;
+ alignment = 16;
+ break;
+ case DataFormat::BC7_UNORM_BLOCK:
+ internalFormat = GL_COMPRESSED_RGBA_BPTC_UNORM;
+ format = GL_RGBA;
+ type = GL_FLOAT;
+ alignment = 16;
+ break;
+#endif
+
+ case DataFormat::ETC2_R8G8B8_UNORM_BLOCK:
+ internalFormat = GL_COMPRESSED_RGB8_ETC2;
+ format = GL_RGB;
+ type = GL_FLOAT;
+ alignment = 8;
+ break;
+
+ case DataFormat::ETC2_R8G8B8A1_UNORM_BLOCK:
+ internalFormat = GL_COMPRESSED_RGB8_PUNCHTHROUGH_ALPHA1_ETC2;
+ format = GL_RGBA;
+ type = GL_FLOAT;
+ alignment = 16;
+ break;
+
+ case DataFormat::ETC2_R8G8B8A8_UNORM_BLOCK:
+ internalFormat = GL_COMPRESSED_RGBA8_ETC2_EAC;
+ format = GL_RGBA;
+ type = GL_FLOAT;
+ alignment = 16;
+ break;
+
+ case DataFormat::ASTC_4x4_UNORM_BLOCK:
+ internalFormat = GL_COMPRESSED_RGBA_ASTC_4x4_KHR;
+ format = GL_RGBA;
+ type = GL_FLOAT;
+ alignment = 16;
+ break;
+
default:
return false;
}
diff --git a/Common/GPU/OpenGL/GLFeatures.cpp b/Common/GPU/OpenGL/GLFeatures.cpp
index 96bc41a6e3..9b1e313372 100644
--- a/Common/GPU/OpenGL/GLFeatures.cpp
+++ b/Common/GPU/OpenGL/GLFeatures.cpp
@@ -182,7 +182,9 @@ bool CheckGLExtensions() {
gl_extensions.gpuVendor = GPU_VENDOR_IMGTEC;
} else if (vendor == "Qualcomm") {
gl_extensions.gpuVendor = GPU_VENDOR_QUALCOMM;
- sscanf(renderer, "Adreno (TM) %d", &gl_extensions.modelNumber);
+ if (1 != sscanf(renderer, "Adreno (TM) %d", &gl_extensions.modelNumber)) {
+ gl_extensions.modelNumber = 300; // or what should we default to?
+ }
} else if (vendor == "Broadcom") {
gl_extensions.gpuVendor = GPU_VENDOR_BROADCOM;
// Just for reference: Galaxy Y has renderer == "VideoCore IV HW"
@@ -378,6 +380,12 @@ bool CheckGLExtensions() {
gl_extensions.ARB_explicit_attrib_location = g_set_gl_extensions.count("GL_ARB_explicit_attrib_location") != 0;
gl_extensions.ARB_texture_non_power_of_two = g_set_gl_extensions.count("GL_ARB_texture_non_power_of_two") != 0;
gl_extensions.ARB_shader_stencil_export = g_set_gl_extensions.count("GL_ARB_shader_stencil_export") != 0;
+ gl_extensions.ARB_texture_compression_bptc = g_set_gl_extensions.count("GL_ARB_texture_compression_bptc") != 0;
+ gl_extensions.ARB_texture_compression_rgtc = g_set_gl_extensions.count("GL_ARB_texture_compression_rgtc") != 0;
+ gl_extensions.KHR_texture_compression_astc_ldr = g_set_gl_extensions.count("GL_KHR_texture_compression_astc_ldr") != 0;
+ gl_extensions.EXT_texture_compression_s3tc = g_set_gl_extensions.count("GL_EXT_texture_compression_s3tc") != 0;
+ gl_extensions.OES_texture_compression_astc = g_set_gl_extensions.count("GL_OES_texture_compression_astc") != 0;
+
if (gl_extensions.IsGLES) {
gl_extensions.EXT_blend_func_extended = g_set_gl_extensions.count("GL_EXT_blend_func_extended") != 0;
gl_extensions.OES_texture_npot = g_set_gl_extensions.count("GL_OES_texture_npot") != 0;
@@ -575,6 +583,38 @@ bool CheckGLExtensions() {
gl_extensions.EXT_clip_cull_distance = false;
}
+ // Check the old query API. It doesn't seem to be very reliable (can miss stuff).
+ GLint numCompressedFormats = 0;
+ glGetIntegerv(GL_NUM_COMPRESSED_TEXTURE_FORMATS, &numCompressedFormats);
+ GLint *compressedFormats = new GLint[numCompressedFormats];
+ if (numCompressedFormats > 0) {
+ glGetIntegerv(GL_COMPRESSED_TEXTURE_FORMATS, compressedFormats);
+ for (int i = 0; i < numCompressedFormats; i++) {
+ switch (compressedFormats[i]) {
+ case GL_COMPRESSED_RGB8_ETC2: gl_extensions.supportsETC2 = true; break;
+ case GL_COMPRESSED_RGBA_ASTC_4x4_KHR: gl_extensions.supportsASTC = true; break;
+#ifndef USING_GLES2
+ case GL_COMPRESSED_RGBA_S3TC_DXT5_EXT: gl_extensions.supportsBC123 = true; break;
+ case GL_COMPRESSED_RGBA_BPTC_UNORM: gl_extensions.supportsBC7 = true; break;
+#endif
+ }
+ }
+ }
+
+ // Enable additional formats based on extensions.
+ if (gl_extensions.EXT_texture_compression_s3tc) gl_extensions.supportsBC123 = true;
+ if (gl_extensions.ARB_texture_compression_bptc) gl_extensions.supportsBC7 = true;
+ if (gl_extensions.ARB_texture_compression_rgtc) gl_extensions.supportsBC45 = true;
+ if (gl_extensions.KHR_texture_compression_astc_ldr) gl_extensions.supportsASTC = true;
+ if (gl_extensions.OES_texture_compression_astc) gl_extensions.supportsASTC = true;
+
+ // Now, disable known-emulated texture formats.
+ if (gl_extensions.gpuVendor == GPU_VENDOR_NVIDIA && !gl_extensions.IsGLES) {
+ gl_extensions.supportsETC2 = false;
+ gl_extensions.supportsASTC = false;
+ }
+ delete[] compressedFormats;
+
ProcessGPUFeatures();
int error = glGetError();
diff --git a/Common/GPU/OpenGL/GLFeatures.h b/Common/GPU/OpenGL/GLFeatures.h
index fe72d037b2..04df20f87f 100644
--- a/Common/GPU/OpenGL/GLFeatures.h
+++ b/Common/GPU/OpenGL/GLFeatures.h
@@ -52,6 +52,7 @@ struct GLExtensions {
bool OES_copy_image;
bool OES_texture_float;
bool OES_texture_3D;
+ bool OES_texture_compression_astc;
// ARB
bool ARB_framebuffer_object;
@@ -73,8 +74,14 @@ struct GLExtensions {
bool ARB_texture_non_power_of_two;
bool ARB_stencil_texturing;
bool ARB_shader_stencil_export;
+ bool ARB_texture_compression_bptc;
+ bool ARB_texture_compression_rgtc;
+
+ // KHR
+ bool KHR_texture_compression_astc_ldr;
// EXT
+ bool EXT_texture_compression_s3tc;
bool EXT_swap_control_tear;
bool EXT_discard_framebuffer;
bool EXT_unpack_subimage; // always supported on desktop and ES3
@@ -115,6 +122,12 @@ struct GLExtensions {
int maxVertexTextureUnits;
+ bool supportsETC2;
+ bool supportsBC123;
+ bool supportsBC45;
+ bool supportsBC7;
+ bool supportsASTC;
+
// greater-or-equal than
bool VersionGEThan(int major, int minor, int sub = 0);
int GLSLVersion();
diff --git a/Common/GPU/OpenGL/GLFrameData.cpp b/Common/GPU/OpenGL/GLFrameData.cpp
new file mode 100644
index 0000000000..fa5a051d30
--- /dev/null
+++ b/Common/GPU/OpenGL/GLFrameData.cpp
@@ -0,0 +1,75 @@
+#include "Common/GPU/OpenGL/GLCommon.h"
+#include "Common/GPU/OpenGL/GLFrameData.h"
+#include "Common/GPU/OpenGL/GLRenderManager.h"
+#include "Common/Log.h"
+
+void GLDeleter::Take(GLDeleter &other) {
+ _assert_msg_(IsEmpty(), "Deleter already has stuff");
+ shaders = std::move(other.shaders);
+ programs = std::move(other.programs);
+ buffers = std::move(other.buffers);
+ textures = std::move(other.textures);
+ inputLayouts = std::move(other.inputLayouts);
+ framebuffers = std::move(other.framebuffers);
+ pushBuffers = std::move(other.pushBuffers);
+ other.shaders.clear();
+ other.programs.clear();
+ other.buffers.clear();
+ other.textures.clear();
+ other.inputLayouts.clear();
+ other.framebuffers.clear();
+ other.pushBuffers.clear();
+}
+
+// Runs on the GPU thread.
+void GLDeleter::Perform(GLRenderManager *renderManager, bool skipGLCalls) {
+ for (auto pushBuffer : pushBuffers) {
+ renderManager->UnregisterPushBuffer(pushBuffer);
+ if (skipGLCalls) {
+ pushBuffer->Destroy(false);
+ }
+ delete pushBuffer;
+ }
+ pushBuffers.clear();
+ for (auto shader : shaders) {
+ if (skipGLCalls)
+ shader->shader = 0; // prevent the glDeleteShader
+ delete shader;
+ }
+ shaders.clear();
+ for (auto program : programs) {
+ if (skipGLCalls)
+ program->program = 0; // prevent the glDeleteProgram
+ delete program;
+ }
+ programs.clear();
+ for (auto buffer : buffers) {
+ if (skipGLCalls)
+ buffer->buffer_ = 0;
+ delete buffer;
+ }
+ buffers.clear();
+ for (auto texture : textures) {
+ if (skipGLCalls)
+ texture->texture = 0;
+ delete texture;
+ }
+ textures.clear();
+ for (auto inputLayout : inputLayouts) {
+ // No GL objects in an inputLayout yet
+ delete inputLayout;
+ }
+ inputLayouts.clear();
+ for (auto framebuffer : framebuffers) {
+ if (skipGLCalls) {
+ framebuffer->handle = 0;
+ framebuffer->color_texture.texture = 0;
+ framebuffer->z_stencil_buffer = 0;
+ framebuffer->z_stencil_texture.texture = 0;
+ framebuffer->z_buffer = 0;
+ framebuffer->stencil_buffer = 0;
+ }
+ delete framebuffer;
+ }
+ framebuffers.clear();
+}
diff --git a/Common/GPU/OpenGL/GLFrameData.h b/Common/GPU/OpenGL/GLFrameData.h
new file mode 100644
index 0000000000..a50fbe7d55
--- /dev/null
+++ b/Common/GPU/OpenGL/GLFrameData.h
@@ -0,0 +1,52 @@
+#pragma once
+
+#include
+#include
+#include
+#include
+
+#include "Common/GPU/OpenGL/GLCommon.h"
+
+class GLRShader;
+class GLRBuffer;
+class GLRTexture;
+class GLRInputLayout;
+class GLRFramebuffer;
+class GLPushBuffer;
+class GLRProgram;
+class GLRenderManager;
+
+class GLDeleter {
+public:
+ void Perform(GLRenderManager *renderManager, bool skipGLCalls);
+
+ bool IsEmpty() const {
+ return shaders.empty() && programs.empty() && buffers.empty() && textures.empty() && inputLayouts.empty() && framebuffers.empty() && pushBuffers.empty();
+ }
+
+ void Take(GLDeleter &other);
+
+ std::vector shaders;
+ std::vector programs;
+ std::vector buffers;
+ std::vector textures;
+ std::vector inputLayouts;
+ std::vector framebuffers;
+ std::vector pushBuffers;
+};
+
+// Per-frame data, round-robin so we can overlap submission with execution of the previous frame.
+struct GLFrameData {
+ bool skipSwap = false;
+
+ std::mutex fenceMutex;
+ std::condition_variable fenceCondVar;
+ bool readyForFence = true;
+
+ // Swapchain.
+ bool hasBegun = false;
+
+ GLDeleter deleter;
+ GLDeleter deleter_prev;
+ std::set activePushBuffers;
+};
diff --git a/Common/GPU/OpenGL/GLQueueRunner.cpp b/Common/GPU/OpenGL/GLQueueRunner.cpp
index 63f75af330..097ecdc2c4 100644
--- a/Common/GPU/OpenGL/GLQueueRunner.cpp
+++ b/Common/GPU/OpenGL/GLQueueRunner.cpp
@@ -88,9 +88,6 @@ void GLQueueRunner::DestroyDeviceObjects() {
delete[] readbackBuffer_;
readbackBuffer_ = nullptr;
readbackBufferSize_ = 0;
- delete[] tempBuffer_;
- tempBuffer_ = nullptr;
- tempBufferSize_ = 0;
CHECK_GL_ERROR_IF_DEBUG();
}
@@ -112,14 +109,14 @@ static std::string GetInfoLog(GLuint name, Getiv getiv, GetLog getLog) {
return infoLog;
}
-int GLQueueRunner::GetStereoBufferIndex(const char *uniformName) {
+static int GetStereoBufferIndex(const char *uniformName) {
if (!uniformName) return -1;
else if (strcmp(uniformName, "u_view") == 0) return 0;
else if (strcmp(uniformName, "u_proj_lens") == 0) return 1;
else return -1;
}
-std::string GLQueueRunner::GetStereoBufferLayout(const char *uniformName) {
+static std::string GetStereoBufferLayout(const char *uniformName) {
if (strcmp(uniformName, "u_view") == 0) return "ViewMatrices";
else if (strcmp(uniformName, "u_proj_lens") == 0) return "ProjectionMatrix";
else return "undefined";
@@ -390,14 +387,23 @@ void GLQueueRunner::RunInitSteps(const std::vector &steps, bool ski
// For things to show in RenderDoc, need to split into glTexImage2D(..., nullptr) and glTexSubImage.
+ int blockSize = 0;
+ bool bc = Draw::DataFormatIsBlockCompressed(step.texture_image.format, &blockSize);
+
GLenum internalFormat, format, type;
int alignment;
Thin3DFormatToGLFormatAndType(step.texture_image.format, internalFormat, format, type, alignment);
if (step.texture_image.depth == 1) {
- glTexImage2D(tex->target,
- step.texture_image.level, internalFormat,
- step.texture_image.width, step.texture_image.height, 0,
- format, type, step.texture_image.data);
+ if (bc) {
+ int dataSize = ((step.texture_image.width + 3) & ~3) * ((step.texture_image.height + 3) & ~3) * blockSize / 16;
+ glCompressedTexImage2D(tex->target, step.texture_image.level, internalFormat,
+ step.texture_image.width, step.texture_image.height, 0, dataSize, step.texture_image.data);
+ } else {
+ glTexImage2D(tex->target,
+ step.texture_image.level, internalFormat,
+ step.texture_image.width, step.texture_image.height, 0,
+ format, type, step.texture_image.data);
+ }
} else {
glTexImage3D(tex->target,
step.texture_image.level, internalFormat,
@@ -508,8 +514,8 @@ void GLQueueRunner::InitCreateFramebuffer(const GLRInitStep &step) {
glTexParameteri(GL_TEXTURE_2D, GL_TEXTURE_WRAP_S, tex.wrapS);
glTexParameteri(GL_TEXTURE_2D, GL_TEXTURE_WRAP_T, tex.wrapT);
- glTexParameteri(GL_TEXTURE_2D, GL_TEXTURE_MIN_FILTER, tex.magFilter);
- glTexParameteri(GL_TEXTURE_2D, GL_TEXTURE_MAG_FILTER, tex.minFilter);
+ glTexParameteri(GL_TEXTURE_2D, GL_TEXTURE_MAG_FILTER, tex.magFilter);
+ glTexParameteri(GL_TEXTURE_2D, GL_TEXTURE_MIN_FILTER, tex.minFilter);
if (!gl_extensions.IsGLES || gl_extensions.GLES3) {
glTexParameteri(GL_TEXTURE_2D, GL_TEXTURE_MAX_LEVEL, 0);
}
@@ -747,11 +753,6 @@ void GLQueueRunner::RunSteps(const std::vector &steps, bool skipGLCal
CHECK_GL_ERROR_IF_DEBUG();
}
-void GLQueueRunner::LogSteps(const std::vector &steps) {
-
-}
-
-
void GLQueueRunner::PerformBlit(const GLRStep &step) {
CHECK_GL_ERROR_IF_DEBUG();
// Without FBO_ARB / GLES3, this will collide with bind_for_read, but there's nothing
@@ -848,7 +849,7 @@ void GLQueueRunner::PerformRenderPass(const GLRStep &step, bool first, bool last
glDepthFunc(c.depth.func);
depthFunc = c.depth.func;
}
- } else if (!c.depth.enabled && depthEnabled) {
+ } else if (/* !c.depth.enabled && */ depthEnabled) {
glDisable(GL_DEPTH_TEST);
depthEnabled = false;
}
@@ -860,7 +861,7 @@ void GLQueueRunner::PerformRenderPass(const GLRStep &step, bool first, bool last
stencilEnabled = true;
}
glStencilFunc(c.stencilFunc.func, c.stencilFunc.ref, c.stencilFunc.compareMask);
- } else if (stencilEnabled) {
+ } else if (/* !c.stencilFunc.enabled && */stencilEnabled) {
glDisable(GL_STENCIL_TEST);
stencilEnabled = false;
}
@@ -882,7 +883,7 @@ void GLQueueRunner::PerformRenderPass(const GLRStep &step, bool first, bool last
blendEqAlpha = c.blend.funcAlpha;
}
glBlendFuncSeparate(c.blend.srcColor, c.blend.dstColor, c.blend.srcAlpha, c.blend.dstAlpha);
- } else if (!c.blend.enabled && blendEnabled) {
+ } else if (/* !c.blend.enabled && */ blendEnabled) {
glDisable(GL_BLEND);
blendEnabled = false;
}
@@ -902,7 +903,7 @@ void GLQueueRunner::PerformRenderPass(const GLRStep &step, bool first, bool last
if (logicOp != c.logic.logicOp) {
glLogicOp(c.logic.logicOp);
}
- } else if (!c.logic.enabled && logicEnabled) {
+ } else if (/* !c.logic.enabled && */ logicEnabled) {
glDisable(GL_COLOR_LOGIC_OP);
logicEnabled = false;
}
@@ -983,24 +984,18 @@ void GLQueueRunner::PerformRenderPass(const GLRStep &step, bool first, bool last
}
case GLRRenderCommand::UNIFORM4F:
{
+ _dbg_assert_(curProgram);
int loc = c.uniform4.loc ? *c.uniform4.loc : -1;
if (c.uniform4.name) {
loc = curProgram->GetUniformLoc(c.uniform4.name);
}
if (loc >= 0) {
+ _dbg_assert_(c.uniform4.count >=1 && c.uniform4.count <=4);
switch (c.uniform4.count) {
- case 1:
- glUniform1f(loc, c.uniform4.v[0]);
- break;
- case 2:
- glUniform2fv(loc, 1, c.uniform4.v);
- break;
- case 3:
- glUniform3fv(loc, 1, c.uniform4.v);
- break;
- case 4:
- glUniform4fv(loc, 1, c.uniform4.v);
- break;
+ case 1: glUniform1f(loc, c.uniform4.v[0]); break;
+ case 2: glUniform2fv(loc, 1, c.uniform4.v); break;
+ case 3: glUniform3fv(loc, 1, c.uniform4.v); break;
+ case 4: glUniform4fv(loc, 1, c.uniform4.v); break;
}
}
CHECK_GL_ERROR_IF_DEBUG();
@@ -1014,6 +1009,7 @@ void GLQueueRunner::PerformRenderPass(const GLRStep &step, bool first, bool last
loc = curProgram->GetUniformLoc(c.uniform4.name);
}
if (loc >= 0) {
+ _dbg_assert_(c.uniform4.count >=1 && c.uniform4.count <=4);
switch (c.uniform4.count) {
case 1: glUniform1uiv(loc, 1, (GLuint *)c.uniform4.v); break;
case 2: glUniform2uiv(loc, 1, (GLuint *)c.uniform4.v); break;
@@ -1032,6 +1028,7 @@ void GLQueueRunner::PerformRenderPass(const GLRStep &step, bool first, bool last
loc = curProgram->GetUniformLoc(c.uniform4.name);
}
if (loc >= 0) {
+ _dbg_assert_(c.uniform4.count >=1 && c.uniform4.count <=4);
switch (c.uniform4.count) {
case 1: glUniform1iv(loc, 1, (GLint *)c.uniform4.v); break;
case 2: glUniform2iv(loc, 1, (GLint *)c.uniform4.v); break;
@@ -1193,14 +1190,14 @@ void GLQueueRunner::PerformRenderPass(const GLRStep &step, bool first, bool last
Crash();
} else if (c.bind_buffer.target == GL_ELEMENT_ARRAY_BUFFER) {
GLuint buf = c.bind_buffer.buffer ? c.bind_buffer.buffer->buffer_ : 0;
- _dbg_assert_(!c.bind_buffer.buffer->Mapped());
+ _dbg_assert_(!(c.bind_buffer.buffer && c.bind_buffer.buffer->Mapped()));
if (buf != curElemArrayBuffer) {
glBindBuffer(GL_ELEMENT_ARRAY_BUFFER, buf);
curElemArrayBuffer = buf;
}
} else {
GLuint buf = c.bind_buffer.buffer ? c.bind_buffer.buffer->buffer_ : 0;
- _dbg_assert_(!c.bind_buffer.buffer->Mapped());
+ _dbg_assert_(!(c.bind_buffer.buffer && c.bind_buffer.buffer->Mapped()));
glBindBuffer(c.bind_buffer.target, buf);
}
CHECK_GL_ERROR_IF_DEBUG();
@@ -1267,7 +1264,7 @@ void GLQueueRunner::PerformRenderPass(const GLRStep &step, bool first, bool last
}
case GLRRenderCommand::TEXTURELOD:
{
- GLint slot = c.textureSampler.slot;
+ GLint slot = c.textureLod.slot;
if (slot != activeSlot) {
glActiveTexture(GL_TEXTURE0 + slot);
activeSlot = slot;
@@ -1325,7 +1322,7 @@ void GLQueueRunner::PerformRenderPass(const GLRStep &step, bool first, bool last
}
glFrontFace(c.raster.frontFace);
glCullFace(c.raster.cullFace);
- } else if (!c.raster.cullEnable && cullEnabled) {
+ } else if (/* !c.raster.cullEnable && */ cullEnabled) {
glDisable(GL_CULL_FACE);
cullEnabled = false;
}
@@ -1334,7 +1331,7 @@ void GLQueueRunner::PerformRenderPass(const GLRStep &step, bool first, bool last
glEnable(GL_DITHER);
ditherEnabled = true;
}
- } else if (!c.raster.ditherEnable && ditherEnabled) {
+ } else if (/* !c.raster.ditherEnable && */ ditherEnabled) {
glDisable(GL_DITHER);
ditherEnabled = false;
}
@@ -1344,7 +1341,7 @@ void GLQueueRunner::PerformRenderPass(const GLRStep &step, bool first, bool last
glEnable(GL_DEPTH_CLAMP);
depthClampEnabled = true;
}
- } else if (!c.raster.depthClampEnable && depthClampEnabled) {
+ } else if (/* !c.raster.depthClampEnable && */ depthClampEnabled) {
glDisable(GL_DEPTH_CLAMP);
depthClampEnabled = false;
}
@@ -1481,26 +1478,24 @@ void GLQueueRunner::PerformReadback(const GLRStep &pass) {
CHECK_GL_ERROR_IF_DEBUG();
// Always read back in 8888 format for the color aspect.
- GLuint internalFormat = GL_RGBA;
GLuint format = GL_RGBA;
GLuint type = GL_UNSIGNED_BYTE;
int srcAlignment = 4;
- int dstAlignment = (int)DataFormatSizeInBytes(pass.readback.dstFormat);
#ifndef USING_GLES2
if (pass.readback.aspectMask & GL_DEPTH_BUFFER_BIT) {
- internalFormat = GL_DEPTH_COMPONENT;
format = GL_DEPTH_COMPONENT;
type = GL_FLOAT;
srcAlignment = 4;
} else if (pass.readback.aspectMask & GL_STENCIL_BUFFER_BIT) {
- internalFormat = GL_STENCIL_INDEX;
format = GL_STENCIL_INDEX;
type = GL_UNSIGNED_BYTE;
srcAlignment = 1;
}
#endif
+ readbackAspectMask_ = pass.readback.aspectMask;
+
int pixelStride = pass.readback.srcRect.w;
// Apply the correct alignment.
glPixelStorei(GL_PACK_ALIGNMENT, srcAlignment);
@@ -1511,31 +1506,20 @@ void GLQueueRunner::PerformReadback(const GLRStep &pass) {
GLRect2D rect = pass.readback.srcRect;
- bool convert = internalFormat == GL_RGBA && pass.readback.dstFormat != DataFormat::R8G8B8A8_UNORM;
-
- int tempSize = srcAlignment * rect.w * rect.h;
- int readbackSize = dstAlignment * rect.w * rect.h;
- if (convert && tempSize > tempBufferSize_) {
- delete[] tempBuffer_;
- tempBuffer_ = new uint8_t[tempSize];
- tempBufferSize_ = tempSize;
- }
+ int readbackSize = srcAlignment * rect.w * rect.h;
if (readbackSize > readbackBufferSize_) {
delete[] readbackBuffer_;
readbackBuffer_ = new uint8_t[readbackSize];
readbackBufferSize_ = readbackSize;
}
- glReadPixels(rect.x, rect.y, rect.w, rect.h, format, type, convert ? tempBuffer_ : readbackBuffer_);
+ glReadPixels(rect.x, rect.y, rect.w, rect.h, format, type, readbackBuffer_);
#ifdef DEBUG_READ_PIXELS
LogReadPixelsError(glGetError());
#endif
if (!gl_extensions.IsGLES || gl_extensions.GLES3) {
glPixelStorei(GL_PACK_ROW_LENGTH, 0);
}
- if (convert && tempBuffer_ && readbackBuffer_) {
- ConvertFromRGBA8888(readbackBuffer_, tempBuffer_, pixelStride, pixelStride, rect.w, rect.h, pass.readback.dstFormat);
- }
CHECK_GL_ERROR_IF_DEBUG();
}
@@ -1564,7 +1548,7 @@ void GLQueueRunner::PerformReadbackImage(const GLRStep &pass) {
glGetTexLevelParameteriv(GL_TEXTURE_2D, pass.readback_image.mipLevel, GL_TEXTURE_WIDTH, &w);
glGetTexLevelParameteriv(GL_TEXTURE_2D, pass.readback_image.mipLevel, GL_TEXTURE_HEIGHT, &h);
- int size = 4 * std::max((int)w, rect.x + rect.w) * std::max((int)h, rect.h);
+ int size = 4 * std::max((int)w, rect.x + rect.w) * std::max((int)h, rect.y + rect.h);
if (size > readbackBufferSize_) {
delete[] readbackBuffer_;
readbackBuffer_ = new uint8_t[size];
@@ -1615,7 +1599,7 @@ void GLQueueRunner::PerformBindFramebufferAsRenderTarget(const GLRStep &pass) {
CHECK_GL_ERROR_IF_DEBUG();
}
-void GLQueueRunner::CopyReadbackBuffer(int width, int height, Draw::DataFormat srcFormat, Draw::DataFormat destFormat, int pixelStride, uint8_t *pixels) {
+void GLQueueRunner::CopyFromReadbackBuffer(GLRFramebuffer *framebuffer, int width, int height, Draw::DataFormat srcFormat, Draw::DataFormat destFormat, int pixelStride, uint8_t *pixels) {
// TODO: Maybe move data format conversion here, and always read back 8888. Drivers
// don't usually provide very optimized conversion implementations, though some do.
// Just need to be careful about dithering, which may break Danganronpa.
@@ -1624,8 +1608,25 @@ void GLQueueRunner::CopyReadbackBuffer(int width, int height, Draw::DataFormat s
// Something went wrong during the read and no readback buffer was allocated, probably.
return;
}
- for (int y = 0; y < height; y++) {
- memcpy(pixels + y * pixelStride * bpp, readbackBuffer_ + y * width * bpp, width * bpp);
+
+ // Always read back in 8888 format for the color aspect.
+ GLuint internalFormat = GL_RGBA;
+#ifndef USING_GLES2
+ if (readbackAspectMask_ & GL_DEPTH_BUFFER_BIT) {
+ internalFormat = GL_DEPTH_COMPONENT;
+ } else if (readbackAspectMask_ & GL_STENCIL_BUFFER_BIT) {
+ internalFormat = GL_STENCIL_INDEX;
+ }
+#endif
+
+ bool convert = internalFormat == GL_RGBA && destFormat != Draw::DataFormat::R8G8B8A8_UNORM;
+ if (convert) {
+ // srcStride is width because we read back "packed" (with no gaps) from GL.
+ ConvertFromRGBA8888(pixels, readbackBuffer_, pixelStride, width, width, height, destFormat);
+ } else {
+ for (int y = 0; y < height; y++) {
+ memcpy(pixels + y * pixelStride * bpp, readbackBuffer_ + y * width * bpp, width * bpp);
+ }
}
}
@@ -1758,7 +1759,7 @@ void GLQueueRunner::fbo_unbind() {
glBindFramebuffer(GL_FRAMEBUFFER, g_defaultFBO);
#endif
-#if PPSSPP_PLATFORM(IOS)
+#if PPSSPP_PLATFORM(IOS) && !defined(__LIBRETRO__)
bindDefaultFBO();
#endif
diff --git a/Common/GPU/OpenGL/GLQueueRunner.h b/Common/GPU/OpenGL/GLQueueRunner.h
index b91648a1fa..f51f8e9999 100644
--- a/Common/GPU/OpenGL/GLQueueRunner.h
+++ b/Common/GPU/OpenGL/GLQueueRunner.h
@@ -6,6 +6,7 @@
#include
#include "Common/GPU/OpenGL/GLCommon.h"
+#include "Common/GPU/OpenGL/GLFrameData.h"
#include "Common/GPU/DataFormat.h"
#include "Common/GPU/Shader.h"
#include "Common/GPU/thin3d.h"
@@ -358,22 +359,14 @@ public:
caps_ = caps;
}
- int GetStereoBufferIndex(const char *uniformName);
- std::string GetStereoBufferLayout(const char *uniformName);
-
void RunInitSteps(const std::vector &steps, bool skipGLCalls);
void RunSteps(const std::vector &steps, bool skipGLCalls, bool keepSteps, bool useVR);
- void LogSteps(const std::vector &steps);
void CreateDeviceObjects();
void DestroyDeviceObjects();
- inline int RPIndex(GLRRenderPassAction color, GLRRenderPassAction depth) {
- return (int)depth * 3 + (int)color;
- }
-
- void CopyReadbackBuffer(int width, int height, Draw::DataFormat srcFormat, Draw::DataFormat destFormat, int pixelStride, uint8_t *pixels);
+ void CopyFromReadbackBuffer(GLRFramebuffer *framebuffer, int width, int height, Draw::DataFormat srcFormat, Draw::DataFormat destFormat, int pixelStride, uint8_t *pixels);
void Resize(int width, int height) {
targetWidth_ = width;
@@ -419,9 +412,7 @@ private:
// We size it generously.
uint8_t *readbackBuffer_ = nullptr;
int readbackBufferSize_ = 0;
- // Temp buffer for color conversion
- uint8_t *tempBuffer_ = nullptr;
- int tempBufferSize_ = 0;
+ uint32_t readbackAspectMask_ = 0;
float maxAnisotropyLevel_ = 0.0f;
diff --git a/Common/GPU/OpenGL/GLRenderManager.cpp b/Common/GPU/OpenGL/GLRenderManager.cpp
index f0a1cce354..b10caa2541 100644
--- a/Common/GPU/OpenGL/GLRenderManager.cpp
+++ b/Common/GPU/OpenGL/GLRenderManager.cpp
@@ -41,77 +41,6 @@ GLRTexture::~GLRTexture() {
}
}
-void GLDeleter::Take(GLDeleter &other) {
- _assert_msg_(IsEmpty(), "Deleter already has stuff");
- shaders = std::move(other.shaders);
- programs = std::move(other.programs);
- buffers = std::move(other.buffers);
- textures = std::move(other.textures);
- inputLayouts = std::move(other.inputLayouts);
- framebuffers = std::move(other.framebuffers);
- pushBuffers = std::move(other.pushBuffers);
- other.shaders.clear();
- other.programs.clear();
- other.buffers.clear();
- other.textures.clear();
- other.inputLayouts.clear();
- other.framebuffers.clear();
- other.pushBuffers.clear();
-}
-
-// Runs on the GPU thread.
-void GLDeleter::Perform(GLRenderManager *renderManager, bool skipGLCalls) {
- for (auto pushBuffer : pushBuffers) {
- renderManager->UnregisterPushBuffer(pushBuffer);
- if (skipGLCalls) {
- pushBuffer->Destroy(false);
- }
- delete pushBuffer;
- }
- pushBuffers.clear();
- for (auto shader : shaders) {
- if (skipGLCalls)
- shader->shader = 0; // prevent the glDeleteShader
- delete shader;
- }
- shaders.clear();
- for (auto program : programs) {
- if (skipGLCalls)
- program->program = 0; // prevent the glDeleteProgram
- delete program;
- }
- programs.clear();
- for (auto buffer : buffers) {
- if (skipGLCalls)
- buffer->buffer_ = 0;
- delete buffer;
- }
- buffers.clear();
- for (auto texture : textures) {
- if (skipGLCalls)
- texture->texture = 0;
- delete texture;
- }
- textures.clear();
- for (auto inputLayout : inputLayouts) {
- // No GL objects in an inputLayout yet
- delete inputLayout;
- }
- inputLayouts.clear();
- for (auto framebuffer : framebuffers) {
- if (skipGLCalls) {
- framebuffer->handle = 0;
- framebuffer->color_texture.texture = 0;
- framebuffer->z_stencil_buffer = 0;
- framebuffer->z_stencil_texture.texture = 0;
- framebuffer->z_buffer = 0;
- framebuffer->stencil_buffer = 0;
- }
- delete framebuffer;
- }
- framebuffers.clear();
-}
-
GLRenderManager::~GLRenderManager() {
_dbg_assert_(!run_);
@@ -360,7 +289,7 @@ void GLRenderManager::BlitFramebuffer(GLRFramebuffer *src, GLRect2D srcRect, GLR
steps_.push_back(step);
}
-bool GLRenderManager::CopyFramebufferToMemorySync(GLRFramebuffer *src, int aspectBits, int x, int y, int w, int h, Draw::DataFormat destFormat, uint8_t *pixels, int pixelStride, const char *tag) {
+bool GLRenderManager::CopyFramebufferToMemory(GLRFramebuffer *src, int aspectBits, int x, int y, int w, int h, Draw::DataFormat destFormat, uint8_t *pixels, int pixelStride, Draw::ReadbackMode mode, const char *tag) {
_assert_(pixels);
GLRStep *step = new GLRStep{ GLRStepType::READBACK };
@@ -387,7 +316,7 @@ bool GLRenderManager::CopyFramebufferToMemorySync(GLRFramebuffer *src, int aspec
} else {
return false;
}
- queueRunner_.CopyReadbackBuffer(w, h, srcFormat, destFormat, pixelStride, pixels);
+ queueRunner_.CopyFromReadbackBuffer(src, w, h, srcFormat, destFormat, pixelStride, pixels);
return true;
}
@@ -404,7 +333,7 @@ void GLRenderManager::CopyImageToMemorySync(GLRTexture *texture, int mipLevel, i
curRenderStep_ = nullptr;
FlushSync();
- queueRunner_.CopyReadbackBuffer(w, h, Draw::DataFormat::R8G8B8A8_UNORM, destFormat, pixelStride, pixels);
+ queueRunner_.CopyFromReadbackBuffer(nullptr, w, h, Draw::DataFormat::R8G8B8A8_UNORM, destFormat, pixelStride, pixels);
}
void GLRenderManager::BeginFrame() {
@@ -414,7 +343,7 @@ void GLRenderManager::BeginFrame() {
int curFrame = GetCurFrame();
- FrameData &frameData = frameData_[curFrame];
+ GLFrameData &frameData = frameData_[curFrame];
{
VLOG("PUSH: BeginFrame (curFrame = %d, readyForFence = %d, time=%0.3f)", curFrame, (int)frameData.readyForFence, time_now_d());
std::unique_lock lock(frameData.fenceMutex);
@@ -435,7 +364,7 @@ void GLRenderManager::Finish() {
curRenderStep_ = nullptr; // EndCurRenderStep is this simple here.
int curFrame = GetCurFrame();
- FrameData &frameData = frameData_[curFrame];
+ GLFrameData &frameData = frameData_[curFrame];
frameData_[curFrame].deleter.Take(deleter_);
@@ -463,7 +392,7 @@ void GLRenderManager::Finish() {
// Render thread. Returns true if the caller should handle a swap.
bool GLRenderManager::Run(GLRRenderThreadTask &task) {
- FrameData &frameData = frameData_[task.frame];
+ GLFrameData &frameData = frameData_[task.frame];
if (!frameData.hasBegun) {
frameData.hasBegun = true;
diff --git a/Common/GPU/OpenGL/GLRenderManager.h b/Common/GPU/OpenGL/GLRenderManager.h
index 2d213be8b8..281a42353f 100644
--- a/Common/GPU/OpenGL/GLRenderManager.h
+++ b/Common/GPU/OpenGL/GLRenderManager.h
@@ -10,11 +10,12 @@
#include
#include
-#include "Common/GPU/OpenGL/GLCommon.h"
#include "Common/GPU/MiscTypes.h"
#include "Common/Data/Convert/SmallDataConvert.h"
#include "Common/Log.h"
-#include "GLQueueRunner.h"
+#include "Common/GPU/OpenGL/GLQueueRunner.h"
+#include "Common/GPU/OpenGL/GLFrameData.h"
+#include "Common/GPU/OpenGL/GLCommon.h"
class GLRInputLayout;
class GLPushBuffer;
@@ -373,25 +374,6 @@ enum class GLRRunType {
class GLRenderManager;
class GLPushBuffer;
-class GLDeleter {
-public:
- void Perform(GLRenderManager *renderManager, bool skipGLCalls);
-
- bool IsEmpty() const {
- return shaders.empty() && programs.empty() && buffers.empty() && textures.empty() && inputLayouts.empty() && framebuffers.empty() && pushBuffers.empty();
- }
-
- void Take(GLDeleter &other);
-
- std::vector shaders;
- std::vector programs;
- std::vector buffers;
- std::vector textures;
- std::vector inputLayouts;
- std::vector framebuffers;
- std::vector pushBuffers;
-};
-
// These are enqueued from the main thread,
// and the render thread pops them off
struct GLRRenderThreadTask {
@@ -570,7 +552,7 @@ public:
// Binds a framebuffer as a texture, for the following draws.
void BindFramebufferAsTexture(GLRFramebuffer *fb, int binding, int aspectBit);
- bool CopyFramebufferToMemorySync(GLRFramebuffer *src, int aspectBits, int x, int y, int w, int h, Draw::DataFormat destFormat, uint8_t *pixels, int pixelStride, const char *tag);
+ bool CopyFramebufferToMemory(GLRFramebuffer *src, int aspectBits, int x, int y, int w, int h, Draw::DataFormat destFormat, uint8_t *pixels, int pixelStride, Draw::ReadbackMode mode, const char *tag);
void CopyImageToMemorySync(GLRTexture *texture, int mipLevel, int x, int y, int w, int h, Draw::DataFormat destFormat, uint8_t *pixels, int pixelStride, const char *tag);
void CopyFramebuffer(GLRFramebuffer *src, GLRect2D srcRect, GLRFramebuffer *dst, GLOffset2D dstPos, int aspectMask, const char *tag);
@@ -1035,23 +1017,7 @@ private:
frameData_[frame].activePushBuffers.insert(buffer);
}
- // Per-frame data, round-robin so we can overlap submission with execution of the previous frame.
- struct FrameData {
- bool skipSwap = false;
-
- std::mutex fenceMutex;
- std::condition_variable fenceCondVar;
- bool readyForFence = true;
-
- // Swapchain.
- bool hasBegun = false;
-
- GLDeleter deleter;
- GLDeleter deleter_prev;
- std::set activePushBuffers;
- };
-
- FrameData frameData_[MAX_INFLIGHT_FRAMES];
+ GLFrameData frameData_[MAX_INFLIGHT_FRAMES];
// Submission time state
bool insideFrame_ = false;
diff --git a/Common/GPU/OpenGL/GLSLProgram.cpp b/Common/GPU/OpenGL/GLSLProgram.cpp
index 45b098d714..fc5698085e 100644
--- a/Common/GPU/OpenGL/GLSLProgram.cpp
+++ b/Common/GPU/OpenGL/GLSLProgram.cpp
@@ -102,7 +102,7 @@ bool glsl_recompile(GLSLProgram *program, std::string *error_message) {
if (!program->vshader_source && !vsh_src) {
size_t sz;
- vsh_src.reset((char *)VFSReadFile(program->vshader_filename, &sz));
+ vsh_src.reset((char *)g_VFS.ReadFile(program->vshader_filename, &sz));
}
if (!program->vshader_source && !vsh_src) {
ERROR_LOG(G3D, "File missing: %s", program->vshader_filename);
@@ -113,7 +113,7 @@ bool glsl_recompile(GLSLProgram *program, std::string *error_message) {
}
if (!program->fshader_source && !fsh_src) {
size_t sz;
- fsh_src.reset((char *)VFSReadFile(program->fshader_filename, &sz));
+ fsh_src.reset((char *)g_VFS.ReadFile(program->fshader_filename, &sz));
}
if (!program->fshader_source && !fsh_src) {
ERROR_LOG(G3D, "File missing: %s", program->fshader_filename);
diff --git a/Common/GPU/OpenGL/thin3d_gl.cpp b/Common/GPU/OpenGL/thin3d_gl.cpp
index 480f0cc3bb..0f04ee0415 100644
--- a/Common/GPU/OpenGL/thin3d_gl.cpp
+++ b/Common/GPU/OpenGL/thin3d_gl.cpp
@@ -365,7 +365,7 @@ public:
void CopyFramebufferImage(Framebuffer *src, int level, int x, int y, int z, Framebuffer *dst, int dstLevel, int dstX, int dstY, int dstZ, int width, int height, int depth, int channelBits, const char *tag) override;
bool BlitFramebuffer(Framebuffer *src, int srcX1, int srcY1, int srcX2, int srcY2, Framebuffer *dst, int dstX1, int dstY1, int dstX2, int dstY2, int channelBits, FBBlitFilter filter, const char *tag) override;
- bool CopyFramebufferToMemorySync(Framebuffer *src, int channelBits, int x, int y, int w, int h, Draw::DataFormat format, void *pixels, int pixelStride, const char *tag) override;
+ bool CopyFramebufferToMemory(Framebuffer *src, int channelBits, int x, int y, int w, int h, Draw::DataFormat format, void *pixels, int pixelStride, ReadbackMode mode, const char *tag) override;
// These functions should be self explanatory.
void BindFramebufferAsRenderTarget(Framebuffer *fbo, const RenderPassInfo &rp, const char *tag) override;
@@ -385,9 +385,9 @@ public:
renderManager_.SetScissor({ left, top, width, height });
}
- void SetViewports(int count, Viewport *viewports) override {
+ void SetViewport(const Viewport &viewport) override {
// Same structure, different name.
- renderManager_.SetViewport((GLRViewport &)*viewports);
+ renderManager_.SetViewport((GLRViewport &)viewport);
}
void SetBlendFactor(float color[4]) override {
@@ -988,7 +988,7 @@ static void LogReadPixelsError(GLenum error) {
}
#endif
-bool OpenGLContext::CopyFramebufferToMemorySync(Framebuffer *src, int channelBits, int x, int y, int w, int h, Draw::DataFormat dataFormat, void *pixels, int pixelStride, const char *tag) {
+bool OpenGLContext::CopyFramebufferToMemory(Framebuffer *src, int channelBits, int x, int y, int w, int h, Draw::DataFormat dataFormat, void *pixels, int pixelStride, ReadbackMode mode, const char *tag) {
if (gl_extensions.IsGLES && (channelBits & FB_COLOR_BIT) == 0) {
// Can't readback depth or stencil on GLES.
return false;
@@ -1001,8 +1001,7 @@ bool OpenGLContext::CopyFramebufferToMemorySync(Framebuffer *src, int channelBit
aspect |= GL_DEPTH_BUFFER_BIT;
if (channelBits & FB_STENCIL_BIT)
aspect |= GL_STENCIL_BUFFER_BIT;
- renderManager_.CopyFramebufferToMemorySync(fb ? fb->framebuffer_ : nullptr, aspect, x, y, w, h, dataFormat, (uint8_t *)pixels, pixelStride, tag);
- return true;
+ return renderManager_.CopyFramebufferToMemory(fb ? fb->framebuffer_ : nullptr, aspect, x, y, w, h, dataFormat, (uint8_t *)pixels, pixelStride, mode, tag);
}
@@ -1557,7 +1556,23 @@ uint32_t OpenGLContext::GetDataFormatSupport(DataFormat fmt) const {
case DataFormat::BC1_RGBA_UNORM_BLOCK:
case DataFormat::BC2_UNORM_BLOCK:
case DataFormat::BC3_UNORM_BLOCK:
- return FMT_TEXTURE;
+ return gl_extensions.supportsBC123 ? FMT_TEXTURE : 0;
+
+ case DataFormat::BC4_UNORM_BLOCK:
+ case DataFormat::BC5_UNORM_BLOCK:
+ return gl_extensions.supportsBC45 ? FMT_TEXTURE : 0;
+
+ case DataFormat::BC7_UNORM_BLOCK:
+ return gl_extensions.supportsBC7 ? FMT_TEXTURE : 0;
+
+ case DataFormat::ASTC_4x4_UNORM_BLOCK:
+ return gl_extensions.supportsASTC ? FMT_TEXTURE : 0;
+
+ case DataFormat::ETC2_R8G8B8_UNORM_BLOCK:
+ case DataFormat::ETC2_R8G8B8A1_UNORM_BLOCK:
+ case DataFormat::ETC2_R8G8B8A8_UNORM_BLOCK:
+ return gl_extensions.supportsETC2 ? FMT_TEXTURE : 0;
+
default:
return 0;
}
diff --git a/Common/GPU/Vulkan/VulkanContext.cpp b/Common/GPU/Vulkan/VulkanContext.cpp
index aa34ad3b0d..032fc128f6 100644
--- a/Common/GPU/Vulkan/VulkanContext.cpp
+++ b/Common/GPU/Vulkan/VulkanContext.cpp
@@ -535,7 +535,7 @@ int VulkanContext::GetBestPhysicalDevice() {
void VulkanContext::ChooseDevice(int physical_device) {
physical_device_ = physical_device;
- INFO_LOG(G3D, "Chose physical device %d: %p", physical_device, physical_devices_[physical_device]);
+ INFO_LOG(G3D, "Chose physical device %d: %s", physical_device, physicalDeviceProperties_[physical_device].properties.deviceName);
GetDeviceLayerProperties();
if (!CheckLayers(device_layer_properties_, device_layer_names_)) {
@@ -711,7 +711,9 @@ VkResult VulkanContext::CreateDevice() {
} else {
VulkanLoadDeviceFunctions(device_, extensionsLookup_);
}
- INFO_LOG(G3D, "Vulkan Device created");
+ INFO_LOG(G3D, "Vulkan Device created: %s", physicalDeviceProperties_[physical_device_].properties.deviceName);
+
+ // Since we successfully created a device (however we got here, might be interesting in debug), we force the choice to be visible in the menu.
VulkanSetAvailable(true);
VmaAllocatorCreateInfo allocatorInfo = {};
@@ -886,6 +888,187 @@ VkResult VulkanContext::ReinitSurface() {
case WINDOWSYSTEM_DISPLAY:
{
VkDisplaySurfaceCreateInfoKHR display{ VK_STRUCTURE_TYPE_DISPLAY_SURFACE_CREATE_INFO_KHR };
+#if !defined(__LIBRETRO__)
+ /*
+ And when not to use libretro need VkDisplaySurfaceCreateInfoKHR this extension,
+ then you need to use dlopen to read vulkan loader in VulkanLoader.cpp.
+ huangzihan China
+ */
+
+ if(!vkGetPhysicalDeviceDisplayPropertiesKHR ||
+ !vkGetPhysicalDeviceDisplayPlanePropertiesKHR ||
+ !vkGetDisplayModePropertiesKHR ||
+ !vkGetDisplayPlaneSupportedDisplaysKHR ||
+ !vkGetDisplayPlaneCapabilitiesKHR ) {
+ _assert_msg_(false, "DISPLAY Vulkan cannot find any vulkan function symbols.");
+ return VK_ERROR_INITIALIZATION_FAILED;
+ }
+
+ //The following code is for reference:
+ // https://github.com/vanfanel/ppsspp
+ // When using the VK_KHR_display extension and not using LIBRETRO, a complete
+ // VkDisplaySurfaceCreateInfoKHR is needed.
+
+ uint32_t display_count;
+ uint32_t plane_count;
+
+ VkDisplayPropertiesKHR *display_props = NULL;
+ VkDisplayPlanePropertiesKHR *plane_props = NULL;
+ VkDisplayModePropertiesKHR* mode_props = NULL;
+
+ VkExtent2D image_size;
+ // This is the chosen physical_device, it has been chosen elsewhere.
+ VkPhysicalDevice phys_device = physical_devices_[physical_device_];
+ VkDisplayModeKHR display_mode = VK_NULL_HANDLE;
+ VkDisplayPlaneAlphaFlagBitsKHR alpha_mode = VK_DISPLAY_PLANE_ALPHA_OPAQUE_BIT_KHR;
+ uint32_t plane = UINT32_MAX;
+
+ // For now, use the first available (connected) display.
+ int display_index = 0;
+
+ VkResult result;
+ bool ret = false;
+ bool mode_found = false;
+
+ int i, j;
+
+ // 1 physical device can have N displays connected.
+ // Vulkan only counts the connected displays.
+
+ // Get a list of displays on the physical device.
+ display_count = 0;
+ vkGetPhysicalDeviceDisplayPropertiesKHR(phys_device, &display_count, NULL);
+ if (display_count == 0) {
+ _assert_msg_(false, "DISPLAY Vulkan couldn't find any displays.");
+ return VK_ERROR_INITIALIZATION_FAILED;
+ }
+ display_props = new VkDisplayPropertiesKHR[display_count];
+ vkGetPhysicalDeviceDisplayPropertiesKHR(phys_device, &display_count, display_props);
+
+ // Get a list of display planes on the physical device.
+ plane_count = 0;
+ vkGetPhysicalDeviceDisplayPlanePropertiesKHR(phys_device, &plane_count, NULL);
+ if (plane_count == 0) {
+ _assert_msg_(false, "DISPLAY Vulkan couldn't find any planes on the physical device");
+ return VK_ERROR_INITIALIZATION_FAILED;
+
+ }
+ plane_props = new VkDisplayPlanePropertiesKHR[plane_count];
+ vkGetPhysicalDeviceDisplayPlanePropertiesKHR(phys_device, &plane_count, plane_props);
+
+ // Get the Vulkan display we are going to use.
+ VkDisplayKHR myDisplay = display_props[display_index].display;
+
+ // Get the list of display modes of the display
+ uint32_t mode_count = 0;
+ vkGetDisplayModePropertiesKHR(phys_device, myDisplay, &mode_count, NULL);
+ if (mode_count == 0) {
+ _assert_msg_(false, "DISPLAY Vulkan couldn't find any video modes on the display");
+ return VK_ERROR_INITIALIZATION_FAILED;
+ }
+ mode_props = new VkDisplayModePropertiesKHR[mode_count];
+ vkGetDisplayModePropertiesKHR(phys_device, myDisplay, &mode_count, mode_props);
+
+ // See if there's an appropiate mode available on the display
+ display_mode = VK_NULL_HANDLE;
+ for (i = 0; i < mode_count; ++i)
+ {
+ const VkDisplayModePropertiesKHR* mode = &mode_props[i];
+
+ if (mode->parameters.visibleRegion.width == pixel_xres &&
+ mode->parameters.visibleRegion.height == pixel_yres)
+ {
+ display_mode = mode->displayMode;
+ mode_found = true;
+ break;
+ }
+ }
+
+ // Free the mode list now.
+ delete [] mode_props;
+
+ // If there are no useable modes found on the display, error out
+ if (display_mode == VK_NULL_HANDLE)
+ {
+ _assert_msg_(false, "DISPLAY Vulkan couldn't find any video modes on the display");
+ return VK_ERROR_INITIALIZATION_FAILED;
+ }
+
+ /* Iterate on the list of planes of the physical device
+ to find a plane that matches these criteria:
+ -It must be compatible with the chosen display + mode.
+ -It isn't currently bound to another display.
+ -It supports per-pixel alpha, if possible. */
+ for (i = 0; i < plane_count; i++) {
+ uint32_t supported_displays_count = 0;
+ VkDisplayKHR* supported_displays;
+ VkDisplayPlaneCapabilitiesKHR plane_caps;
+
+ /* See if the plane is compatible with the current display. */
+ vkGetDisplayPlaneSupportedDisplaysKHR(phys_device, i, &supported_displays_count, NULL);
+ if (supported_displays_count == 0) {
+ /* This plane doesn't support any displays. Continue to the next plane. */
+ continue;
+ }
+
+ /* Get the list of displays supported by this plane. */
+ supported_displays = new VkDisplayKHR[supported_displays_count];
+ vkGetDisplayPlaneSupportedDisplaysKHR(phys_device, i,
+ &supported_displays_count, supported_displays);
+
+ /* The plane must be bound to the chosen display, or not in use.
+ If none of these is true, iterate to another plane. */
+ if ( !( (plane_props[i].currentDisplay == myDisplay) ||
+ (plane_props[i].currentDisplay == VK_NULL_HANDLE)))
+ continue;
+
+ /* Iterate the list of displays supported by this plane
+ in order to find out if the chosen display is among them. */
+ bool plane_supports_display = false;
+ for (j = 0; j < supported_displays_count; j++) {
+ if (supported_displays[j] == myDisplay) {
+ plane_supports_display = true;
+ break;
+ }
+ }
+
+ /* Free the list of displays supported by this plane. */
+ delete [] supported_displays;
+
+ /* If the display is not supported by this plane, iterate to the next plane. */
+ if (!plane_supports_display)
+ continue;
+
+ /* Want a plane that supports the alpha mode we have chosen. */
+ vkGetDisplayPlaneCapabilitiesKHR(phys_device, display_mode, i, &plane_caps);
+ if (plane_caps.supportedAlpha & alpha_mode) {
+ /* Yep, this plane is alright. */
+ plane = i;
+ break;
+ }
+ }
+
+ /* If we couldn't find an appropiate plane, error out. */
+ if (plane == UINT32_MAX) {
+ _assert_msg_(false, "DISPLAY Vulkan couldn't find an appropiate plane");
+ return VK_ERROR_INITIALIZATION_FAILED;
+ }
+
+ // Finally, create the vulkan surface.
+ image_size.width = pixel_xres;
+ image_size.height = pixel_yres;
+
+ display.displayMode = display_mode;
+ display.imageExtent = image_size;
+ display.transform = VK_SURFACE_TRANSFORM_IDENTITY_BIT_KHR;
+ display.alphaMode = alpha_mode;
+ display.globalAlpha = 1.0f;
+ display.planeIndex = plane;
+ display.planeStackIndex = plane_props[plane].currentStackIndex;
+ display.pNext = nullptr;
+ delete [] display_props;
+ delete [] plane_props;
+#endif
display.flags = 0;
retval = vkCreateDisplayPlaneSurfaceKHR(instance_, &display, nullptr, &surface_);
break;
@@ -1107,8 +1290,8 @@ bool VulkanContext::InitSwapchain() {
VkSurfaceTransformFlagBitsKHR preTransform;
std::string supportedTransforms = surface_transforms_to_string(surfCapabilities_.supportedTransforms);
std::string currentTransform = surface_transforms_to_string(surfCapabilities_.currentTransform);
- g_display_rotation = DisplayRotation::ROTATE_0;
- g_display_rot_matrix.setIdentity();
+ g_display.rotation = DisplayRotation::ROTATE_0;
+ g_display.rot_matrix.setIdentity();
uint32_t allowedRotations = VK_SURFACE_TRANSFORM_ROTATE_90_BIT_KHR | VK_SURFACE_TRANSFORM_ROTATE_180_BIT_KHR | VK_SURFACE_TRANSFORM_ROTATE_270_BIT_KHR;
// Hack: Don't allow 270 degrees pretransform (inverse landscape), it creates bizarre issues on some devices (see #15773).
@@ -1119,20 +1302,20 @@ bool VulkanContext::InitSwapchain() {
} else if (surfCapabilities_.currentTransform & allowedRotations) {
// Normal, sensible rotations. Let's handle it.
preTransform = surfCapabilities_.currentTransform;
- g_display_rot_matrix.setIdentity();
+ g_display.rot_matrix.setIdentity();
switch (surfCapabilities_.currentTransform) {
case VK_SURFACE_TRANSFORM_ROTATE_90_BIT_KHR:
- g_display_rotation = DisplayRotation::ROTATE_90;
- g_display_rot_matrix.setRotationZ90();
+ g_display.rotation = DisplayRotation::ROTATE_90;
+ g_display.rot_matrix.setRotationZ90();
std::swap(swapChainExtent_.width, swapChainExtent_.height);
break;
case VK_SURFACE_TRANSFORM_ROTATE_180_BIT_KHR:
- g_display_rotation = DisplayRotation::ROTATE_180;
- g_display_rot_matrix.setRotationZ180();
+ g_display.rotation = DisplayRotation::ROTATE_180;
+ g_display.rot_matrix.setRotationZ180();
break;
case VK_SURFACE_TRANSFORM_ROTATE_270_BIT_KHR:
- g_display_rotation = DisplayRotation::ROTATE_270;
- g_display_rot_matrix.setRotationZ270();
+ g_display.rotation = DisplayRotation::ROTATE_270;
+ g_display.rot_matrix.setRotationZ270();
std::swap(swapChainExtent_.width, swapChainExtent_.height);
break;
default:
diff --git a/Common/GPU/Vulkan/VulkanContext.h b/Common/GPU/Vulkan/VulkanContext.h
index 4cc097b758..49e6eb96c3 100644
--- a/Common/GPU/Vulkan/VulkanContext.h
+++ b/Common/GPU/Vulkan/VulkanContext.h
@@ -222,7 +222,8 @@ public:
// Simple workaround for the casting warning.
template
void SetDebugName(T handle, VkObjectType type, const char *name) {
- if (extensionsLookup_.EXT_debug_utils) {
+ if (extensionsLookup_.EXT_debug_utils && handle != VK_NULL_HANDLE) {
+ _dbg_assert_(handle != VK_NULL_HANDLE);
SetDebugNameImpl((uint64_t)handle, type, name);
}
}
@@ -329,6 +330,7 @@ public:
}
int GetInflightFrames() const {
+ // out of MAX_INFLIGHT_FRAMES.
return inflightFrames_;
}
// Don't call while a frame is in progress.
diff --git a/Common/GPU/Vulkan/VulkanDebug.cpp b/Common/GPU/Vulkan/VulkanDebug.cpp
index 4daf41bdf7..29a71cf61e 100644
--- a/Common/GPU/Vulkan/VulkanDebug.cpp
+++ b/Common/GPU/Vulkan/VulkanDebug.cpp
@@ -62,6 +62,12 @@ VKAPI_ATTR VkBool32 VKAPI_CALL VulkanDebugUtilsCallback(
// See https://github.com/hrydgard/ppsspp/pull/16354
return false;
}
+ if (messageCode == -375211665) {
+ // VUID-vkAllocateMemory-pAllocateInfo-01713
+ // Can happen when VMA aggressively tries to allocate aperture memory for upload. It gracefully
+ // falls back to regular video memory, so we just ignore this. I'd argue this is a VMA bug, actually.
+ return false;
+ }
int count;
{
diff --git a/Common/GPU/Vulkan/VulkanFrameData.cpp b/Common/GPU/Vulkan/VulkanFrameData.cpp
index fcb91c3cc1..90d2c43483 100644
--- a/Common/GPU/Vulkan/VulkanFrameData.cpp
+++ b/Common/GPU/Vulkan/VulkanFrameData.cpp
@@ -4,6 +4,13 @@
#include "Common/Log.h"
#include "Common/StringUtils.h"
+void CachedReadback::Destroy(VulkanContext *vulkan) {
+ if (buffer) {
+ vulkan->Delete().QueueDeleteBufferAllocation(buffer, allocation);
+ }
+ bufferSize = 0;
+}
+
void FrameData::Init(VulkanContext *vulkan, int index) {
this->index = index;
VkDevice device = vulkan->GetDevice();
@@ -36,11 +43,6 @@ void FrameData::Init(VulkanContext *vulkan, int index) {
vulkan->SetDebugName(fence, VK_OBJECT_TYPE_FENCE, StringFromFormat("fence%d", index).c_str());
readyForFence = true;
- // This fence is used for synchronizing readbacks. Does not need preinitialization.
- // TODO: Put this in frameDataShared, only one is needed.
- readbackFence = vulkan->CreateFence(false);
- vulkan->SetDebugName(fence, VK_OBJECT_TYPE_FENCE, "readbackFence");
-
VkQueryPoolCreateInfo query_ci{ VK_STRUCTURE_TYPE_QUERY_POOL_CREATE_INFO };
query_ci.queryCount = MAX_TIMESTAMP_QUERIES;
query_ci.queryType = VK_QUERY_TYPE_TIMESTAMP;
@@ -52,8 +54,13 @@ void FrameData::Destroy(VulkanContext *vulkan) {
vkDestroyCommandPool(device, cmdPoolInit, nullptr);
vkDestroyCommandPool(device, cmdPoolMain, nullptr);
vkDestroyFence(device, fence, nullptr);
- vkDestroyFence(device, readbackFence, nullptr);
vkDestroyQueryPool(device, profile.queryPool, nullptr);
+
+ readbacks_.IterateMut([=](const ReadbackKey &key, CachedReadback *value) {
+ value->Destroy(vulkan);
+ delete value;
+ });
+ readbacks_.Clear();
}
void FrameData::AcquireNextImage(VulkanContext *vulkan, FrameDataShared &shared) {
@@ -144,7 +151,7 @@ void FrameData::SubmitPending(VulkanContext *vulkan, FrameSubmitType type, Frame
}
if ((hasMainCommands || hasPresentCommands) && type == FrameSubmitType::Sync) {
- fenceToTrigger = readbackFence;
+ fenceToTrigger = sharedData.readbackFence;
}
if (hasMainCommands) {
@@ -206,8 +213,8 @@ void FrameData::SubmitPending(VulkanContext *vulkan, FrameSubmitType type, Frame
if (type == FrameSubmitType::Sync) {
// Hard stall of the GPU, not ideal, but necessary so the CPU has the contents of the readback.
- vkWaitForFences(vulkan->GetDevice(), 1, &readbackFence, true, UINT64_MAX);
- vkResetFences(vulkan->GetDevice(), 1, &readbackFence);
+ vkWaitForFences(vulkan->GetDevice(), 1, &sharedData.readbackFence, true, UINT64_MAX);
+ vkResetFences(vulkan->GetDevice(), 1, &sharedData.readbackFence);
syncDone = true;
}
}
@@ -219,10 +226,15 @@ void FrameDataShared::Init(VulkanContext *vulkan) {
_dbg_assert_(res == VK_SUCCESS);
res = vkCreateSemaphore(vulkan->GetDevice(), &semaphoreCreateInfo, nullptr, &renderingCompleteSemaphore);
_dbg_assert_(res == VK_SUCCESS);
+
+ // This fence is used for synchronizing readbacks. Does not need preinitialization.
+ readbackFence = vulkan->CreateFence(false);
+ vulkan->SetDebugName(readbackFence, VK_OBJECT_TYPE_FENCE, "readbackFence");
}
void FrameDataShared::Destroy(VulkanContext *vulkan) {
VkDevice device = vulkan->GetDevice();
vkDestroySemaphore(device, acquireSemaphore, nullptr);
vkDestroySemaphore(device, renderingCompleteSemaphore, nullptr);
+ vkDestroyFence(device, readbackFence, nullptr);
}
diff --git a/Common/GPU/Vulkan/VulkanFrameData.h b/Common/GPU/Vulkan/VulkanFrameData.h
index 88d4c185d2..0e1344f24e 100644
--- a/Common/GPU/Vulkan/VulkanFrameData.h
+++ b/Common/GPU/Vulkan/VulkanFrameData.h
@@ -6,6 +6,7 @@
#include
#include "Common/GPU/Vulkan/VulkanContext.h"
+#include "Common/Data/Collections/Hashmaps.h"
enum {
MAX_TIMESTAMP_QUERIES = 128,
@@ -25,11 +26,31 @@ struct QueueProfileContext {
double cpuEndTime;
};
+class VKRFramebuffer;
+
+struct ReadbackKey {
+ const VKRFramebuffer *framebuf;
+ int width;
+ int height;
+};
+
+struct CachedReadback {
+ VkBuffer buffer;
+ VmaAllocation allocation;
+ VkDeviceSize bufferSize;
+ bool isCoherent;
+
+ void Destroy(VulkanContext *vulkan);
+};
+
struct FrameDataShared {
// Permanent objects
VkSemaphore acquireSemaphore = VK_NULL_HANDLE;
VkSemaphore renderingCompleteSemaphore = VK_NULL_HANDLE;
+ // For synchronous readbacks.
+ VkFence readbackFence = VK_NULL_HANDLE;
+
void Init(VulkanContext *vulkan);
void Destroy(VulkanContext *vulkan);
};
@@ -49,7 +70,6 @@ struct FrameData {
bool readyForFence = true;
VkFence fence = VK_NULL_HANDLE;
- VkFence readbackFence = VK_NULL_HANDLE; // Strictly speaking we might only need one global of these.
// These are on different threads so need separate pools.
VkCommandPool cmdPoolInit = VK_NULL_HANDLE; // Written to from main thread
@@ -75,6 +95,11 @@ struct FrameData {
QueueProfileContext profile;
bool profilingEnabled_ = false;
+ // Async readback cache.
+ DenseHashMap readbacks_;
+
+ FrameData() : readbacks_(8) {}
+
void Init(VulkanContext *vulkan, int index);
void Destroy(VulkanContext *vulkan);
@@ -89,5 +114,5 @@ struct FrameData {
private:
// Metadata for logging etc
- int index;
+ int index = -1;
};
diff --git a/Common/GPU/Vulkan/VulkanImage.cpp b/Common/GPU/Vulkan/VulkanImage.cpp
index 8efcb69834..d40e422ff4 100644
--- a/Common/GPU/Vulkan/VulkanImage.cpp
+++ b/Common/GPU/Vulkan/VulkanImage.cpp
@@ -142,8 +142,7 @@ bool VulkanTexture::CreateDirect(VkCommandBuffer cmd, int w, int h, int depth, i
return true;
}
-// TODO: Batch these.
-void VulkanTexture::UploadMip(VkCommandBuffer cmd, int mip, int mipWidth, int mipHeight, int depthLayer, VkBuffer buffer, uint32_t offset, size_t rowLength) {
+void VulkanTexture::CopyBufferToMipLevel(VkCommandBuffer cmd, TextureCopyBatch *copyBatch, int mip, int mipWidth, int mipHeight, int depthLayer, VkBuffer buffer, uint32_t offset, size_t rowLength) {
VkBufferImageCopy copy_region{};
copy_region.bufferOffset = offset;
copy_region.bufferRowLength = (uint32_t)rowLength;
@@ -157,7 +156,23 @@ void VulkanTexture::UploadMip(VkCommandBuffer cmd, int mip, int mipWidth, int mi
copy_region.imageSubresource.baseArrayLayer = 0;
copy_region.imageSubresource.layerCount = 1;
- vkCmdCopyBufferToImage(cmd, buffer, image_, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, 1, ©_region);
+ _dbg_assert_(mip < numMips_);
+
+ if (!copyBatch->buffer) {
+ copyBatch->buffer = buffer;
+ } else if (copyBatch->buffer != buffer) {
+ // Need to flush the batch if this image isn't from the same buffer as the previous ones.
+ FinishCopyBatch(cmd, copyBatch);
+ copyBatch->buffer = buffer;
+ }
+ copyBatch->copies.push_back(copy_region);
+}
+
+void VulkanTexture::FinishCopyBatch(VkCommandBuffer cmd, TextureCopyBatch *copyBatch) {
+ if (!copyBatch->copies.empty()) {
+ vkCmdCopyBufferToImage(cmd, copyBatch->buffer, image_, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, (uint32_t)copyBatch->copies.size(), copyBatch->copies.data());
+ copyBatch->copies.clear();
+ }
}
void VulkanTexture::ClearMip(VkCommandBuffer cmd, int mip, uint32_t value) {
diff --git a/Common/GPU/Vulkan/VulkanImage.h b/Common/GPU/Vulkan/VulkanImage.h
index c088fdca5b..580e663713 100644
--- a/Common/GPU/Vulkan/VulkanImage.h
+++ b/Common/GPU/Vulkan/VulkanImage.h
@@ -7,6 +7,13 @@ class VulkanDeviceAllocator;
VK_DEFINE_HANDLE(VmaAllocation);
+struct TextureCopyBatch {
+ std::vector copies;
+ VkBuffer buffer = VK_NULL_HANDLE;
+ void reserve(size_t mips) { copies.reserve(mips); }
+ bool empty() const { return copies.empty(); }
+};
+
// Wrapper around what you need to use a texture.
// ALWAYS use an allocator when calling CreateDirect.
class VulkanTexture {
@@ -23,7 +30,9 @@ public:
void ClearMip(VkCommandBuffer cmd, int mip, uint32_t value);
// Can also be used to copy individual levels of a 3D texture.
- void UploadMip(VkCommandBuffer cmd, int mip, int mipWidth, int mipHeight, int depthLayer, VkBuffer buffer, uint32_t offset, size_t rowLength); // rowLength is in pixels
+ // If possible, will just add to the batch instead of submitting a copy.
+ void CopyBufferToMipLevel(VkCommandBuffer cmd, TextureCopyBatch *copyBatch, int mip, int mipWidth, int mipHeight, int depthLayer, VkBuffer buffer, uint32_t offset, size_t rowLength); // rowLength is in pixels
+ void FinishCopyBatch(VkCommandBuffer cmd, TextureCopyBatch *copyBatch);
void GenerateMips(VkCommandBuffer cmd, int firstMipToGenerate, bool fromCompute);
void EndCreate(VkCommandBuffer cmd, bool vertexTexture, VkPipelineStageFlags prevStage, VkImageLayout layout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL);
diff --git a/Common/GPU/Vulkan/VulkanLoader.cpp b/Common/GPU/Vulkan/VulkanLoader.cpp
index 94092b2166..5dd7baf4f6 100644
--- a/Common/GPU/Vulkan/VulkanLoader.cpp
+++ b/Common/GPU/Vulkan/VulkanLoader.cpp
@@ -195,6 +195,11 @@ PFN_vkCreateWaylandSurfaceKHR vkCreateWaylandSurfaceKHR;
#endif
#if defined(VK_USE_PLATFORM_DISPLAY_KHR)
PFN_vkCreateDisplayPlaneSurfaceKHR vkCreateDisplayPlaneSurfaceKHR;
+PFN_vkGetPhysicalDeviceDisplayPropertiesKHR vkGetPhysicalDeviceDisplayPropertiesKHR;
+PFN_vkGetPhysicalDeviceDisplayPlanePropertiesKHR vkGetPhysicalDeviceDisplayPlanePropertiesKHR;
+PFN_vkGetDisplayModePropertiesKHR vkGetDisplayModePropertiesKHR;
+PFN_vkGetDisplayPlaneSupportedDisplaysKHR vkGetDisplayPlaneSupportedDisplaysKHR;
+PFN_vkGetDisplayPlaneCapabilitiesKHR vkGetDisplayPlaneCapabilitiesKHR;
#endif
PFN_vkDestroySurfaceKHR vkDestroySurfaceKHR;
@@ -305,6 +310,7 @@ static void VulkanFreeLibrary(VulkanLibraryHandle &h) {
}
void VulkanSetAvailable(bool available) {
+ INFO_LOG(G3D, "Forcing Vulkan availability to true");
g_vulkanAvailabilityChecked = true;
g_vulkanMayBeAvailable = available;
}
@@ -465,6 +471,7 @@ bool VulkanMayBeAvailable() {
case VK_PHYSICAL_DEVICE_TYPE_INTEGRATED_GPU:
case VK_PHYSICAL_DEVICE_TYPE_VIRTUAL_GPU:
anyGood = true;
+ INFO_LOG(G3D, "VulkanMayBeAvailable: Eligible device found: '%s'", props.deviceName);
break;
default:
INFO_LOG(G3D, "VulkanMayBeAvailable: Ineligible device found and ignored: '%s'", props.deviceName);
@@ -489,8 +496,9 @@ bail:
}
if (lib) {
VulkanFreeLibrary(lib);
- } else {
- ERROR_LOG(G3D, "Vulkan with working device not detected.");
+ }
+ if (!g_vulkanMayBeAvailable) {
+ WARN_LOG(G3D, "Vulkan with working device not detected.");
}
return g_vulkanMayBeAvailable;
}
@@ -564,6 +572,11 @@ void VulkanLoadInstanceFunctions(VkInstance instance, const VulkanExtensions &en
#endif
#if defined(VK_USE_PLATFORM_DISPLAY_KHR)
LOAD_INSTANCE_FUNC(instance, vkCreateDisplayPlaneSurfaceKHR);
+ LOAD_INSTANCE_FUNC(instance, vkGetPhysicalDeviceDisplayPropertiesKHR);
+ LOAD_INSTANCE_FUNC(instance, vkGetPhysicalDeviceDisplayPlanePropertiesKHR);
+ LOAD_INSTANCE_FUNC(instance, vkGetDisplayModePropertiesKHR);
+ LOAD_INSTANCE_FUNC(instance, vkGetDisplayPlaneSupportedDisplaysKHR);
+ LOAD_INSTANCE_FUNC(instance, vkGetDisplayPlaneCapabilitiesKHR);
#endif
LOAD_INSTANCE_FUNC(instance, vkDestroySurfaceKHR);
diff --git a/Common/GPU/Vulkan/VulkanLoader.h b/Common/GPU/Vulkan/VulkanLoader.h
index caaef0d597..f85a09f558 100644
--- a/Common/GPU/Vulkan/VulkanLoader.h
+++ b/Common/GPU/Vulkan/VulkanLoader.h
@@ -33,6 +33,11 @@
#include "ext/vulkan/vulkan.h"
+// Hacky X11 header workaround
+#ifdef Opposite
+#undef Opposite
+#endif
+
namespace PPSSPP_VK {
// Putting our own Vulkan function pointers in a namespace ensures that ppsspp_libretro.so doesn't collide with libvulkan.so.
extern PFN_vkCreateInstance vkCreateInstance;
@@ -194,6 +199,11 @@ extern PFN_vkCreateWaylandSurfaceKHR vkCreateWaylandSurfaceKHR;
#endif
#if defined(VK_USE_PLATFORM_DISPLAY_KHR)
extern PFN_vkCreateDisplayPlaneSurfaceKHR vkCreateDisplayPlaneSurfaceKHR;
+extern PFN_vkGetPhysicalDeviceDisplayPropertiesKHR vkGetPhysicalDeviceDisplayPropertiesKHR;
+extern PFN_vkGetPhysicalDeviceDisplayPlanePropertiesKHR vkGetPhysicalDeviceDisplayPlanePropertiesKHR;
+extern PFN_vkGetDisplayModePropertiesKHR vkGetDisplayModePropertiesKHR;
+extern PFN_vkGetDisplayPlaneSupportedDisplaysKHR vkGetDisplayPlaneSupportedDisplaysKHR;
+extern PFN_vkGetDisplayPlaneCapabilitiesKHR vkGetDisplayPlaneCapabilitiesKHR;
#endif
extern PFN_vkDestroySurfaceKHR vkDestroySurfaceKHR;
diff --git a/Common/GPU/Vulkan/VulkanMemory.cpp b/Common/GPU/Vulkan/VulkanMemory.cpp
index d8184098a5..3d65980912 100644
--- a/Common/GPU/Vulkan/VulkanMemory.cpp
+++ b/Common/GPU/Vulkan/VulkanMemory.cpp
@@ -18,21 +18,54 @@
// Additionally, Common/Vulkan/* , including this file, are also licensed
// under the public domain.
+#include
+#include
+#include
+
#include "Common/Math/math_util.h"
#include "Common/Log.h"
#include "Common/TimeUtil.h"
+#include "Common/Math/math_util.h"
#include "Common/GPU/Vulkan/VulkanMemory.h"
+#include "Common/Data/Text/Parsers.h"
using namespace PPSSPP_VK;
-VulkanPushBuffer::VulkanPushBuffer(VulkanContext *vulkan, const char *name, size_t size, VkBufferUsageFlags usage, PushBufferType type)
- : vulkan_(vulkan), name_(name), size_(size), usage_(usage), type_(type) {
+// Always keep around push buffers at least this long (seconds).
+static const double PUSH_GARBAGE_COLLECTION_DELAY = 10.0;
+
+// Global push buffer tracker for vulkan memory profiling.
+// Don't want to manually dig up all the active push buffers.
+static std::mutex g_pushBufferListMutex;
+static std::set g_pushBuffers;
+
+std::vector GetActiveVulkanMemoryManagers() {
+ std::vector buffers;
+ std::lock_guard guard(g_pushBufferListMutex);
+ for (auto iter : g_pushBuffers) {
+ buffers.push_back(iter);
+ }
+ return buffers;
+}
+
+VulkanPushBuffer::VulkanPushBuffer(VulkanContext *vulkan, const char *name, size_t size, VkBufferUsageFlags usage)
+ : vulkan_(vulkan), name_(name), size_(size), usage_(usage) {
+ {
+ std::lock_guard guard(g_pushBufferListMutex);
+ g_pushBuffers.insert(this);
+ }
+
bool res = AddBuffer();
_assert_(res);
}
VulkanPushBuffer::~VulkanPushBuffer() {
+ {
+ std::lock_guard guard(g_pushBufferListMutex);
+ g_pushBuffers.erase(this);
+ }
+
_dbg_assert_(!writePtr_);
_assert_(buffers_.empty());
}
@@ -50,7 +83,7 @@ bool VulkanPushBuffer::AddBuffer() {
b.pQueueFamilyIndices = nullptr;
VmaAllocationCreateInfo allocCreateInfo{};
- allocCreateInfo.usage = type_ == PushBufferType::CPU_TO_GPU ? VMA_MEMORY_USAGE_CPU_TO_GPU : VMA_MEMORY_USAGE_GPU_ONLY;
+ allocCreateInfo.usage = VMA_MEMORY_USAGE_CPU_TO_GPU;
VmaAllocationInfo allocInfo{};
VkResult res = vmaCreateBuffer(vulkan_->Allocator(), &b, &allocCreateInfo, &info.buffer, &info.allocation, &allocInfo);
@@ -76,8 +109,7 @@ void VulkanPushBuffer::Destroy(VulkanContext *vulkan) {
void VulkanPushBuffer::NextBuffer(size_t minSize) {
// First, unmap the current memory.
- if (type_ == PushBufferType::CPU_TO_GPU)
- Unmap();
+ Unmap();
buf_++;
if (buf_ >= buffers_.size() || minSize > size_) {
@@ -96,8 +128,7 @@ void VulkanPushBuffer::NextBuffer(size_t minSize) {
// Now, move to the next buffer and map it.
offset_ = 0;
- if (type_ == PushBufferType::CPU_TO_GPU)
- Map();
+ Map();
}
void VulkanPushBuffer::Defragment(VulkanContext *vulkan) {
@@ -122,6 +153,15 @@ size_t VulkanPushBuffer::GetTotalSize() const {
return sum;
}
+void VulkanPushBuffer::GetDebugString(char *buffer, size_t bufSize) const {
+ size_t sum = 0;
+ if (buffers_.size() > 1)
+ sum += size_ * (buffers_.size() - 1);
+ sum += offset_;
+ size_t capacity = size_ * buffers_.size();
+ snprintf(buffer, bufSize, "Push %s: %s / %s", name_, NiceSizeFormat(sum).c_str(), NiceSizeFormat(capacity).c_str());
+}
+
void VulkanPushBuffer::Map() {
_dbg_assert_(!writePtr_);
VkResult res = vmaMapMemory(vulkan_->Allocator(), buffers_[buf_].allocation, (void **)(&writePtr_));
@@ -233,3 +273,148 @@ VkResult VulkanDescSetPool::Recreate(bool grow) {
}
return result;
}
+
+VulkanPushPool::VulkanPushPool(VulkanContext *vulkan, const char *name, size_t originalBlockSize, VkBufferUsageFlags usage)
+ : vulkan_(vulkan), name_(name), originalBlockSize_(originalBlockSize), usage_(usage) {
+ {
+ std::lock_guard guard(g_pushBufferListMutex);
+ g_pushBuffers.insert(this);
+ }
+
+ for (int i = 0; i < VulkanContext::MAX_INFLIGHT_FRAMES; i++) {
+ blocks_.push_back(CreateBlock(originalBlockSize));
+ blocks_.back().original = true;
+ blocks_.back().frameIndex = i;
+ }
+}
+
+VulkanPushPool::~VulkanPushPool() {
+ {
+ std::lock_guard guard(g_pushBufferListMutex);
+ g_pushBuffers.erase(this);
+ }
+
+ _dbg_assert_(blocks_.empty());
+}
+
+void VulkanPushPool::Destroy() {
+ for (auto &block : blocks_) {
+ block.Destroy(vulkan_);
+ }
+ blocks_.clear();
+}
+
+VulkanPushPool::Block VulkanPushPool::CreateBlock(size_t size) {
+ Block block{};
+ block.size = size;
+ block.frameIndex = -1;
+
+ VkBufferCreateInfo b{ VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO };
+ b.size = size;
+ b.usage = usage_;
+ b.sharingMode = VK_SHARING_MODE_EXCLUSIVE;
+ VmaAllocationCreateInfo allocCreateInfo{};
+ allocCreateInfo.usage = VMA_MEMORY_USAGE_CPU_TO_GPU;
+ VmaAllocationInfo allocInfo{};
+
+ VkResult result = vmaCreateBuffer(vulkan_->Allocator(), &b, &allocCreateInfo, &block.buffer, &block.allocation, &allocInfo);
+ _assert_(result == VK_SUCCESS);
+
+ result = vmaMapMemory(vulkan_->Allocator(), block.allocation, (void **)(&block.writePtr));
+ _assert_(result == VK_SUCCESS);
+
+ return block;
+}
+
+VulkanPushPool::Block::~Block() {}
+
+void VulkanPushPool::Block::Destroy(VulkanContext *vulkan) {
+ vmaUnmapMemory(vulkan->Allocator(), allocation);
+ vulkan->Delete().QueueDeleteBufferAllocation(buffer, allocation);
+}
+
+void VulkanPushPool::BeginFrame() {
+ double now = time_now_d();
+ curBlockIndex_ = -1;
+ for (auto &block : blocks_) {
+ if (block.frameIndex == vulkan_->GetCurFrame()) {
+ if (curBlockIndex_ == -1) {
+ // Pick a block associated with the current frame to start at.
+ // We always start with one block per frame index.
+ curBlockIndex_ = block.frameIndex;
+ block.lastUsed = now;
+ }
+ block.used = 0;
+ if (!block.original) {
+ // Return block to the common pool
+ block.frameIndex = -1;
+ }
+ }
+ }
+
+ // Do a single pass of bubblesort to move the bigger buffers earlier in the sequence.
+ // Over multiple frames this will quickly converge to the right order.
+ for (size_t i = 3; i < blocks_.size() - 1; i++) {
+ if (blocks_[i].frameIndex == -1 && blocks_[i + 1].frameIndex == -1 && blocks_[i].size < blocks_[i + 1].size) {
+ std::swap(blocks_[i], blocks_[i + 1]);
+ }
+ }
+
+ // If we have lots of little buffers and the last one hasn't been used in a while, drop it.
+ // Still, let's keep around a few big ones (6 - 3).
+ if (blocks_.size() > 6 && blocks_.back().lastUsed < now - PUSH_GARBAGE_COLLECTION_DELAY) {
+ double start = time_now_d();
+ size_t size = blocks_.back().size;
+ blocks_.back().Destroy(vulkan_);
+ blocks_.pop_back();
+ DEBUG_LOG(G3D, "%s: Garbage collected block of size %s in %0.2f ms", name_, NiceSizeFormat(size).c_str(), time_now_d() - start);
+ }
+}
+
+void VulkanPushPool::NextBlock(VkDeviceSize allocationSize) {
+ int curFrameIndex = vulkan_->GetCurFrame();
+ curBlockIndex_++;
+ while (curBlockIndex_ < blocks_.size()) {
+ Block &block = blocks_[curBlockIndex_];
+ // Grab the first matching block, or unused block (frameIndex == -1).
+ if ((block.frameIndex == curFrameIndex || block.frameIndex == -1) && block.size >= allocationSize) {
+ _assert_(block.used == 0);
+ block.used = allocationSize;
+ block.lastUsed = time_now_d();
+ block.frameIndex = curFrameIndex;
+ return;
+ }
+ curBlockIndex_++;
+ }
+
+ double start = time_now_d();
+ VkDeviceSize newBlockSize = std::max(originalBlockSize_ * 2, (VkDeviceSize)RoundUpToPowerOf2((uint32_t)allocationSize));
+ // We're still here and ran off the end of blocks. Create a new one.
+ blocks_.push_back(CreateBlock(newBlockSize));
+ blocks_.back().frameIndex = curFrameIndex;
+ blocks_.back().used = allocationSize;
+ blocks_.back().lastUsed = time_now_d();
+ // curBlockIndex_ is already set correctly here.
+ DEBUG_LOG(G3D, "%s: Created new block of size %s in %0.2f ms", name_, NiceSizeFormat(newBlockSize).c_str(), 1000.0 * (time_now_d() - start));
+}
+
+size_t VulkanPushPool::GetUsedThisFrame() const {
+ size_t used = 0;
+ for (auto &block : blocks_) {
+ if (block.frameIndex == vulkan_->GetCurFrame()) {
+ used += block.used;
+ }
+ }
+ return used;
+}
+
+void VulkanPushPool::GetDebugString(char *buffer, size_t bufSize) const {
+ size_t used = 0;
+ size_t capacity = 0;
+ for (auto &block : blocks_) {
+ used += block.used;
+ capacity += block.size;
+ }
+
+ snprintf(buffer, bufSize, "Pool %s: %s / %s (%d extra blocks)", name_, NiceSizeFormat(used).c_str(), NiceSizeFormat(capacity).c_str(), (int)blocks_.size() - 3);
+}
diff --git a/Common/GPU/Vulkan/VulkanMemory.h b/Common/GPU/Vulkan/VulkanMemory.h
index 275ca4bb78..f4d502e138 100644
--- a/Common/GPU/Vulkan/VulkanMemory.h
+++ b/Common/GPU/Vulkan/VulkanMemory.h
@@ -14,19 +14,22 @@ VK_DEFINE_HANDLE(VmaAllocation);
//
// Vulkan memory management utils.
-enum class PushBufferType {
- CPU_TO_GPU,
- GPU_ONLY,
+// Just an abstract thing to get debug information.
+class VulkanMemoryManager {
+public:
+ virtual ~VulkanMemoryManager() {}
+
+ virtual void GetDebugString(char *buffer, size_t bufSize) const = 0;
+ virtual const char *Name() const = 0; // for sorting
};
// VulkanPushBuffer
// Simple incrementing allocator.
-// Use these to push vertex, index and uniform data. Generally you'll have two of these
+// Use these to push vertex, index and uniform data. Generally you'll have two or three of these
// and alternate on each frame. Make sure not to reset until the fence from the last time you used it
// has completed.
-//
-// TODO: Make it possible to suballocate pushbuffers from a large DeviceMemory block.
-class VulkanPushBuffer {
+// NOTE: This has now been replaced with VulkanPushPool for all uses except the vertex cache.
+class VulkanPushBuffer : public VulkanMemoryManager {
struct BufInfo {
VkBuffer buffer;
VmaAllocation allocation;
@@ -36,101 +39,54 @@ public:
// NOTE: If you create a push buffer with PushBufferType::GPU_ONLY,
// then you can't use any of the push functions as pointers will not be reachable from the CPU.
// You must in this case use Allocate() only, and pass the returned offset and the VkBuffer to Vulkan APIs.
- VulkanPushBuffer(VulkanContext *vulkan, const char *name, size_t size, VkBufferUsageFlags usage, PushBufferType type);
+ VulkanPushBuffer(VulkanContext *vulkan, const char *name, size_t size, VkBufferUsageFlags usage);
~VulkanPushBuffer();
void Destroy(VulkanContext *vulkan);
void Reset() { offset_ = 0; }
+ void GetDebugString(char *buffer, size_t bufSize) const override;
+ const char *Name() const override {
+ return name_;
+ }
+
// Needs context in case of defragment.
void Begin(VulkanContext *vulkan) {
buf_ = 0;
offset_ = 0;
// Note: we must defrag because some buffers may be smaller than size_.
Defragment(vulkan);
- if (type_ == PushBufferType::CPU_TO_GPU)
- Map();
+ Map();
}
- void BeginNoReset() {
- if (type_ == PushBufferType::CPU_TO_GPU)
- Map();
- }
-
- void End() {
- if (type_ == PushBufferType::CPU_TO_GPU)
- Unmap();
- }
+ void BeginNoReset() { Map(); }
+ void End() { Unmap(); }
void Map();
-
void Unmap();
// When using the returned memory, make sure to bind the returned vkbuf.
- // This will later allow for handling overflow correctly.
- size_t Allocate(size_t numBytes, VkBuffer *vkbuf) {
- size_t out = offset_;
- offset_ += (numBytes + 3) & ~3; // Round up to 4 bytes.
-
- if (offset_ >= size_) {
+ uint8_t *Allocate(VkDeviceSize numBytes, VkDeviceSize alignment, VkBuffer *vkbuf, uint32_t *bindOffset) {
+ size_t offset = (offset_ + alignment - 1) & ~(alignment - 1);
+ if (offset + numBytes > size_) {
NextBuffer(numBytes);
- out = offset_;
- offset_ += (numBytes + 3) & ~3;
+ offset = offset_;
}
+ offset_ = offset + numBytes;
+ *bindOffset = (uint32_t)offset;
*vkbuf = buffers_[buf_].buffer;
- return out;
+ return writePtr_ + offset;
}
- // Returns the offset that should be used when binding this buffer to get this data.
- size_t Push(const void *data, size_t size, VkBuffer *vkbuf) {
- _dbg_assert_(writePtr_);
- size_t off = Allocate(size, vkbuf);
- memcpy(writePtr_ + off, data, size);
- return off;
- }
-
- uint32_t PushAligned(const void *data, size_t size, int align, VkBuffer *vkbuf) {
- _dbg_assert_(writePtr_);
- offset_ = (offset_ + align - 1) & ~(align - 1);
- size_t off = Allocate(size, vkbuf);
- memcpy(writePtr_ + off, data, size);
- return (uint32_t)off;
- }
-
- size_t GetOffset() const {
- return offset_;
- }
-
- const char *Name() const {
- return name_;
- }
-
- // "Zero-copy" variant - you can write the data directly as you compute it.
- // Recommended.
- void *Push(size_t size, uint32_t *bindOffset, VkBuffer *vkbuf) {
- _dbg_assert_(writePtr_);
- size_t off = Allocate(size, vkbuf);
- *bindOffset = (uint32_t)off;
- return writePtr_ + off;
- }
- void *PushAligned(size_t size, uint32_t *bindOffset, VkBuffer *vkbuf, int align) {
- _dbg_assert_(writePtr_);
- offset_ = (offset_ + align - 1) & ~(align - 1);
- size_t off = Allocate(size, vkbuf);
- *bindOffset = (uint32_t)off;
- return writePtr_ + off;
- }
-
- template
- void PushUBOData(const T &data, VkDescriptorBufferInfo *info) {
+ VkDeviceSize Push(const void *data, VkDeviceSize numBytes, int alignment, VkBuffer *vkbuf) {
uint32_t bindOffset;
- void *ptr = PushAligned(sizeof(T), &bindOffset, &info->buffer, vulkan_->GetPhysicalDeviceProperties().properties.limits.minUniformBufferOffsetAlignment);
- memcpy(ptr, &data, sizeof(T));
- info->offset = bindOffset;
- info->range = sizeof(T);
+ uint8_t *ptr = Allocate(numBytes, alignment, vkbuf, &bindOffset);
+ memcpy(ptr, data, numBytes);
+ return bindOffset;
}
+ size_t GetOffset() const { return offset_; }
size_t GetTotalSize() const;
private:
@@ -139,7 +95,6 @@ private:
void Defragment(VulkanContext *vulkan);
VulkanContext *vulkan_;
- PushBufferType type_;
std::vector buffers_;
size_t buf_ = 0;
@@ -150,6 +105,81 @@ private:
const char *name_;
};
+// Simple memory pushbuffer pool that can share blocks between the "frames", to reduce the impact of push memory spikes -
+// a later frame can gobble up redundant buffers from an earlier frame even if they don't share frame index.
+class VulkanPushPool : public VulkanMemoryManager {
+public:
+ VulkanPushPool(VulkanContext *vulkan, const char *name, size_t originalBlockSize, VkBufferUsageFlags usage);
+ ~VulkanPushPool();
+
+ void Destroy();
+ void BeginFrame();
+
+ const char *Name() const override {
+ return name_;
+ }
+ void GetDebugString(char *buffer, size_t bufSize) const override;
+
+ // When using the returned memory, make sure to bind the returned vkbuf.
+ uint8_t *Allocate(VkDeviceSize numBytes, VkDeviceSize alignment, VkBuffer *vkbuf, uint32_t *bindOffset) {
+ _dbg_assert_(curBlockIndex_ >= 0);
+
+ Block &block = blocks_[curBlockIndex_];
+
+ VkDeviceSize offset = (block.used + (alignment - 1)) & ~(alignment - 1);
+ if (offset + numBytes <= block.size) {
+ block.used = offset + numBytes;
+ *vkbuf = block.buffer;
+ *bindOffset = (uint32_t)offset;
+ return block.writePtr + offset;
+ }
+
+ NextBlock(numBytes);
+
+ *vkbuf = blocks_[curBlockIndex_].buffer;
+ *bindOffset = 0; // Newly allocated buffer will start at 0.
+ return blocks_[curBlockIndex_].writePtr;
+ }
+
+ VkDeviceSize Push(const void *data, VkDeviceSize numBytes, int alignment, VkBuffer *vkbuf) {
+ uint32_t bindOffset;
+ uint8_t *ptr = Allocate(numBytes, alignment, vkbuf, &bindOffset);
+ memcpy(ptr, data, numBytes);
+ return bindOffset;
+ }
+
+ size_t GetUsedThisFrame() const;
+
+private:
+ void NextBlock(VkDeviceSize allocationSize);
+
+ struct Block {
+ ~Block();
+ VkBuffer buffer;
+ VmaAllocation allocation;
+
+ VkDeviceSize size;
+ VkDeviceSize used;
+
+ int frameIndex;
+ bool original; // these blocks aren't garbage collected.
+ double lastUsed;
+
+ uint8_t *writePtr;
+
+ void Destroy(VulkanContext *vulkan);
+ };
+
+ Block CreateBlock(size_t sz);
+
+ VulkanContext *vulkan_;
+ VkDeviceSize originalBlockSize_;
+ std::vector blocks_;
+ VkBufferUsageFlags usage_;
+ int curBlockIndex_ = -1;
+ const char *name_;
+};
+
// Only appropriate for use in a per-frame pool.
class VulkanDescSetPool {
public:
@@ -173,9 +203,12 @@ private:
const char *tag_;
VulkanContext *vulkan_ = nullptr;
VkDescriptorPool descPool_ = VK_NULL_HANDLE;
- VkDescriptorPoolCreateInfo info_;
+ VkDescriptorPoolCreateInfo info_{};
std::vector sizes_;
std::function clear_;
uint32_t usage_ = 0;
bool grow_;
};
+
+std::vector GetActiveVulkanMemoryManagers();
+
diff --git a/Common/GPU/Vulkan/VulkanQueueRunner.cpp b/Common/GPU/Vulkan/VulkanQueueRunner.cpp
index 137703bb0d..3b2bbdd9be 100644
--- a/Common/GPU/Vulkan/VulkanQueueRunner.cpp
+++ b/Common/GPU/Vulkan/VulkanQueueRunner.cpp
@@ -67,72 +67,10 @@ void VulkanQueueRunner::CreateDeviceObjects() {
#endif
}
-void VulkanQueueRunner::ResizeReadbackBuffer(VkDeviceSize requiredSize) {
- if (readbackBuffer_ && requiredSize <= readbackBufferSize_) {
- return;
- }
- if (readbackMemory_) {
- vulkan_->Delete().QueueDeleteDeviceMemory(readbackMemory_);
- }
- if (readbackBuffer_) {
- vulkan_->Delete().QueueDeleteBuffer(readbackBuffer_);
- }
-
- readbackBufferSize_ = requiredSize;
-
- VkDevice device = vulkan_->GetDevice();
-
- VkBufferCreateInfo buf{ VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO };
- buf.size = readbackBufferSize_;
- buf.usage = VK_BUFFER_USAGE_TRANSFER_DST_BIT;
-
- VkResult res = vkCreateBuffer(device, &buf, nullptr, &readbackBuffer_);
- _assert_(res == VK_SUCCESS);
-
- VkMemoryRequirements reqs{};
- vkGetBufferMemoryRequirements(device, readbackBuffer_, &reqs);
-
- VkMemoryAllocateInfo allocInfo{ VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO };
- allocInfo.allocationSize = reqs.size;
-
- // For speedy readbacks, we want the CPU cache to be enabled. However on most hardware we then have to
- // sacrifice coherency, which means manual flushing. But try to find such memory first! If no cached
- // memory type is available we fall back to just coherent.
- const VkFlags desiredTypes[] = {
- VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT | VK_MEMORY_PROPERTY_HOST_CACHED_BIT,
- VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_CACHED_BIT,
- VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT,
- };
- VkFlags successTypeReqs = 0;
- for (VkFlags typeReqs : desiredTypes) {
- if (vulkan_->MemoryTypeFromProperties(reqs.memoryTypeBits, typeReqs, &allocInfo.memoryTypeIndex)) {
- successTypeReqs = typeReqs;
- break;
- }
- }
- _assert_(successTypeReqs != 0);
- readbackBufferIsCoherent_ = (successTypeReqs & VK_MEMORY_PROPERTY_HOST_COHERENT_BIT) != 0;
-
- res = vkAllocateMemory(device, &allocInfo, nullptr, &readbackMemory_);
- if (res != VK_SUCCESS) {
- readbackMemory_ = VK_NULL_HANDLE;
- vkDestroyBuffer(device, readbackBuffer_, nullptr);
- readbackBuffer_ = VK_NULL_HANDLE;
- return;
- }
- uint32_t offset = 0;
- vkBindBufferMemory(device, readbackBuffer_, readbackMemory_, offset);
-}
-
void VulkanQueueRunner::DestroyDeviceObjects() {
INFO_LOG(G3D, "VulkanQueueRunner::DestroyDeviceObjects");
- if (readbackMemory_) {
- vulkan_->Delete().QueueDeleteDeviceMemory(readbackMemory_);
- }
- if (readbackBuffer_) {
- vulkan_->Delete().QueueDeleteBuffer(readbackBuffer_);
- }
- readbackBufferSize_ = 0;
+
+ syncReadback_.Destroy(vulkan_);
renderPasses_.IterateMut([&](const RPKey &rpkey, VKRRenderPass *rp) {
_assert_(rp);
@@ -482,7 +420,7 @@ void VulkanQueueRunner::RunSteps(std::vector &steps, FrameData &frame
PerformBlit(step, cmd);
break;
case VKRStepType::READBACK:
- PerformReadback(step, cmd);
+ PerformReadback(step, cmd, frameData);
break;
case VKRStepType::READBACK_IMAGE:
PerformReadbackImage(step, cmd);
@@ -2007,18 +1945,35 @@ void VulkanQueueRunner::SetupTransferDstWriteAfterWrite(VKRImage &img, VkImageAs
);
}
-void VulkanQueueRunner::PerformReadback(const VKRStep &step, VkCommandBuffer cmd) {
- ResizeReadbackBuffer(sizeof(uint32_t) * step.readback.srcRect.extent.width * step.readback.srcRect.extent.height);
+void VulkanQueueRunner::ResizeReadbackBuffer(CachedReadback *readback, VkDeviceSize requiredSize) {
+ if (readback->buffer && requiredSize <= readback->bufferSize) {
+ return;
+ }
- VkBufferImageCopy region{};
- region.imageOffset = { step.readback.srcRect.offset.x, step.readback.srcRect.offset.y, 0 };
- region.imageExtent = { step.readback.srcRect.extent.width, step.readback.srcRect.extent.height, 1 };
- region.imageSubresource.aspectMask = step.readback.aspectMask;
- region.imageSubresource.layerCount = 1;
- region.bufferOffset = 0;
- region.bufferRowLength = step.readback.srcRect.extent.width;
- region.bufferImageHeight = step.readback.srcRect.extent.height;
+ if (readback->buffer) {
+ vulkan_->Delete().QueueDeleteBufferAllocation(readback->buffer, readback->allocation);
+ }
+ readback->bufferSize = requiredSize;
+
+ VkDevice device = vulkan_->GetDevice();
+
+ VkBufferCreateInfo buf{ VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO };
+ buf.size = readback->bufferSize;
+ buf.usage = VK_BUFFER_USAGE_TRANSFER_DST_BIT;
+
+ VmaAllocationCreateInfo allocCreateInfo{};
+ allocCreateInfo.usage = VMA_MEMORY_USAGE_GPU_TO_CPU;
+ VmaAllocationInfo allocInfo{};
+
+ VkResult res = vmaCreateBuffer(vulkan_->Allocator(), &buf, &allocCreateInfo, &readback->buffer, &readback->allocation, &allocInfo);
+ _assert_(res == VK_SUCCESS);
+
+ const VkMemoryType &memoryType = vulkan_->GetMemoryProperties().memoryTypes[allocInfo.memoryType];
+ readback->isCoherent = (memoryType.propertyFlags & VK_MEMORY_PROPERTY_HOST_COHERENT_BIT) != 0;
+}
+
+void VulkanQueueRunner::PerformReadback(const VKRStep &step, VkCommandBuffer cmd, FrameData &frameData) {
VkImage image;
VkImageLayout copyLayout;
// Special case for backbuffer readbacks.
@@ -2052,7 +2007,40 @@ void VulkanQueueRunner::PerformReadback(const VKRStep &step, VkCommandBuffer cmd
copyLayout = srcImage->layout;
}
- vkCmdCopyImageToBuffer(cmd, image, copyLayout, readbackBuffer_, 1, ®ion);
+ // TODO: Handle different readback formats!
+ u32 readbackSizeInBytes = sizeof(uint32_t) * step.readback.srcRect.extent.width * step.readback.srcRect.extent.height;
+
+ CachedReadback *cached = nullptr;
+
+ if (step.readback.delayed) {
+ ReadbackKey key;
+ key.framebuf = step.readback.src;
+ key.width = step.readback.srcRect.extent.width;
+ key.height = step.readback.srcRect.extent.height;
+
+ // See if there's already a buffer we can reuse
+ cached = frameData.readbacks_.Get(key);
+ if (!cached) {
+ cached = new CachedReadback();
+ cached->bufferSize = 0;
+ frameData.readbacks_.Insert(key, cached);
+ }
+ } else {
+ cached = &syncReadback_;
+ }
+
+ ResizeReadbackBuffer(cached, readbackSizeInBytes);
+
+ VkBufferImageCopy region{};
+ region.imageOffset = { step.readback.srcRect.offset.x, step.readback.srcRect.offset.y, 0 };
+ region.imageExtent = { step.readback.srcRect.extent.width, step.readback.srcRect.extent.height, 1 };
+ region.imageSubresource.aspectMask = step.readback.aspectMask;
+ region.imageSubresource.layerCount = 1;
+ region.bufferOffset = 0;
+ region.bufferRowLength = step.readback.srcRect.extent.width;
+ region.bufferImageHeight = step.readback.srcRect.extent.height;
+
+ vkCmdCopyImageToBuffer(cmd, image, copyLayout, cached->buffer, 1, ®ion);
// NOTE: Can't read the buffer using the CPU here - need to sync first.
@@ -2079,7 +2067,7 @@ void VulkanQueueRunner::PerformReadbackImage(const VKRStep &step, VkCommandBuffe
SetupTransitionToTransferSrc(srcImage, VK_IMAGE_ASPECT_COLOR_BIT, &recordBarrier_);
recordBarrier_.Flush(cmd);
- ResizeReadbackBuffer(sizeof(uint32_t) * step.readback_image.srcRect.extent.width * step.readback_image.srcRect.extent.height);
+ ResizeReadbackBuffer(&syncReadback_, sizeof(uint32_t) * step.readback_image.srcRect.extent.width * step.readback_image.srcRect.extent.height);
VkBufferImageCopy region{};
region.imageOffset = { step.readback_image.srcRect.offset.x, step.readback_image.srcRect.offset.y, 0 };
@@ -2090,7 +2078,7 @@ void VulkanQueueRunner::PerformReadbackImage(const VKRStep &step, VkCommandBuffe
region.bufferOffset = 0;
region.bufferRowLength = step.readback_image.srcRect.extent.width;
region.bufferImageHeight = step.readback_image.srcRect.extent.height;
- vkCmdCopyImageToBuffer(cmd, step.readback_image.image, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, readbackBuffer_, 1, ®ion);
+ vkCmdCopyImageToBuffer(cmd, step.readback_image.image, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, syncReadback_.buffer, 1, ®ion);
// Now transfer it back to a texture.
TransitionImageLayout2(cmd, step.readback_image.image, 0, 1, 1, // I don't think we have any multilayer cases for regular textures. Above in PerformReadback, though..
@@ -2103,26 +2091,39 @@ void VulkanQueueRunner::PerformReadbackImage(const VKRStep &step, VkCommandBuffe
// Doing that will also act like a heavyweight barrier ensuring that device writes are visible on the host.
}
-void VulkanQueueRunner::CopyReadbackBuffer(int width, int height, Draw::DataFormat srcFormat, Draw::DataFormat destFormat, int pixelStride, uint8_t *pixels) {
- if (!readbackMemory_)
- return; // Something has gone really wrong.
+bool VulkanQueueRunner::CopyReadbackBuffer(FrameData &frameData, VKRFramebuffer *src, int width, int height, Draw::DataFormat srcFormat, Draw::DataFormat destFormat, int pixelStride, uint8_t *pixels) {
+ CachedReadback *readback = &syncReadback_;
+
+ // Look up in readback cache.
+ if (src) {
+ ReadbackKey key;
+ key.framebuf = src;
+ key.width = width;
+ key.height = height;
+ CachedReadback *cached = frameData.readbacks_.Get(key);
+ if (cached) {
+ readback = cached;
+ } else {
+ // Didn't have a cached image ready yet
+ return false;
+ }
+ }
+
+ if (!readback->buffer)
+ return false; // Didn't find anything in cache, or something has gone really wrong.
// Read back to the requested address in ram from buffer.
void *mappedData;
const size_t srcPixelSize = DataFormatSizeInBytes(srcFormat);
-
- VkResult res = vkMapMemory(vulkan_->GetDevice(), readbackMemory_, 0, width * height * srcPixelSize, 0, &mappedData);
- if (!readbackBufferIsCoherent_) {
- VkMappedMemoryRange range{};
- range.memory = readbackMemory_;
- range.offset = 0;
- range.size = width * height * srcPixelSize;
- vkInvalidateMappedMemoryRanges(vulkan_->GetDevice(), 1, &range);
- }
+ VkResult res = vmaMapMemory(vulkan_->Allocator(), readback->allocation, &mappedData);
if (res != VK_SUCCESS) {
ERROR_LOG(G3D, "CopyReadbackBuffer: vkMapMemory failed! result=%d", (int)res);
- return;
+ return false;
+ }
+
+ if (!readback->isCoherent) {
+ vmaInvalidateAllocation(vulkan_->Allocator(), readback->allocation, 0, width * height * srcPixelSize);
}
// TODO: Perform these conversions in a compute shader on the GPU.
@@ -2148,5 +2149,7 @@ void VulkanQueueRunner::CopyReadbackBuffer(int width, int height, Draw::DataForm
ERROR_LOG(G3D, "CopyReadbackBuffer: Unknown format");
_assert_msg_(false, "CopyReadbackBuffer: Unknown src format %d", (int)srcFormat);
}
- vkUnmapMemory(vulkan_->GetDevice(), readbackMemory_);
+
+ vmaUnmapMemory(vulkan_->Allocator(), readback->allocation);
+ return true;
}
diff --git a/Common/GPU/Vulkan/VulkanQueueRunner.h b/Common/GPU/Vulkan/VulkanQueueRunner.h
index da46a07b0a..6df7dfc681 100644
--- a/Common/GPU/Vulkan/VulkanQueueRunner.h
+++ b/Common/GPU/Vulkan/VulkanQueueRunner.h
@@ -199,6 +199,7 @@ struct VKRStep {
int aspectMask;
VKRFramebuffer *src;
VkRect2D srcRect;
+ bool delayed;
} readback;
struct {
VkImage image;
@@ -252,7 +253,8 @@ public:
return (int)depth * 3 + (int)color;
}
- void CopyReadbackBuffer(int width, int height, Draw::DataFormat srcFormat, Draw::DataFormat destFormat, int pixelStride, uint8_t *pixels);
+ // src == 0 means to copy from the sync readback buffer.
+ bool CopyReadbackBuffer(FrameData &frameData, VKRFramebuffer *src, int width, int height, Draw::DataFormat srcFormat, Draw::DataFormat destFormat, int pixelStride, uint8_t *pixels);
VKRRenderPass *GetRenderPass(const RPKey &key);
@@ -288,7 +290,7 @@ private:
void PerformRenderPass(const VKRStep &pass, VkCommandBuffer cmd);
void PerformCopy(const VKRStep &pass, VkCommandBuffer cmd);
void PerformBlit(const VKRStep &pass, VkCommandBuffer cmd);
- void PerformReadback(const VKRStep &pass, VkCommandBuffer cmd);
+ void PerformReadback(const VKRStep &pass, VkCommandBuffer cmd, FrameData &frameData);
void PerformReadbackImage(const VKRStep &pass, VkCommandBuffer cmd);
void LogRenderPass(const VKRStep &pass, bool verbose);
@@ -297,7 +299,7 @@ private:
void LogReadback(const VKRStep &pass);
void LogReadbackImage(const VKRStep &pass);
- void ResizeReadbackBuffer(VkDeviceSize requiredSize);
+ void ResizeReadbackBuffer(CachedReadback *readback, VkDeviceSize requiredSize);
void ApplyMGSHack(std::vector &steps);
void ApplySonicHack(std::vector &steps);
@@ -323,10 +325,7 @@ private:
// Readback buffer. Currently we only support synchronous readback, so we only really need one.
// We size it generously.
- VkDeviceMemory readbackMemory_ = VK_NULL_HANDLE;
- VkBuffer readbackBuffer_ = VK_NULL_HANDLE;
- VkDeviceSize readbackBufferSize_ = 0;
- bool readbackBufferIsCoherent_ = false;
+ CachedReadback syncReadback_{};
// TODO: Enable based on compat.ini.
uint32_t hacksEnabled_ = 0;
diff --git a/Common/GPU/Vulkan/VulkanRenderManager.cpp b/Common/GPU/Vulkan/VulkanRenderManager.cpp
index 32e859936a..7ab778d8c2 100644
--- a/Common/GPU/Vulkan/VulkanRenderManager.cpp
+++ b/Common/GPU/Vulkan/VulkanRenderManager.cpp
@@ -403,6 +403,10 @@ public:
return TaskType::CPU_COMPUTE;
}
+ TaskPriority Priority() const override {
+ return TaskPriority::HIGH;
+ }
+
void Run() override {
for (auto &task : tasks_) {
task.pipeline->Create(vulkan_, task.compatibleRenderPass, task.rpType, task.sampleCount, task.scheduleTime, task.countToCompile);
@@ -880,7 +884,8 @@ void VulkanRenderManager::BindFramebufferAsRenderTarget(VKRFramebuffer *fb, VKRR
} else {
curWidthRaw_ = vulkan_->GetBackbufferWidth();
curHeightRaw_ = vulkan_->GetBackbufferHeight();
- if (g_display_rotation == DisplayRotation::ROTATE_90 || g_display_rotation == DisplayRotation::ROTATE_270) {
+ if (g_display.rotation == DisplayRotation::ROTATE_90 ||
+ g_display.rotation == DisplayRotation::ROTATE_270) {
curWidth_ = curHeightRaw_;
curHeight_ = curWidthRaw_;
} else {
@@ -908,8 +913,9 @@ void VulkanRenderManager::BindFramebufferAsRenderTarget(VKRFramebuffer *fb, VKRR
}
}
-bool VulkanRenderManager::CopyFramebufferToMemorySync(VKRFramebuffer *src, VkImageAspectFlags aspectBits, int x, int y, int w, int h, Draw::DataFormat destFormat, uint8_t *pixels, int pixelStride, const char *tag) {
+bool VulkanRenderManager::CopyFramebufferToMemory(VKRFramebuffer *src, VkImageAspectFlags aspectBits, int x, int y, int w, int h, Draw::DataFormat destFormat, uint8_t *pixels, int pixelStride, Draw::ReadbackMode mode, const char *tag) {
_dbg_assert_(insideFrame_);
+
for (int i = (int)steps_.size() - 1; i >= 0; i--) {
if (steps_[i]->stepType == VKRStepType::RENDER && steps_[i]->render.framebuffer == src) {
steps_[i]->render.numReads++;
@@ -924,11 +930,14 @@ bool VulkanRenderManager::CopyFramebufferToMemorySync(VKRFramebuffer *src, VkIma
step->readback.src = src;
step->readback.srcRect.offset = { x, y };
step->readback.srcRect.extent = { (uint32_t)w, (uint32_t)h };
+ step->readback.delayed = mode == Draw::ReadbackMode::OLD_DATA_OK;
step->dependencies.insert(src);
step->tag = tag;
steps_.push_back(step);
- FlushSync();
+ if (mode == Draw::ReadbackMode::BLOCK) {
+ FlushSync();
+ }
Draw::DataFormat srcFormat = Draw::DataFormat::UNDEFINED;
if (aspectBits & VK_IMAGE_ASPECT_COLOR_BIT) {
@@ -967,8 +976,8 @@ bool VulkanRenderManager::CopyFramebufferToMemorySync(VKRFramebuffer *src, VkIma
}
// Need to call this after FlushSync so the pixels are guaranteed to be ready in CPU-accessible VRAM.
- queueRunner_.CopyReadbackBuffer(w, h, srcFormat, destFormat, pixelStride, pixels);
- return true;
+ return queueRunner_.CopyReadbackBuffer(frameData_[vulkan_->GetCurFrame()],
+ mode == Draw::ReadbackMode::OLD_DATA_OK ? src : nullptr, w, h, srcFormat, destFormat, pixelStride, pixels);
}
void VulkanRenderManager::CopyImageToMemorySync(VkImage image, int mipLevel, int x, int y, int w, int h, Draw::DataFormat destFormat, uint8_t *pixels, int pixelStride, const char *tag) {
@@ -987,7 +996,7 @@ void VulkanRenderManager::CopyImageToMemorySync(VkImage image, int mipLevel, int
FlushSync();
// Need to call this after FlushSync so the pixels are guaranteed to be ready in CPU-accessible VRAM.
- queueRunner_.CopyReadbackBuffer(w, h, destFormat, destFormat, pixelStride, pixels);
+ queueRunner_.CopyReadbackBuffer(frameData_[vulkan_->GetCurFrame()], nullptr, w, h, destFormat, destFormat, pixelStride, pixels);
}
static void RemoveDrawCommands(std::vector *cmds) {
diff --git a/Common/GPU/Vulkan/VulkanRenderManager.h b/Common/GPU/Vulkan/VulkanRenderManager.h
index 4536fcf932..3cad859e8d 100644
--- a/Common/GPU/Vulkan/VulkanRenderManager.h
+++ b/Common/GPU/Vulkan/VulkanRenderManager.h
@@ -217,7 +217,7 @@ public:
void BindCurrentFramebufferAsInputAttachment0(VkImageAspectFlags aspectBits);
- bool CopyFramebufferToMemorySync(VKRFramebuffer *src, VkImageAspectFlags aspectBits, int x, int y, int w, int h, Draw::DataFormat destFormat, uint8_t *pixels, int pixelStride, const char *tag);
+ bool CopyFramebufferToMemory(VKRFramebuffer *src, VkImageAspectFlags aspectBits, int x, int y, int w, int h, Draw::DataFormat destFormat, uint8_t *pixels, int pixelStride, Draw::ReadbackMode mode, const char *tag);
void CopyImageToMemorySync(VkImage image, int mipLevel, int x, int y, int w, int h, Draw::DataFormat destFormat, uint8_t *pixels, int pixelStride, const char *tag);
void CopyFramebuffer(VKRFramebuffer *src, VkRect2D srcRect, VKRFramebuffer *dst, VkOffset2D dstPos, VkImageAspectFlags aspectMask, const char *tag);
@@ -416,8 +416,10 @@ public:
// These can be useful both when inspecting in RenderDoc, and when manually inspecting recorded commands
// in the debugger.
void DebugAnnotate(const char *annotation) {
+ _dbg_assert_(curRenderStep_);
VkRenderData data{ VKRRenderCommand::DEBUG_ANNOTATION };
data.debugAnnotation.annotation = annotation;
+ curRenderStep_->commands.push_back(data);
}
VkCommandBuffer GetInitCmd();
diff --git a/Common/GPU/Vulkan/thin3d_vulkan.cpp b/Common/GPU/Vulkan/thin3d_vulkan.cpp
index 7046121f20..fe1929cd52 100644
--- a/Common/GPU/Vulkan/thin3d_vulkan.cpp
+++ b/Common/GPU/Vulkan/thin3d_vulkan.cpp
@@ -283,8 +283,8 @@ public:
}
// Returns the binding offset, and the VkBuffer to bind.
- size_t PushUBO(VulkanPushBuffer *buf, VulkanContext *vulkan, VkBuffer *vkbuf) {
- return buf->PushAligned(ubo_, uboSize_, vulkan->GetPhysicalDeviceProperties().properties.limits.minUniformBufferOffsetAlignment, vkbuf);
+ size_t PushUBO(VulkanPushPool *buf, VulkanContext *vulkan, VkBuffer *vkbuf) {
+ return buf->Push(ubo_, uboSize_, vulkan->GetPhysicalDeviceProperties().properties.limits.minUniformBufferOffsetAlignment, vkbuf);
}
int GetUBOSize() const {
@@ -334,9 +334,9 @@ struct DescriptorSetKey {
class VKTexture : public Texture {
public:
- VKTexture(VulkanContext *vulkan, VkCommandBuffer cmd, VulkanPushBuffer *pushBuffer, const TextureDesc &desc)
+ VKTexture(VulkanContext *vulkan, VkCommandBuffer cmd, VulkanPushPool *pushBuffer, const TextureDesc &desc)
: vulkan_(vulkan), mipLevels_(desc.mipLevels), format_(desc.format) {}
- bool Create(VkCommandBuffer cmd, VulkanPushBuffer *pushBuffer, const TextureDesc &desc);
+ bool Create(VkCommandBuffer cmd, VulkanPushPool *pushBuffer, const TextureDesc &desc);
~VKTexture() {
Destroy();
@@ -414,7 +414,7 @@ public:
void CopyFramebufferImage(Framebuffer *src, int level, int x, int y, int z, Framebuffer *dst, int dstLevel, int dstX, int dstY, int dstZ, int width, int height, int depth, int channelBits, const char *tag) override;
bool BlitFramebuffer(Framebuffer *src, int srcX1, int srcY1, int srcX2, int srcY2, Framebuffer *dst, int dstX1, int dstY1, int dstX2, int dstY2, int channelBits, FBBlitFilter filter, const char *tag) override;
- bool CopyFramebufferToMemorySync(Framebuffer *src, int channelBits, int x, int y, int w, int h, Draw::DataFormat format, void *pixels, int pixelStride, const char *tag) override;
+ bool CopyFramebufferToMemory(Framebuffer *src, int channelBits, int x, int y, int w, int h, Draw::DataFormat format, void *pixels, int pixelStride, ReadbackMode mode, const char *tag) override;
DataFormat PreferredFramebufferReadbackFormat(Framebuffer *src) override;
// These functions should be self explanatory.
@@ -425,7 +425,7 @@ public:
void GetFramebufferDimensions(Framebuffer *fbo, int *w, int *h) override;
void SetScissorRect(int left, int top, int width, int height) override;
- void SetViewports(int count, Viewport *viewports) override;
+ void SetViewport(const Viewport &viewport) override;
void SetBlendFactor(float color[4]) override;
void SetStencilParams(uint8_t refValue, uint8_t writeMask, uint8_t compareMask) override;
@@ -536,12 +536,12 @@ private:
VkImageView boundImageView_[MAX_BOUND_TEXTURES]{};
TextureBindFlags boundTextureFlags_[MAX_BOUND_TEXTURES];
+ VulkanPushPool *push_ = nullptr;
+
struct FrameData {
FrameData() : descriptorPool("VKContext", false) {
descriptorPool.Setup([this] { descSets_.clear(); });
}
-
- VulkanPushBuffer *pushBuffer = nullptr;
// Per-frame descriptor set cache. As it's per frame and reset every frame, we don't need to
// worry about invalidating descriptors pointing to deleted textures.
// However! ARM is not a fan of doing it this way.
@@ -551,8 +551,6 @@ private:
FrameData frame_[VulkanContext::MAX_INFLIGHT_FRAMES];
- VulkanPushBuffer *push_ = nullptr;
-
DeviceCaps caps_{};
uint8_t stencilRef_ = 0;
@@ -560,6 +558,7 @@ private:
uint8_t stencilCompareMask_ = 0xFF;
};
+// Bits per pixel, not bytes.
static int GetBpp(VkFormat format) {
switch (format) {
case VK_FORMAT_R8G8B8A8_UNORM:
@@ -582,6 +581,21 @@ static int GetBpp(VkFormat format) {
return 32;
case VK_FORMAT_D16_UNORM:
return 16;
+ case VK_FORMAT_ETC2_R8G8B8_UNORM_BLOCK:
+ return 4;
+ case VK_FORMAT_ETC2_R8G8B8A1_UNORM_BLOCK:
+ case VK_FORMAT_ETC2_R8G8B8A8_UNORM_BLOCK:
+ return 8;
+ case VK_FORMAT_ASTC_4x4_UNORM_BLOCK:
+ return 8;
+ case VK_FORMAT_BC1_RGBA_UNORM_BLOCK:
+ return 4;
+ case VK_FORMAT_BC2_UNORM_BLOCK:
+ case VK_FORMAT_BC3_UNORM_BLOCK:
+ case VK_FORMAT_BC4_UNORM_BLOCK:
+ case VK_FORMAT_BC5_UNORM_BLOCK:
+ case VK_FORMAT_BC7_UNORM_BLOCK:
+ return 8;
default:
return 0;
}
@@ -625,13 +639,15 @@ static VkFormat DataFormatToVulkan(DataFormat format) {
case DataFormat::BC2_UNORM_BLOCK: return VK_FORMAT_BC2_UNORM_BLOCK;
case DataFormat::BC3_UNORM_BLOCK: return VK_FORMAT_BC3_UNORM_BLOCK;
case DataFormat::BC4_UNORM_BLOCK: return VK_FORMAT_BC4_UNORM_BLOCK;
- case DataFormat::BC4_SNORM_BLOCK: return VK_FORMAT_BC4_SNORM_BLOCK;
case DataFormat::BC5_UNORM_BLOCK: return VK_FORMAT_BC5_UNORM_BLOCK;
- case DataFormat::BC5_SNORM_BLOCK: return VK_FORMAT_BC5_SNORM_BLOCK;
- case DataFormat::BC6H_SFLOAT_BLOCK: return VK_FORMAT_BC6H_SFLOAT_BLOCK;
- case DataFormat::BC6H_UFLOAT_BLOCK: return VK_FORMAT_BC6H_UFLOAT_BLOCK;
case DataFormat::BC7_UNORM_BLOCK: return VK_FORMAT_BC7_UNORM_BLOCK;
- case DataFormat::BC7_SRGB_BLOCK: return VK_FORMAT_BC7_SRGB_BLOCK;
+
+ case DataFormat::ETC2_R8G8B8A1_UNORM_BLOCK: return VK_FORMAT_ETC2_R8G8B8A1_UNORM_BLOCK;
+ case DataFormat::ETC2_R8G8B8A8_UNORM_BLOCK: return VK_FORMAT_ETC2_R8G8B8A8_UNORM_BLOCK;
+ case DataFormat::ETC2_R8G8B8_UNORM_BLOCK: return VK_FORMAT_ETC2_R8G8B8_UNORM_BLOCK;
+
+ case DataFormat::ASTC_4x4_UNORM_BLOCK: return VK_FORMAT_ASTC_4x4_UNORM_BLOCK;
+
default:
return VK_FORMAT_UNDEFINED;
}
@@ -657,14 +673,16 @@ VulkanTexture *VKContext::GetNullTexture() {
VK_IMAGE_USAGE_TRANSFER_DST_BIT | VK_IMAGE_USAGE_SAMPLED_BIT);
uint32_t bindOffset;
VkBuffer bindBuf;
- uint32_t *data = (uint32_t *)push_->Push(w * h * 4, &bindOffset, &bindBuf);
+ uint32_t *data = (uint32_t *)push_->Allocate(w * h * 4, 4, &bindBuf, &bindOffset);
for (int y = 0; y < h; y++) {
for (int x = 0; x < w; x++) {
// data[y*w + x] = ((x ^ y) & 1) ? 0xFF808080 : 0xFF000000; // gray/black checkerboard
data[y*w + x] = 0; // black
}
}
- nullTexture_->UploadMip(cmdInit, 0, w, h, 0, bindBuf, bindOffset, w);
+ TextureCopyBatch batch;
+ nullTexture_->CopyBufferToMipLevel(cmdInit, &batch, 0, w, h, 0, bindBuf, bindOffset, w);
+ nullTexture_->FinishCopyBatch(cmdInit, &batch);
nullTexture_->EndCreate(cmdInit, false, VK_PIPELINE_STAGE_TRANSFER_BIT);
}
return nullTexture_;
@@ -719,7 +737,7 @@ enum class TextureState {
PENDING_DESTRUCTION,
};
-bool VKTexture::Create(VkCommandBuffer cmd, VulkanPushBuffer *push, const TextureDesc &desc) {
+bool VKTexture::Create(VkCommandBuffer cmd, VulkanPushPool *push, const TextureDesc &desc) {
// Zero-sized textures not allowed.
_assert_(desc.width * desc.height * desc.depth > 0); // remember to set depth to 1!
if (desc.width * desc.height * desc.depth <= 0) {
@@ -742,7 +760,9 @@ bool VKTexture::Create(VkCommandBuffer cmd, VulkanPushBuffer *push, const Textur
usageBits |= VK_IMAGE_USAGE_TRANSFER_SRC_BIT;
}
- if (!vkTex_->CreateDirect(cmd, width_, height_, 1, mipLevels_, vulkanFormat, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, usageBits)) {
+ VkComponentMapping r8AsAlpha[4] = { VK_COMPONENT_SWIZZLE_ONE, VK_COMPONENT_SWIZZLE_ONE, VK_COMPONENT_SWIZZLE_ONE, VK_COMPONENT_SWIZZLE_R };
+
+ if (!vkTex_->CreateDirect(cmd, width_, height_, 1, mipLevels_, vulkanFormat, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, usageBits, desc.swizzle == TextureSwizzle::R8_AS_ALPHA ? r8AsAlpha : nullptr)) {
ERROR_LOG(G3D, "Failed to create VulkanTexture: %dx%dx%d fmt %d, %d levels", width_, height_, depth_, (int)vulkanFormat, mipLevels_);
return false;
}
@@ -757,14 +777,16 @@ bool VKTexture::Create(VkCommandBuffer cmd, VulkanPushBuffer *push, const Textur
VkBuffer buf;
size_t size = w * h * d * bytesPerPixel;
if (desc.initDataCallback) {
- uint8_t *dest = (uint8_t *)push->PushAligned(size, &offset, &buf, 16);
+ uint8_t *dest = (uint8_t *)push->Allocate(size, 16, &buf, &offset);
if (!desc.initDataCallback(dest, desc.initData[i], w, h, d, w * bytesPerPixel, h * w * bytesPerPixel)) {
memcpy(dest, desc.initData[i], size);
}
} else {
- offset = push->PushAligned((const void *)desc.initData[i], size, 16, &buf);
+ offset = push->Push((const void *)desc.initData[i], size, 16, &buf);
}
- vkTex_->UploadMip(cmd, i, w, h, 0, buf, offset, w);
+ TextureCopyBatch batch;
+ vkTex_->CopyBufferToMipLevel(cmd, &batch, i, w, h, 0, buf, offset, w);
+ vkTex_->FinishCopyBatch(cmd, &batch);
w = (w + 1) / 2;
h = (h + 1) / 2;
d = (d + 1) / 2;
@@ -959,9 +981,10 @@ VKContext::VKContext(VulkanContext *vulkan)
// 200 textures per frame was not enough for the UI.
dp.maxSets = 4096;
+ VkBufferUsageFlags usage = VK_BUFFER_USAGE_INDEX_BUFFER_BIT | VK_BUFFER_USAGE_UNIFORM_BUFFER_BIT | VK_BUFFER_USAGE_STORAGE_BUFFER_BIT | VK_BUFFER_USAGE_VERTEX_BUFFER_BIT | VK_BUFFER_USAGE_TRANSFER_SRC_BIT;
+ push_ = new VulkanPushPool(vulkan_, "pushBuffer", 4 * 1024 * 1024, usage);
+
for (int i = 0; i < VulkanContext::MAX_INFLIGHT_FRAMES; i++) {
- VkBufferUsageFlags usage = VK_BUFFER_USAGE_INDEX_BUFFER_BIT | VK_BUFFER_USAGE_UNIFORM_BUFFER_BIT | VK_BUFFER_USAGE_VERTEX_BUFFER_BIT | VK_BUFFER_USAGE_TRANSFER_SRC_BIT;
- frame_[i].pushBuffer = new VulkanPushBuffer(vulkan_, "pushBuffer", 1024 * 1024, usage, PushBufferType::CPU_TO_GPU);
frame_[i].descriptorPool.Create(vulkan_, dp, dpTypes);
}
@@ -1013,9 +1036,9 @@ VKContext::~VKContext() {
// This also destroys all descriptor sets.
for (int i = 0; i < VulkanContext::MAX_INFLIGHT_FRAMES; i++) {
frame_[i].descriptorPool.Destroy();
- frame_[i].pushBuffer->Destroy(vulkan_);
- delete frame_[i].pushBuffer;
}
+ push_->Destroy();
+ delete push_;
vulkan_->Delete().QueueDeleteDescriptorSetLayout(descriptorSetLayout_);
vulkan_->Delete().QueueDeletePipelineLayout(pipelineLayout_);
vulkan_->Delete().QueueDeletePipelineCache(pipelineCache_);
@@ -1026,23 +1049,15 @@ void VKContext::BeginFrame() {
renderManager_.BeginFrame(debugFlags_ & DebugFlags::PROFILE_TIMESTAMPS, debugFlags_ & DebugFlags::PROFILE_SCOPES);
FrameData &frame = frame_[vulkan_->GetCurFrame()];
- push_ = frame.pushBuffer;
- // OK, we now know that nothing is reading from this frame's data pushbuffer,
- push_->Reset();
- push_->Begin(vulkan_);
+ push_->BeginFrame();
frame.descriptorPool.Reset();
}
void VKContext::EndFrame() {
- // Stop collecting data in the frame's data pushbuffer.
- push_->End();
-
renderManager_.Finish();
- push_ = nullptr;
-
// Unbind stuff, to avoid accidentally relying on it across frames (and provide some protection against forgotten unbinds of deleted things).
Invalidate(InvalidationFlags::CACHED_RENDER_STATE);
}
@@ -1237,18 +1252,16 @@ void VKContext::SetScissorRect(int left, int top, int width, int height) {
renderManager_.SetScissor(left, top, width, height);
}
-void VKContext::SetViewports(int count, Viewport *viewports) {
- if (count > 0) {
- // Ignore viewports more than the first.
- VkViewport viewport;
- viewport.x = viewports[0].TopLeftX;
- viewport.y = viewports[0].TopLeftY;
- viewport.width = viewports[0].Width;
- viewport.height = viewports[0].Height;
- viewport.minDepth = viewports[0].MinDepth;
- viewport.maxDepth = viewports[0].MaxDepth;
- renderManager_.SetViewport(viewport);
- }
+void VKContext::SetViewport(const Viewport &viewport) {
+ // Ignore viewports more than the first.
+ VkViewport vkViewport;
+ vkViewport.x = viewport.TopLeftX;
+ vkViewport.y = viewport.TopLeftY;
+ vkViewport.width = viewport.Width;
+ vkViewport.height = viewport.Height;
+ vkViewport.minDepth = viewport.MinDepth;
+ vkViewport.maxDepth = viewport.MaxDepth;
+ renderManager_.SetViewport(vkViewport);
}
void VKContext::SetBlendFactor(float color[4]) {
@@ -1430,7 +1443,7 @@ void VKContext::Draw(int vertexCount, int offset) {
VkBuffer vulkanVbuf;
VkBuffer vulkanUBObuf;
uint32_t ubo_offset = (uint32_t)curPipeline_->PushUBO(push_, vulkan_, &vulkanUBObuf);
- size_t vbBindOffset = push_->Push(vbuf->GetData(), vbuf->GetSize(), &vulkanVbuf);
+ size_t vbBindOffset = push_->Push(vbuf->GetData(), vbuf->GetSize(), 4, &vulkanVbuf);
VkDescriptorSet descSet = GetOrCreateDescriptorSet(vulkanUBObuf);
if (descSet == VK_NULL_HANDLE) {
@@ -1449,8 +1462,8 @@ void VKContext::DrawIndexed(int vertexCount, int offset) {
VkBuffer vulkanVbuf, vulkanIbuf, vulkanUBObuf;
uint32_t ubo_offset = (uint32_t)curPipeline_->PushUBO(push_, vulkan_, &vulkanUBObuf);
- size_t vbBindOffset = push_->Push(vbuf->GetData(), vbuf->GetSize(), &vulkanVbuf);
- size_t ibBindOffset = push_->Push(ibuf->GetData(), ibuf->GetSize(), &vulkanIbuf);
+ size_t vbBindOffset = push_->Push(vbuf->GetData(), vbuf->GetSize(), 4, &vulkanVbuf);
+ size_t ibBindOffset = push_->Push(ibuf->GetData(), ibuf->GetSize(), 4, &vulkanIbuf);
VkDescriptorSet descSet = GetOrCreateDescriptorSet(vulkanUBObuf);
if (descSet == VK_NULL_HANDLE) {
@@ -1465,7 +1478,7 @@ void VKContext::DrawIndexed(int vertexCount, int offset) {
void VKContext::DrawUP(const void *vdata, int vertexCount) {
VkBuffer vulkanVbuf, vulkanUBObuf;
- size_t vbBindOffset = push_->Push(vdata, vertexCount * curPipeline_->stride[0], &vulkanVbuf);
+ size_t vbBindOffset = push_->Push(vdata, vertexCount * curPipeline_->stride[0], 4, &vulkanVbuf);
uint32_t ubo_offset = (uint32_t)curPipeline_->PushUBO(push_, vulkan_, &vulkanUBObuf);
VkDescriptorSet descSet = GetOrCreateDescriptorSet(vulkanUBObuf);
@@ -1632,7 +1645,7 @@ bool VKContext::BlitFramebuffer(Framebuffer *srcfb, int srcX1, int srcY1, int sr
return true;
}
-bool VKContext::CopyFramebufferToMemorySync(Framebuffer *srcfb, int channelBits, int x, int y, int w, int h, Draw::DataFormat format, void *pixels, int pixelStride, const char *tag) {
+bool VKContext::CopyFramebufferToMemory(Framebuffer *srcfb, int channelBits, int x, int y, int w, int h, Draw::DataFormat format, void *pixels, int pixelStride, ReadbackMode mode, const char *tag) {
VKFramebuffer *src = (VKFramebuffer *)srcfb;
int aspectMask = 0;
@@ -1640,7 +1653,7 @@ bool VKContext::CopyFramebufferToMemorySync(Framebuffer *srcfb, int channelBits,
if (channelBits & FBChannel::FB_DEPTH_BIT) aspectMask |= VK_IMAGE_ASPECT_DEPTH_BIT;
if (channelBits & FBChannel::FB_STENCIL_BIT) aspectMask |= VK_IMAGE_ASPECT_STENCIL_BIT;
- return renderManager_.CopyFramebufferToMemorySync(src ? src->GetFB() : nullptr, aspectMask, x, y, w, h, format, (uint8_t *)pixels, pixelStride, tag);
+ return renderManager_.CopyFramebufferToMemory(src ? src->GetFB() : nullptr, aspectMask, x, y, w, h, format, (uint8_t *)pixels, pixelStride, mode, tag);
}
DataFormat VKContext::PreferredFramebufferReadbackFormat(Framebuffer *src) {
@@ -1757,7 +1770,8 @@ uint64_t VKContext::GetNativeObject(NativeObject obj, void *srcObject) {
return (uint64_t)curFramebuffer_->GetFB()->GetRTView();
case NativeObject::THIN3D_PIPELINE_LAYOUT:
return (uint64_t)pipelineLayout_;
-
+ case NativeObject::PUSH_POOL:
+ return (uint64_t)push_;
default:
Crash();
return 0;
diff --git a/Common/GPU/thin3d.cpp b/Common/GPU/thin3d.cpp
index 83726b7391..f5dc4b452e 100644
--- a/Common/GPU/thin3d.cpp
+++ b/Common/GPU/thin3d.cpp
@@ -93,6 +93,31 @@ bool DataFormatIsDepthStencil(DataFormat fmt) {
}
}
+// We don't bother listing the formats that are irrelevant for PPSSPP, like BC6 (HDR format)
+// or weird-shaped ASTC formats. We only support 4x4 block size formats for now.
+// If you pass in a blockSize parameter, it receives byte count that a 4x4 block takes in this format.
+bool DataFormatIsBlockCompressed(DataFormat fmt, int *blockSize) {
+ switch (fmt) {
+ case DataFormat::BC1_RGBA_UNORM_BLOCK:
+ case DataFormat::BC4_UNORM_BLOCK:
+ case DataFormat::ETC2_R8G8B8_UNORM_BLOCK:
+ if (blockSize) *blockSize = 8; // 64 bits
+ return true;
+ case DataFormat::BC2_UNORM_BLOCK:
+ case DataFormat::BC3_UNORM_BLOCK:
+ case DataFormat::BC5_UNORM_BLOCK:
+ case DataFormat::BC7_UNORM_BLOCK:
+ case DataFormat::ETC2_R8G8B8A1_UNORM_BLOCK:
+ case DataFormat::ETC2_R8G8B8A8_UNORM_BLOCK:
+ case DataFormat::ASTC_4x4_UNORM_BLOCK:
+ if (blockSize) *blockSize = 16; // 128 bits
+ return true;
+ default:
+ if (blockSize) *blockSize = 0;
+ return false;
+ }
+}
+
RefCountedObject::~RefCountedObject() {
_dbg_assert_(refcount_ == 0xDEDEDE);
}
diff --git a/Common/GPU/thin3d.h b/Common/GPU/thin3d.h
index bf10cb3801..ae3db6af00 100644
--- a/Common/GPU/thin3d.h
+++ b/Common/GPU/thin3d.h
@@ -252,6 +252,7 @@ enum class NativeObject {
NULL_IMAGEVIEW,
NULL_IMAGEVIEW_ARRAY,
THIN3D_PIPELINE_LAYOUT,
+ PUSH_POOL,
};
enum FBChannel {
@@ -292,6 +293,11 @@ enum class Event {
PRESENTED,
};
+enum class ReadbackMode {
+ BLOCK,
+ OLD_DATA_OK, // Lets the backend return old results that won't need any waiting to get.
+};
+
constexpr uint32_t MAX_TEXTURE_SLOTS = 3;
struct FramebufferDesc {
@@ -593,6 +599,11 @@ struct DeviceCaps {
// Important: only write to the provided pointer, don't read from it.
typedef std::function TextureCallback;
+enum class TextureSwizzle {
+ DEFAULT,
+ R8_AS_ALPHA,
+};
+
struct TextureDesc {
TextureType type;
DataFormat format;
@@ -602,6 +613,7 @@ struct TextureDesc {
int depth;
int mipLevels;
bool generateMips;
+ TextureSwizzle swizzle;
// Optional, for tracking memory usage and graphcis debuggers.
const char *tag;
// Does not take ownership over pointed-to data.
@@ -693,7 +705,9 @@ public:
virtual void CopyFramebufferImage(Framebuffer *src, int level, int x, int y, int z, Framebuffer *dst, int dstLevel, int dstX, int dstY, int dstZ, int width, int height, int depth, int channelBits, const char *tag) = 0;
virtual bool BlitFramebuffer(Framebuffer *src, int srcX1, int srcY1, int srcX2, int srcY2, Framebuffer *dst, int dstX1, int dstY1, int dstX2, int dstY2, int channelBits, FBBlitFilter filter, const char *tag) = 0;
- virtual bool CopyFramebufferToMemorySync(Framebuffer *src, int channelBits, int x, int y, int w, int h, Draw::DataFormat format, void *pixels, int pixelStride, const char *tag) {
+
+ // If the backend doesn't support old data, it's "OK" to block.
+ virtual bool CopyFramebufferToMemory(Framebuffer *src, int channelBits, int x, int y, int w, int h, Draw::DataFormat format, void *pixels, int pixelStride, ReadbackMode mode, const char *tag) {
return false;
}
virtual DataFormat PreferredFramebufferReadbackFormat(Framebuffer *src) {
@@ -726,7 +740,7 @@ public:
// Dynamic state
virtual void SetScissorRect(int left, int top, int width, int height) = 0;
- virtual void SetViewports(int count, Viewport *viewports) = 0;
+ virtual void SetViewport(const Viewport &viewport) = 0;
virtual void SetBlendFactor(float color[4]) = 0;
virtual void SetStencilParams(uint8_t refValue, uint8_t writeMask, uint8_t compareMask) = 0;
diff --git a/Common/Log.cpp b/Common/Log.cpp
index d3f7ed9b02..28c2f821d0 100644
--- a/Common/Log.cpp
+++ b/Common/Log.cpp
@@ -24,6 +24,7 @@
#include "Common/Log.h"
#include "StringUtils.h"
#include "Common/Data/Encoding/Utf8.h"
+#include "Common/Thread/ThreadUtil.h"
#if PPSSPP_PLATFORM(ANDROID)
#include
@@ -71,7 +72,7 @@ bool HandleAssert(const char *function, const char *file, int line, const char *
if (!getenv("CI")) {
int msgBoxStyle = MB_ICONINFORMATION | MB_YESNO;
std::wstring wtext = ConvertUTF8ToWString(formatted) + L"\n\nTry to continue?";
- std::wstring wcaption = ConvertUTF8ToWString(caption);
+ std::wstring wcaption = ConvertUTF8ToWString(std::string(caption) + " " + GetCurrentThreadName());
OutputDebugString(wtext.c_str());
if (IDYES != MessageBox(0, wtext.c_str(), wcaption.c_str(), msgBoxStyle)) {
return false;
diff --git a/Common/Math/math_util.h b/Common/Math/math_util.h
index fd47662b54..b76e1f3937 100644
--- a/Common/Math/math_util.h
+++ b/Common/Math/math_util.h
@@ -40,6 +40,10 @@ inline uint32_t RoundUpToPowerOf2(uint32_t v) {
return v;
}
+inline uint32_t RoundUpToPowerOf2(uint32_t v, uint32_t power) {
+ return (v + power - 1) & ~(power - 1);
+}
+
inline uint32_t log2i(uint32_t val) {
unsigned int ret = -1;
while (val != 0) {
diff --git a/Common/MemoryUtil.cpp b/Common/MemoryUtil.cpp
index 28464006a2..139cd4f9e0 100644
--- a/Common/MemoryUtil.cpp
+++ b/Common/MemoryUtil.cpp
@@ -206,11 +206,12 @@ void *AllocateExecutableMemory(size_t size) {
}
void *AllocateMemoryPages(size_t size, uint32_t memProtFlags) {
- size = ppsspp_round_page(size);
#ifdef _WIN32
if (sys_info.dwPageSize == 0)
GetSystemInfo(&sys_info);
uint32_t protect = ConvertProtFlagsWin32(memProtFlags);
+ // Make sure to do this after GetSystemInfo().
+ size = ppsspp_round_page(size);
#if PPSSPP_PLATFORM(UWP)
void* ptr = VirtualAllocFromApp(0, size, MEM_COMMIT, protect);
#else
@@ -221,6 +222,7 @@ void *AllocateMemoryPages(size_t size, uint32_t memProtFlags) {
return nullptr;
}
#else
+ size = ppsspp_round_page(size);
uint32_t protect = ConvertProtFlagsUnix(memProtFlags);
void *ptr = mmap(0, size, protect, MAP_ANON | MAP_PRIVATE, -1, 0);
if (ptr == MAP_FAILED) {
@@ -248,7 +250,7 @@ void *AllocateAlignedMemory(size_t size, size_t alignment) {
#endif
#endif
- _assert_msg_(ptr != nullptr, "Failed to allocate aligned memory");
+ _assert_msg_(ptr != nullptr, "Failed to allocate aligned memory of size %llu", size);
return ptr;
}
diff --git a/Common/Render/DrawBuffer.cpp b/Common/Render/DrawBuffer.cpp
index bf49bc4774..73c0f30a68 100644
--- a/Common/Render/DrawBuffer.cpp
+++ b/Common/Render/DrawBuffer.cpp
@@ -111,14 +111,14 @@ void DrawBuffer::Rect(float x, float y, float w, float h, uint32_t color, int al
void DrawBuffer::hLine(float x1, float y, float x2, uint32_t color) {
// Round Y to the closest full pixel, since we're making it 1-pixel-thin.
- y -= fmodf(y, pixel_in_dps_y);
- Rect(x1, y, x2 - x1, pixel_in_dps_y, color);
+ y -= fmodf(y, g_display.pixel_in_dps_y);
+ Rect(x1, y, x2 - x1, g_display.pixel_in_dps_y, color);
}
void DrawBuffer::vLine(float x, float y1, float y2, uint32_t color) {
// Round X to the closest full pixel, since we're making it 1-pixel-thin.
- x -= fmodf(x, pixel_in_dps_x);
- Rect(x, y1, pixel_in_dps_x, y2 - y1, color);
+ x -= fmodf(x, g_display.pixel_in_dps_x);
+ Rect(x, y1, g_display.pixel_in_dps_x, y2 - y1, color);
}
void DrawBuffer::RectVGradient(float x, float y, float w, float h, uint32_t colorTop, uint32_t colorBottom) {
@@ -131,11 +131,11 @@ void DrawBuffer::RectVGradient(float x, float y, float w, float h, uint32_t colo
}
void DrawBuffer::RectOutline(float x, float y, float w, float h, uint32_t color, int align) {
- hLine(x, y, x + w + pixel_in_dps_x, color);
- hLine(x, y + h, x + w + pixel_in_dps_x, color);
+ hLine(x, y, x + w + g_display.pixel_in_dps_x, color);
+ hLine(x, y + h, x + w + g_display.pixel_in_dps_x, color);
- vLine(x, y, y + h + pixel_in_dps_y, color);
- vLine(x + w, y, y + h + pixel_in_dps_y, color);
+ vLine(x, y, y + h + g_display.pixel_in_dps_y, color);
+ vLine(x + w, y, y + h + g_display.pixel_in_dps_y, color);
}
void DrawBuffer::MultiVGradient(float x, float y, float w, float h, const GradientStop *stops, int numStops) {
diff --git a/Common/Render/ManagedTexture.cpp b/Common/Render/ManagedTexture.cpp
index 771d4f8a92..dce43253c2 100644
--- a/Common/Render/ManagedTexture.cpp
+++ b/Common/Render/ManagedTexture.cpp
@@ -148,7 +148,7 @@ bool ManagedTexture::LoadFromFileData(const uint8_t *data, size_t dataSize, Imag
bool ManagedTexture::LoadFromFile(const std::string &filename, ImageFileType type, bool generateMips) {
generateMips_ = generateMips;
size_t fileSize;
- uint8_t *buffer = VFSReadFile(filename.c_str(), &fileSize);
+ uint8_t *buffer = g_VFS.ReadFile(filename.c_str(), &fileSize);
if (!buffer) {
filename_.clear();
ERROR_LOG(IO, "Failed to read file '%s'", filename.c_str());
diff --git a/Common/Render/Text/draw_text.cpp b/Common/Render/Text/draw_text.cpp
index e236065483..4d4b58e7f9 100644
--- a/Common/Render/Text/draw_text.cpp
+++ b/Common/Render/Text/draw_text.cpp
@@ -38,7 +38,7 @@ void TextDrawer::SetFontScale(float xscale, float yscale) {
float TextDrawer::CalculateDPIScale() {
if (ignoreGlobalDpi_)
return dpiScale_;
- float scale = g_dpi_scale_y;
+ float scale = g_display.dpi_scale_y;
if (scale >= 1.0f) {
scale = 1.0f;
}
diff --git a/Common/Render/Text/draw_text_android.cpp b/Common/Render/Text/draw_text_android.cpp
index b6ada98a94..1cc3dc9d1e 100644
--- a/Common/Render/Text/draw_text_android.cpp
+++ b/Common/Render/Text/draw_text_android.cpp
@@ -27,7 +27,10 @@ TextDrawerAndroid::TextDrawerAndroid(Draw::DrawContext *draw) : TextDrawer(draw)
ERROR_LOG(G3D, "Failed to find class: '%s'", textRendererClassName);
}
dpiScale_ = CalculateDPIScale();
- INFO_LOG(G3D, "Initializing TextDrawerAndroid with DPI scale %f", dpiScale_);
+
+ use4444Format_ = (draw->GetDataFormatSupport(Draw::DataFormat::R4G4B4A4_UNORM_PACK16) & Draw::FMT_TEXTURE) != 0;
+
+ INFO_LOG(G3D, "Initializing TextDrawerAndroid with DPI scale %f, use4444=%d", dpiScale_, (int)use4444Format_);
}
TextDrawerAndroid::~TextDrawerAndroid() {
@@ -244,7 +247,8 @@ void TextDrawerAndroid::DrawString(DrawBuffer &target, const char *str, float x,
entry = iter->second.get();
entry->lastUsedFrame = frameCount_;
} else {
- DataFormat texFormat = Draw::DataFormat::R4G4B4A4_UNORM_PACK16;
+ // Actually, I don't know why we don't always use R8_UNORM..
+ DataFormat texFormat = use4444Format_ ? Draw::DataFormat::R4G4B4A4_UNORM_PACK16 : Draw::DataFormat::R8_UNORM;
entry = new TextStringEntry();
@@ -260,6 +264,7 @@ void TextDrawerAndroid::DrawString(DrawBuffer &target, const char *str, float x,
desc.depth = 1;
desc.mipLevels = 1;
desc.generateMips = false;
+ desc.swizzle = use4444Format_ ? Draw::TextureSwizzle::DEFAULT : Draw::TextureSwizzle::R8_AS_ALPHA,
desc.tag = "TextDrawer";
entry->texture = draw_->CreateTexture(desc);
cache_[key] = std::unique_ptr(entry);
diff --git a/Common/Render/Text/draw_text_android.h b/Common/Render/Text/draw_text_android.h
index 4b95bdeb75..f4b7e431e6 100644
--- a/Common/Render/Text/draw_text_android.h
+++ b/Common/Render/Text/draw_text_android.h
@@ -40,6 +40,7 @@ private:
jmethodID method_renderText;
uint32_t fontHash_;
+ bool use4444Format_ = false;
std::map fontMap_;
diff --git a/Common/RiscVCPUDetect.cpp b/Common/RiscVCPUDetect.cpp
index dfcaf74618..b4f8b608d8 100644
--- a/Common/RiscVCPUDetect.cpp
+++ b/Common/RiscVCPUDetect.cpp
@@ -40,6 +40,7 @@
const char procfile[] = "/proc/cpuinfo";
// https://www.kernel.org/doc/Documentation/ABI/testing/sysfs-devices-system-cpu
const char syscpupresentfile[] = "/sys/devices/system/cpu/present";
+const char firmwarefile[] = "/sys/firmware/devicetree/base/compatible";
class RiscVCPUInfoParser {
public:
@@ -49,9 +50,12 @@ public:
int TotalLogicalCount();
std::string ISAString();
+ bool FirmwareMatchesCompatible(const std::string &str);
private:
std::vector> cores_;
+ std::vector firmware_;
+ bool firmwareLoaded_ = false;
};
RiscVCPUInfoParser::RiscVCPUInfoParser() {
@@ -126,6 +130,25 @@ std::string RiscVCPUInfoParser::ISAString() {
return "Unknown";
}
+
+bool RiscVCPUInfoParser::FirmwareMatchesCompatible(const std::string &str) {
+ if (!firmwareLoaded_) {
+ firmwareLoaded_ = true;
+
+ std::string data;
+ if (!File::ReadFileToString(true, Path(firmwarefile), data))
+ return false;
+
+ SplitString(data, '\0', firmware_);
+ }
+
+ for (auto compatible : firmware_) {
+ if (compatible == str)
+ return true;
+ }
+
+ return false;
+}
#endif
static bool ExtensionSupported(unsigned long v, char c) {
@@ -170,6 +193,12 @@ void CPUInfo::Detect()
logical_cpu_count = 1;
truncate_cpy(cpu_string, parser.ISAString().c_str());
+
+ // A number of CPUs support a limited set of B. It's not all U74, so we use SOC for now...
+ if (parser.FirmwareMatchesCompatible("starfive,jh7110")) {
+ RiscV_Zba = true;
+ RiscV_Zbb = true;
+ }
#endif
unsigned long hwcap = getauxval(AT_HWCAP);
@@ -212,6 +241,10 @@ std::vector CPUInfo::Features() {
{ RiscV_C, "Compressed" },
{ RiscV_V, "Vector" },
{ RiscV_B, "Bitmanip" },
+ { RiscV_Zba, "Zba" },
+ { RiscV_Zbb, "Zbb" },
+ { RiscV_Zbc, "Zbc" },
+ { RiscV_Zbs, "Zbs" },
{ RiscV_Zicsr, "Zicsr" },
{ CPU64bit, "64-bit" },
};
diff --git a/Common/RiscVEmitter.cpp b/Common/RiscVEmitter.cpp
index bbac844fdc..1fa966e1fb 100644
--- a/Common/RiscVEmitter.cpp
+++ b/Common/RiscVEmitter.cpp
@@ -18,6 +18,9 @@
#include "ppsspp_config.h"
#include
#include
+#if PPSSPP_ARCH(RISCV64) && PPSSPP_PLATFORM(LINUX)
+#include
+#endif
#include "Common/BitScan.h"
#include "Common/CPUDetect.h"
#include "Common/RiscVEmitter.h"
@@ -58,8 +61,13 @@ static inline bool SupportsVector() {
}
static inline bool SupportsBitmanip(char zbx) {
- // TODO: Allow and detect sub-support?
- return cpu_info.RiscV_B;
+ switch (zbx) {
+ case 'a': return cpu_info.RiscV_B || cpu_info.RiscV_Zba;
+ case 'b': return cpu_info.RiscV_B || cpu_info.RiscV_Zbb;
+ case 'c': return cpu_info.RiscV_B || cpu_info.RiscV_Zbc;
+ case 's': return cpu_info.RiscV_B || cpu_info.RiscV_Zbs;
+ default: return false;
+ }
}
static inline bool SupportsFloatHalf(bool allowMin = false) {
@@ -1021,8 +1029,12 @@ void RiscVEmitter::ReserveCodeSpace(u32 bytes) {
_assert_msg_((bytes & 3) == 0 || SupportsCompressed(), "Code space should be aligned (no compressed)");
for (u32 i = 0; i < bytes / 4; i++)
EBREAK();
- if (bytes & 2)
- Write16(0);
+ if (bytes & 2) {
+ if (SupportsCompressed())
+ C_EBREAK();
+ else
+ Write16(0);
+ }
}
const u8 *RiscVEmitter::AlignCode16() {
@@ -1047,14 +1059,32 @@ void RiscVEmitter::FlushIcache() {
void RiscVEmitter::FlushIcacheSection(const u8 *start, const u8 *end) {
#if PPSSPP_ARCH(RISCV64)
+#if PPSSPP_PLATFORM(LINUX)
+ __riscv_flush_icache((char *)start, (char *)end, 0);
+#else
+ // TODO: This might only correspond to a local hart icache clear, which is no good.
__builtin___clear_cache((char *)start, (char *)end);
#endif
+#endif
+}
+
+FixupBranch::FixupBranch(FixupBranch &&other) {
+ ptr = other.ptr;
+ type = other.type;
+ other.ptr = nullptr;
}
FixupBranch::~FixupBranch() {
_assert_msg_(ptr == nullptr, "FixupBranch never set (left infinite loop)");
}
+FixupBranch &FixupBranch::operator =(FixupBranch &&other) {
+ ptr = other.ptr;
+ type = other.type;
+ other.ptr = nullptr;
+ return *this;
+}
+
void RiscVEmitter::SetJumpTarget(FixupBranch &branch) {
SetJumpTarget(branch, code_);
}
@@ -1675,6 +1705,10 @@ void RiscVEmitter::EBREAK() {
}
void RiscVEmitter::LWU(RiscVReg rd, RiscVReg rs1, s32 simm12) {
+ if (BitsSupported() == 32) {
+ LW(rd, rs1, simm12);
+ return;
+ }
_assert_msg_(BitsSupported() >= 64, "%s is only valid with R64I", __func__);
Write32(EncodeGI(Opcode32::LOAD, rd, Funct3::LS_WU, rs1, simm12));
}
@@ -3878,67 +3912,67 @@ void RiscVEmitter::REV8(RiscVReg rd, RiscVReg rs) {
void RiscVEmitter::CLMUL(RiscVReg rd, RiscVReg rs1, RiscVReg rs2) {
_assert_msg_(rd != R_ZERO, "%s should avoid write to zero", __func__);
- _assert_msg_(SupportsBitmanip('b'), "%s instruction unsupported without B", __func__);
+ _assert_msg_(SupportsBitmanip('c'), "%s instruction unsupported without B", __func__);
Write32(EncodeGR(Opcode32::OP, rd, Funct3::CLMUL, rs1, rs2, Funct7::MINMAX_CLMUL));
}
void RiscVEmitter::CLMULH(RiscVReg rd, RiscVReg rs1, RiscVReg rs2) {
_assert_msg_(rd != R_ZERO, "%s should avoid write to zero", __func__);
- _assert_msg_(SupportsBitmanip('b'), "%s instruction unsupported without B", __func__);
+ _assert_msg_(SupportsBitmanip('c'), "%s instruction unsupported without B", __func__);
Write32(EncodeGR(Opcode32::OP, rd, Funct3::CLMULH, rs1, rs2, Funct7::MINMAX_CLMUL));
}
void RiscVEmitter::CLMULR(RiscVReg rd, RiscVReg rs1, RiscVReg rs2) {
_assert_msg_(rd != R_ZERO, "%s should avoid write to zero", __func__);
- _assert_msg_(SupportsBitmanip('b'), "%s instruction unsupported without B", __func__);
+ _assert_msg_(SupportsBitmanip('c'), "%s instruction unsupported without B", __func__);
Write32(EncodeGR(Opcode32::OP, rd, Funct3::CLMULR, rs1, rs2, Funct7::MINMAX_CLMUL));
}
void RiscVEmitter::BCLR(RiscVReg rd, RiscVReg rs1, RiscVReg rs2) {
_assert_msg_(rd != R_ZERO, "%s should avoid write to zero", __func__);
- _assert_msg_(SupportsBitmanip('b'), "%s instruction unsupported without B", __func__);
+ _assert_msg_(SupportsBitmanip('s'), "%s instruction unsupported without B", __func__);
Write32(EncodeGR(Opcode32::OP, rd, Funct3::BSET, rs1, rs2, Funct7::BCLREXT));
}
void RiscVEmitter::BCLRI(RiscVReg rd, RiscVReg rs1, u32 shamt) {
_assert_msg_(rd != R_ZERO, "%s should avoid write to zero", __func__);
- _assert_msg_(SupportsBitmanip('b'), "%s instruction unsupported without B", __func__);
+ _assert_msg_(SupportsBitmanip('s'), "%s instruction unsupported without B", __func__);
Write32(EncodeGIShift(Opcode32::OP_IMM, rd, Funct3::BSET, rs1, shamt, Funct7::BCLREXT));
}
void RiscVEmitter::BEXT(RiscVReg rd, RiscVReg rs1, RiscVReg rs2) {
_assert_msg_(rd != R_ZERO, "%s should avoid write to zero", __func__);
- _assert_msg_(SupportsBitmanip('b'), "%s instruction unsupported without B", __func__);
+ _assert_msg_(SupportsBitmanip('s'), "%s instruction unsupported without B", __func__);
Write32(EncodeGR(Opcode32::OP, rd, Funct3::BEXT, rs1, rs2, Funct7::BCLREXT));
}
void RiscVEmitter::BEXTI(RiscVReg rd, RiscVReg rs1, u32 shamt) {
_assert_msg_(rd != R_ZERO, "%s should avoid write to zero", __func__);
- _assert_msg_(SupportsBitmanip('b'), "%s instruction unsupported without B", __func__);
+ _assert_msg_(SupportsBitmanip('s'), "%s instruction unsupported without B", __func__);
Write32(EncodeGIShift(Opcode32::OP_IMM, rd, Funct3::BEXT, rs1, shamt, Funct7::BCLREXT));
}
void RiscVEmitter::BINV(RiscVReg rd, RiscVReg rs1, RiscVReg rs2) {
_assert_msg_(rd != R_ZERO, "%s should avoid write to zero", __func__);
- _assert_msg_(SupportsBitmanip('b'), "%s instruction unsupported without B", __func__);
+ _assert_msg_(SupportsBitmanip('s'), "%s instruction unsupported without B", __func__);
Write32(EncodeGR(Opcode32::OP, rd, Funct3::BSET, rs1, rs2, Funct7::BINV_REV));
}
void RiscVEmitter::BINVI(RiscVReg rd, RiscVReg rs1, u32 shamt) {
_assert_msg_(rd != R_ZERO, "%s should avoid write to zero", __func__);
- _assert_msg_(SupportsBitmanip('b'), "%s instruction unsupported without B", __func__);
+ _assert_msg_(SupportsBitmanip('s'), "%s instruction unsupported without B", __func__);
Write32(EncodeGIShift(Opcode32::OP_IMM, rd, Funct3::BSET, rs1, shamt, Funct7::BINV_REV));
}
void RiscVEmitter::BSET(RiscVReg rd, RiscVReg rs1, RiscVReg rs2) {
_assert_msg_(rd != R_ZERO, "%s should avoid write to zero", __func__);
- _assert_msg_(SupportsBitmanip('b'), "%s instruction unsupported without B", __func__);
+ _assert_msg_(SupportsBitmanip('s'), "%s instruction unsupported without B", __func__);
Write32(EncodeGR(Opcode32::OP, rd, Funct3::BSET, rs1, rs2, Funct7::BSET_ORC));
}
void RiscVEmitter::BSETI(RiscVReg rd, RiscVReg rs1, u32 shamt) {
_assert_msg_(rd != R_ZERO, "%s should avoid write to zero", __func__);
- _assert_msg_(SupportsBitmanip('b'), "%s instruction unsupported without B", __func__);
+ _assert_msg_(SupportsBitmanip('s'), "%s instruction unsupported without B", __func__);
Write32(EncodeGIShift(Opcode32::OP_IMM, rd, Funct3::BSET, rs1, shamt, Funct7::BSET_ORC));
}
@@ -4278,4 +4312,18 @@ void RiscVEmitter::C_SDSP(RiscVReg rs2, u32 uimm9) {
Write16(EncodeCSS(Opcode16::C2, rs2, imm5_4_3_8_7_6, Funct3::C_SDSP));
}
+void RiscVCodeBlock::PoisonMemory(int offset) {
+ // So we can adjust region to writable space. Might be zero.
+ ptrdiff_t writable = writable_ - code_;
+
+ u32 *ptr = (u32 *)(region + offset + writable);
+ u32 *maxptr = (u32 *)(region + region_size - offset + writable);
+ // This will only write an even multiple of u32, but not much else to do.
+ // RiscV: 0x00100073 = EBREAK, 0x9002 = C.EBREAK
+ while (ptr + 1 <= maxptr)
+ *ptr++ = 0x00100073;
+ if (SupportsCompressed() && ptr < maxptr && (intptr_t)maxptr - (intptr_t)ptr >= 2)
+ *(u16 *)ptr = 0x9002;
+}
+
};
diff --git a/Common/RiscVEmitter.h b/Common/RiscVEmitter.h
index 5e9b0b9f16..720db3faac 100644
--- a/Common/RiscVEmitter.h
+++ b/Common/RiscVEmitter.h
@@ -47,6 +47,8 @@ enum RiscVReg {
V8, V9, V10, V11, V12, V13, V14, V15,
V16, V17, V18, V19, V20, V21, V22, V23,
V24, V25, V26, V27, V28, V29, V30, V31,
+
+ INVALID_REG = 0xFFFFFFFF,
};
enum class FixupBranchType {
@@ -180,8 +182,13 @@ enum class VUseMask {
struct FixupBranch {
FixupBranch() {}
FixupBranch(const u8 *p, FixupBranchType t) : ptr(p), type(t) {}
+ FixupBranch(FixupBranch &&other);
+ FixupBranch(const FixupBranch &) = delete;
~FixupBranch();
+ FixupBranch &operator =(FixupBranch &&other);
+ FixupBranch &operator =(const FixupBranch &other) = delete;
+
const u8 *ptr = nullptr;
FixupBranchType type = FixupBranchType::B;
};
@@ -302,7 +309,7 @@ public:
void ECALL();
void EBREAK();
- // 64-bit instructions - oens ending in W sign extend result to 32 bits.
+ // 64-bit instructions - ones ending in W sign extend result to 32 bits.
void LWU(RiscVReg rd, RiscVReg addr, s32 simm12);
void LD(RiscVReg rd, RiscVReg addr, s32 simm12);
void SD(RiscVReg rs2, RiscVReg addr, s32 simm12);
@@ -1024,13 +1031,14 @@ private:
writable_ += 2;
}
+protected:
const u8 *code_ = nullptr;
u8 *writable_ = nullptr;
const u8 *lastCacheFlushEnd_ = nullptr;
bool autoCompress_ = false;
};
-class MIPSCodeBlock : public CodeBlock {
+class RiscVCodeBlock : public CodeBlock {
private:
void PoisonMemory(int offset) override;
};
diff --git a/Common/Serialize/Serializer.h b/Common/Serialize/Serializer.h
index d91b9dfb2f..a7fa964a0c 100644
--- a/Common/Serialize/Serializer.h
+++ b/Common/Serialize/Serializer.h
@@ -192,21 +192,8 @@ public:
return (size_t)ptr;
}
- // Expects ptr to have at least MeasurePtr bytes at ptr.
- template
- static Error SavePtr(u8 *ptr, T &_class, size_t expected_size)
- {
- const u8 *expected_end = ptr + expected_size;
- PointerWrap p(&ptr, PointerWrap::MODE_WRITE);
- _class.DoState(p);
-
- if (p.error != PointerWrap::ERROR_FAILURE && (expected_end == ptr || expected_size == 0)) {
- return ERROR_NONE;
- } else {
- return ERROR_BROKEN_STATE;
- }
- }
-
+ // If *saved is null, will allocate storage using malloc.
+ // If it's not null, it will be used, but only hope can save you from overruns at the end. For libretro.
template
static Error MeasureAndSavePtr(T &_class, u8 **saved, size_t *savedSize)
{
@@ -216,9 +203,14 @@ public:
_assert_(p.error == PointerWrap::ERROR_NONE);
size_t measuredSize = p.Offset();
- u8 *data = (u8 *)malloc(measuredSize);
- if (!data)
- return ERROR_BAD_ALLOC;
+ u8 *data;
+ if (*saved) {
+ data = *saved;
+ } else {
+ data = (u8 *)malloc(measuredSize);
+ if (!data)
+ return ERROR_BAD_ALLOC;
+ }
p.RewindForWrite(data);
_class.DoState(p);
@@ -228,11 +220,37 @@ public:
*savedSize = measuredSize;
return ERROR_NONE;
} else {
- free(data);
+ if (!*saved) {
+ free(data);
+ }
return ERROR_BROKEN_STATE;
}
}
+ // Duplicate of the above but takes and modifies a vector. Less invasive
+ // than modifying the rewind manager to keep things in something else than vectors.
+ template
+ static Error MeasureAndSavePtr(T &_class, std::vector *saved)
+ {
+ u8 *ptr = nullptr;
+ PointerWrap p(&ptr, PointerWrap::MODE_MEASURE);
+ _class.DoState(p);
+ _assert_(p.error == PointerWrap::ERROR_NONE);
+
+ size_t measuredSize = p.Offset();
+ saved->resize(measuredSize);
+ u8 *data = saved->data();
+ p.RewindForWrite(data);
+ _class.DoState(p);
+ if (p.CheckAfterWrite()) {
+ return ERROR_NONE;
+ } else {
+ saved->clear();
+ return ERROR_BROKEN_STATE;
+ }
+ }
+
+
// Load file template
template
static Error Load(const Path &filename, std::string *gitVersion, T& _class, std::string *failureReason)
@@ -257,8 +275,8 @@ public:
template
static Error Save(const Path &filename, const std::string &title, const char *gitVersion, T& _class)
{
- u8 *buffer;
- size_t sz;
+ u8 *buffer = nullptr;
+ size_t sz = 0;
Error error = MeasureAndSavePtr(_class, &buffer, &sz);
// SaveFile takes ownership of buffer (malloc/free)
diff --git a/Common/System/Display.cpp b/Common/System/Display.cpp
index 0e9c1ffb62..15af7de4e3 100644
--- a/Common/System/Display.cpp
+++ b/Common/System/Display.cpp
@@ -1,27 +1,13 @@
+#include
+
#include "Common/System/Display.h"
#include "Common/Math/math_util.h"
-int dp_xres;
-int dp_yres;
-
-int pixel_xres;
-int pixel_yres;
-
-float g_dpi = 1.0f; // will be overwritten with a value that makes sense.
-float g_dpi_scale_x = 1.0f;
-float g_dpi_scale_y = 1.0f;
-float g_dpi_scale_real_x = 1.0f;
-float g_dpi_scale_real_y = 1.0f;
-float pixel_in_dps_x = 1.0f;
-float pixel_in_dps_y = 1.0f;
-float display_hz = 60.0f;
-
-DisplayRotation g_display_rotation;
-Lin::Matrix4x4 g_display_rot_matrix = Lin::Matrix4x4::identity();
+DisplayProperties g_display;
template
void RotateRectToDisplayImpl(DisplayRect &rect, T curRTWidth, T curRTHeight) {
- switch (g_display_rotation) {
+ switch (g_display.rotation) {
case DisplayRotation::ROTATE_180:
rect.x = curRTWidth - rect.w - rect.x;
rect.y = curRTHeight - rect.h - rect.y;
@@ -62,3 +48,21 @@ void RotateRectToDisplay(DisplayRect &rect, int curRTWidth, int curRTHeight
void RotateRectToDisplay(DisplayRect &rect, float curRTWidth, float curRTHeight) {
RotateRectToDisplayImpl(rect, curRTWidth, curRTHeight);
}
+
+DisplayProperties::DisplayProperties() {
+ rot_matrix.setIdentity();
+}
+
+void DisplayProperties::Print() {
+ printf("dp_xres/yres: %d, %d\n", dp_xres, dp_yres);
+ printf("pixel_xres/yres: %d, %d\n", pixel_xres, pixel_yres);
+
+ printf("dpi, x, y: %f, %f, %f\n", dpi, dpi_scale_x, dpi_scale_y);
+ printf("pixel_in_dps: %f, %f\n", pixel_in_dps_x, pixel_in_dps_y);
+
+ printf("dpi_real: %f, %f\n", dpi_scale_real_x, dpi_scale_real_y);
+ printf("display_hz: %f\n", display_hz);
+
+ printf("rotation: %d\n", (int)rotation);
+ rot_matrix.print();
+}
diff --git a/Common/System/Display.h b/Common/System/Display.h
index 586c346f77..0bea49542a 100644
--- a/Common/System/Display.h
+++ b/Common/System/Display.h
@@ -5,20 +5,6 @@
// This is meant to be a framework for handling DPI scaling etc.
// For now, it just consists of these ugly globals.
-extern int dp_xres;
-extern int dp_yres;
-extern int pixel_xres;
-extern int pixel_yres;
-
-extern float g_dpi;
-extern float g_dpi_scale_x;
-extern float g_dpi_scale_y;
-extern float g_dpi_scale_real_x;
-extern float g_dpi_scale_real_y;
-extern float pixel_in_dps_x;
-extern float pixel_in_dps_y;
-extern float display_hz;
-
// On some platforms (currently only Windows UWP) we need to manually rotate
// our rendered output to match the display. Use these to do so.
enum class DisplayRotation {
@@ -28,8 +14,34 @@ enum class DisplayRotation {
ROTATE_270,
};
-extern DisplayRotation g_display_rotation;
-extern Lin::Matrix4x4 g_display_rot_matrix;
+struct DisplayProperties {
+ int dp_xres;
+ int dp_yres;
+ int pixel_xres;
+ int pixel_yres;
+
+ float dpi = 1.0f; // will be overwritten with a value that makes sense.
+ float dpi_scale_x = 1.0f;
+ float dpi_scale_y = 1.0f;
+
+ // pixel_xres/yres in dps
+ float pixel_in_dps_x = 1.0f;
+ float pixel_in_dps_y = 1.0f;
+
+ // If DPI is overridden (like in small window mode), these are still the original DPI.
+ float dpi_scale_real_x = 1.0f;
+ float dpi_scale_real_y = 1.0f;
+
+ float display_hz = 60.0f;
+
+ DisplayRotation rotation;
+ Lin::Matrix4x4 rot_matrix;
+
+ DisplayProperties();
+ void Print();
+};
+
+extern DisplayProperties g_display;
template
struct DisplayRect {
diff --git a/Common/System/NativeApp.h b/Common/System/NativeApp.h
index 6788642eb4..bb0c7af962 100644
--- a/Common/System/NativeApp.h
+++ b/Common/System/NativeApp.h
@@ -23,9 +23,6 @@ void NativeGetAppInfo(std::string *app_dir_name, std::string *app_nice_name, boo
// Generic host->C++ messaging, used for functionality like system-native popup input boxes.
void NativeMessageReceived(const char *message, const char *value);
-// This is used to communicate back and thread requested input box strings.
-void NativeInputBoxReceived(std::function cb, bool result, const std::string &value);
-
// Easy way for the Java side to ask the C++ side for configuration options, such as
// the rotation lock which must be controlled from Java on Android.
// It is currently not called on non-Android platforms.
diff --git a/Common/System/Request.cpp b/Common/System/Request.cpp
new file mode 100644
index 0000000000..13e12f1502
--- /dev/null
+++ b/Common/System/Request.cpp
@@ -0,0 +1,82 @@
+#include "Common/System/Request.h"
+#include "Common/System/System.h"
+#include "Common/Log.h"
+
+RequestManager g_requestManager;
+
+const char *RequestTypeAsString(SystemRequestType type) {
+ switch (type) {
+ case SystemRequestType::INPUT_TEXT_MODAL: return "INPUT_TEXT_MODAL";
+ case SystemRequestType::BROWSE_FOR_IMAGE: return "BROWSE_FOR_IMAGE";
+ case SystemRequestType::BROWSE_FOR_FILE: return "BROWSE_FOR_FILE";
+ case SystemRequestType::BROWSE_FOR_FOLDER: return "BROWSE_FOR_FOLDER";
+ default: return "N/A";
+ }
+}
+
+bool RequestManager::MakeSystemRequest(SystemRequestType type, RequestCallback callback, const std::string ¶m1, const std::string ¶m2, int param3) {
+ int requestId = idCounter_++;
+
+ // NOTE: We need to register immediately, in order to support synchronous implementations.
+ {
+ std::lock_guard guard(callbackMutex_);
+ callbackMap_[requestId] = callback;
+ }
+
+ INFO_LOG(SYSTEM, "Making system request %s: id %d, callback_valid %d", RequestTypeAsString(type), requestId, callback != nullptr);
+ if (!System_MakeRequest(type, requestId, param1, param2, param3)) {
+ {
+ std::lock_guard guard(callbackMutex_);
+ callbackMap_.erase(requestId);
+ }
+ return false;
+ }
+
+ return true;
+}
+
+void RequestManager::PostSystemSuccess(int requestId, const char *responseString, int responseValue) {
+ std::lock_guard guard(callbackMutex_);
+ auto iter = callbackMap_.find(requestId);
+ if (iter == callbackMap_.end()) {
+ ERROR_LOG(SYSTEM, "PostSystemSuccess: Unexpected request ID %d (responseString=%s)", requestId, responseString);
+ return;
+ }
+
+ std::lock_guard responseGuard(responseMutex_);
+ PendingResponse response;
+ response.callback = iter->second;
+ response.responseString = responseString;
+ response.responseValue = responseValue;
+ pendingResponses_.push_back(response);
+ INFO_LOG(SYSTEM, "PostSystemSuccess: Request %d (%s, %d)", requestId, responseString, responseValue);
+}
+
+void RequestManager::PostSystemFailure(int requestId) {
+ std::lock_guard guard(callbackMutex_);
+ auto iter = callbackMap_.find(requestId);
+ if (iter == callbackMap_.end()) {
+ ERROR_LOG(SYSTEM, "PostSystemFailure: Unexpected request ID %d", requestId);
+ return;
+ }
+ INFO_LOG(SYSTEM, "PostSystemFailure: Request %d failed", requestId);
+ callbackMap_.erase(iter);
+}
+
+void RequestManager::ProcessRequests() {
+ std::lock_guard guard(responseMutex_);
+ for (auto &iter : pendingResponses_) {
+ if (iter.callback) {
+ iter.callback(iter.responseString.c_str(), iter.responseValue);
+ }
+ }
+ pendingResponses_.clear();
+}
+
+void RequestManager::Clear() {
+ std::lock_guard guard(callbackMutex_);
+ std::lock_guard responseGuard(responseMutex_);
+
+ pendingResponses_.clear();
+ callbackMap_.clear();
+}
diff --git a/Common/System/Request.h b/Common/System/Request.h
new file mode 100644
index 0000000000..745f4eb555
--- /dev/null
+++ b/Common/System/Request.h
@@ -0,0 +1,123 @@
+#pragma once
+
+#include
+#include
+#include