mirror of
https://github.com/PCSX2/pcsx2.git
synced 2026-08-26 07:34:35 +02:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
3e631e047f | ||
|
|
dd2d4edffc | ||
|
|
fd2960c9cb | ||
|
|
c2907ea58f | ||
|
|
41f62cf53d | ||
|
|
e462f1ff9c | ||
|
|
5b0b6191d8 | ||
|
|
ab9a1e4307 | ||
|
|
029c11c8d2 | ||
|
|
e221d31b45 | ||
|
|
dfbdaa651c | ||
|
|
9de152b8ee | ||
|
|
360f9afb70 | ||
|
|
1b81825218 | ||
|
|
b3e6e28827 | ||
|
|
eaceb27879 | ||
|
|
cd9b6c7ac3 | ||
|
|
d3e527f2a4 | ||
|
|
b47fdcdfab | ||
|
|
2550ad7fd1 | ||
|
|
1717f584a0 | ||
|
|
0822d3e3e5 | ||
|
|
ec41af760a | ||
|
|
a5ed24ca88 | ||
|
|
b3697579c0 | ||
|
|
388da2058b | ||
|
|
aa9a0dca4b | ||
|
|
2a892da0da | ||
|
|
9237bf9429 | ||
|
|
10533dce02 | ||
|
|
b5ebc19eff | ||
|
|
6535e7e43a | ||
|
|
5f9473ef02 | ||
|
|
fbb1c7cb8e | ||
|
|
2d97d85ca5 | ||
|
|
ecd7d0fc35 | ||
|
|
0c389789f3 | ||
|
|
d8239664a8 |
+2
-2
@@ -39,8 +39,8 @@
|
||||
'Debugger':
|
||||
- 'pcsx2/DebugTools/*'
|
||||
- 'pcsx2/DebugTools/**/*'
|
||||
- 'pcsx2/gui/Debugger/*'
|
||||
- 'pcsx2/gui/Debugger/**/*'
|
||||
- 'pcsx2-qt/Debugger/*'
|
||||
- 'pcsx2-qt/Debugger/**/*'
|
||||
'IPC':
|
||||
- 'pcsx2/IPC*'
|
||||
- 'pcsx2/**/IPC*'
|
||||
|
||||
@@ -158,6 +158,7 @@ declare -a SYSLIBS=(
|
||||
"libhx509.so.5"
|
||||
"libsqlite3.so.0"
|
||||
"libcrypt.so.1"
|
||||
"libdbus-1.so.3"
|
||||
)
|
||||
|
||||
declare -a DEPLIBS=(
|
||||
|
||||
@@ -1372,6 +1372,7 @@ SCAJ-20172:
|
||||
region: "NTSC-Unk"
|
||||
gsHWFixes:
|
||||
getSkipCount: "GSC_FFXGames"
|
||||
partialTargetInvalidation: 1 # Fixes broken textures.
|
||||
SCAJ-20173:
|
||||
name: "Ace Combat Zero - The Belkan War"
|
||||
region: "NTSC-Unk"
|
||||
@@ -1457,6 +1458,7 @@ SCAJ-20188:
|
||||
region: "NTSC-Unk"
|
||||
gsHWFixes:
|
||||
getSkipCount: "GSC_FFXGames"
|
||||
partialTargetInvalidation: 1 # Fixes broken textures.
|
||||
SCAJ-20190:
|
||||
name: "God of War II"
|
||||
region: "NTSC-Unk"
|
||||
@@ -5415,6 +5417,9 @@ SCKA-20072:
|
||||
SCKA-20073:
|
||||
name: "Final Fantasy XII"
|
||||
region: "NTSC-K"
|
||||
gsHWFixes:
|
||||
getSkipCount: "GSC_FFXGames"
|
||||
partialTargetInvalidation: 1 # Fixes broken textures.
|
||||
SCKA-20078:
|
||||
name: "Killzone [PlayStation 2 Big Hit Series]"
|
||||
region: "NTSC-K"
|
||||
@@ -5547,6 +5552,12 @@ SCKA-20132:
|
||||
name: "Shin Megami Tensei - Persona 4"
|
||||
region: "NTSC-K"
|
||||
compat: 5
|
||||
SCKA-20138:
|
||||
name: "Final Fantasy XII [Ultimate Hits International Zodiac Job System]"
|
||||
region: "NTSC-K"
|
||||
gsHWFixes:
|
||||
getSkipCount: "GSC_FFXGames"
|
||||
partialTargetInvalidation: 1 # Fixes broken textures.
|
||||
SCKA-24008:
|
||||
name: "SOCOM - U.S. Navy SEALs"
|
||||
region: "NTSC-K"
|
||||
@@ -20323,27 +20334,32 @@ SLES-54354:
|
||||
compat: 5
|
||||
gsHWFixes:
|
||||
getSkipCount: "GSC_FFXGames"
|
||||
partialTargetInvalidation: 1 # Fixes broken textures.
|
||||
SLES-54355:
|
||||
name: "Final Fantasy XII"
|
||||
region: "PAL-F"
|
||||
gsHWFixes:
|
||||
getSkipCount: "GSC_FFXGames"
|
||||
partialTargetInvalidation: 1 # Fixes broken textures.
|
||||
SLES-54356:
|
||||
name: "Final Fantasy XII"
|
||||
region: "PAL-G"
|
||||
gsHWFixes:
|
||||
getSkipCount: "GSC_FFXGames"
|
||||
partialTargetInvalidation: 1 # Fixes broken textures.
|
||||
SLES-54357:
|
||||
name: "Final Fantasy XII"
|
||||
region: "PAL-I"
|
||||
gsHWFixes:
|
||||
getSkipCount: "GSC_FFXGames"
|
||||
partialTargetInvalidation: 1 # Fixes broken textures.
|
||||
SLES-54358:
|
||||
name: "Final Fantasy XII"
|
||||
region: "PAL-S"
|
||||
compat: 5
|
||||
gsHWFixes:
|
||||
getSkipCount: "GSC_FFXGames"
|
||||
partialTargetInvalidation: 1 # Fixes broken textures.
|
||||
SLES-54359:
|
||||
name: "Legend of Spyro, The - A New Beginning"
|
||||
region: "PAL-M6"
|
||||
@@ -25330,6 +25346,12 @@ SLPM-55015:
|
||||
SLPM-55016:
|
||||
name: "Yotsunoha - A Journey of Sincerity"
|
||||
region: "NTSC-J"
|
||||
SLPM-55022:
|
||||
name: "Final Fantasy XII [Ultimate Hits]"
|
||||
region: "NTSC-J"
|
||||
gsHWFixes:
|
||||
getSkipCount: "GSC_FFXGames"
|
||||
partialTargetInvalidation: 1 # Fixes broken textures.
|
||||
SLPM-55024:
|
||||
name: "Jikkyou Powerful Pro Yakyuu 15"
|
||||
region: "NTSC-J"
|
||||
@@ -25672,6 +25694,7 @@ SLPM-55210:
|
||||
region: "NTSC-J"
|
||||
gsHWFixes:
|
||||
getSkipCount: "GSC_FFXGames"
|
||||
partialTargetInvalidation: 1 # Fixes broken textures.
|
||||
SLPM-55211:
|
||||
name: "Pachislot Higurashi no Naku Koro ni Matsuri"
|
||||
region: "NTSC-J"
|
||||
@@ -28244,6 +28267,8 @@ SLPM-62713:
|
||||
SLPM-62714:
|
||||
name: "Shin Sangoku Musou Series Collection Joukan"
|
||||
region: "NTSC-J"
|
||||
gsHWFixes:
|
||||
partialTargetInvalidation: 1 # Fixes invisible text.
|
||||
SLPM-62717:
|
||||
name: "Sega Ages 2500 Series Vol.26 - Dynamite Deka"
|
||||
region: "NTSC-J"
|
||||
@@ -33266,6 +33291,7 @@ SLPM-66320:
|
||||
compat: 5
|
||||
gsHWFixes:
|
||||
getSkipCount: "GSC_FFXGames"
|
||||
partialTargetInvalidation: 1 # Fixes broken textures.
|
||||
SLPM-66321:
|
||||
name: "Kurogane no Houkou - Warship Gunner 2"
|
||||
region: "NTSC-J"
|
||||
@@ -34993,6 +35019,7 @@ SLPM-66750:
|
||||
compat: 5
|
||||
gsHWFixes:
|
||||
getSkipCount: "GSC_FFXGames"
|
||||
partialTargetInvalidation: 1 # Fixes broken textures.
|
||||
SLPM-66751:
|
||||
name: "Mahoroba Stories"
|
||||
region: "NTSC-J"
|
||||
@@ -46529,6 +46556,7 @@ SLUS-20963:
|
||||
compat: 5
|
||||
gsHWFixes:
|
||||
getSkipCount: "GSC_FFXGames"
|
||||
partialTargetInvalidation: 1 # Fixes broken textures.
|
||||
SLUS-20964:
|
||||
name: "Devil May Cry 3 - Dante's Awakening"
|
||||
region: "NTSC-U"
|
||||
@@ -49450,6 +49478,7 @@ SLUS-21475:
|
||||
compat: 5
|
||||
gsHWFixes:
|
||||
getSkipCount: "GSC_FFXGames"
|
||||
partialTargetInvalidation: 1 # Fixes broken textures.
|
||||
SLUS-21476:
|
||||
name: "Madden NFL '07"
|
||||
region: "NTSC-U"
|
||||
@@ -52322,6 +52351,7 @@ SLUS-29171:
|
||||
region: "NTSC-U"
|
||||
gsHWFixes:
|
||||
getSkipCount: "GSC_FFXGames"
|
||||
partialTargetInvalidation: 1 # Fixes broken textures.
|
||||
SLUS-29172:
|
||||
name: "Battlefield 2 - Modern Combat [Demo]"
|
||||
region: "NTSC-U"
|
||||
|
||||
@@ -537,7 +537,7 @@
|
||||
030000009b2800003200000000000000,Raphnet GC and N64 Adapter,a:b0,b:b7,dpdown:b11,dpleft:b12,dpright:b13,dpup:b10,lefttrigger:+a5,leftx:a0,lefty:a1,rightshoulder:b2,righttrigger:+a2,rightx:a3,righty:a4,start:b3,x:b1,y:b8,platform:Windows,
|
||||
030000009b2800006000000000000000,Raphnet GC and N64 Adapter,a:b0,b:b7,dpdown:b11,dpleft:b12,dpright:b13,dpup:b10,lefttrigger:+a5,leftx:a0,lefty:a1,rightshoulder:b2,righttrigger:+a2,rightx:a3,righty:a4,start:b3,x:b1,y:b8,platform:Windows,
|
||||
030000009b2800001800000000000000,Raphnet Jaguar Adapter,a:b2,b:b1,back:b4,dpdown:h0.4,dpleft:h0.8,dpright:h0.2,dpup:h0.1,leftshoulder:b13,lefttrigger:b8,leftx:a0,lefty:a1,rightshoulder:b0,righttrigger:b10,start:b3,x:b11,y:b12,platform:Windows,
|
||||
030000009b2800006300000000000000,Raphnet N64 Adapter,a:b0,b:b1,start:b3,lefttrigger:b2,dpup:b10,dpleft:b12,dpdown:b11,dpright:b13,leftx:a0,lefty:a1,-rightx:b8,+rightx:b9,-righty:b6,+righty:b7,leftshoulder:b4,rightshoulder:b5,platform:Windows,
|
||||
030000009b2800006300000000000000,Raphnet N64 Adapter,+rightx:b9,+righty:b7,-rightx:b8,-righty:b6,a:b0,b:b1,dpdown:b11,dpleft:b12,dpright:b13,dpup:b10,leftshoulder:b4,lefttrigger:b2,leftx:a0,lefty:a1,rightshoulder:b5,start:b3,platform:Windows,
|
||||
030000009b2800000200000000000000,Raphnet NES Adapter,a:b7,b:b6,back:b5,dpdown:h0.4,dpleft:h0.8,dpright:h0.2,dpup:h0.1,leftx:a0,lefty:a1,start:b4,platform:Windows,
|
||||
030000009b2800004400000000000000,Raphnet PS1 and PS2 Adapter,a:b1,b:b2,back:b5,dpdown:b13,dpleft:b14,dpright:b15,dpup:b12,leftshoulder:b6,leftstick:b10,lefttrigger:b8,leftx:a0,lefty:a1,rightshoulder:b7,rightstick:b11,righttrigger:b9,rightx:a3,righty:a4,start:b4,x:b0,y:b3,platform:Windows,
|
||||
030000009b2800004300000000000000,Raphnet Saturn,a:b0,b:b1,dpdown:b13,dpleft:b14,dpright:b15,dpup:b12,leftshoulder:b6,lefttrigger:b7,leftx:a0,lefty:a1,rightshoulder:b5,righttrigger:b2,start:b8,x:b3,y:b4,platform:Windows,
|
||||
@@ -1154,6 +1154,7 @@ xinput,XInput Controller,a:b0,b:b1,back:b6,dpdown:h0.4,dpleft:h0.8,dpright:h0.2,
|
||||
03000000bc2000000055000011010000,GameSir G3w,a:b0,b:b1,back:b10,dpdown:h0.4,dpleft:h0.8,dpright:h0.2,dpup:h0.1,leftshoulder:b6,leftstick:b13,lefttrigger:a5,leftx:a0,lefty:a1,rightshoulder:b7,rightstick:b14,righttrigger:a4,rightx:a2,righty:a3,start:b11,x:b3,y:b4,platform:Linux,
|
||||
03000000558500001b06000010010000,GameSir G4 Pro,a:b0,b:b1,back:b10,dpdown:h0.4,dpleft:h0.8,dpright:h0.2,dpup:h0.1,leftshoulder:b6,leftstick:b13,lefttrigger:b8,leftx:a0,lefty:a1,rightshoulder:b7,rightstick:b14,righttrigger:b9,rightx:a2,righty:a3,start:b11,x:b3,y:b4,platform:Linux,
|
||||
05000000ac0500002d0200001b010000,GameSir G4s,a:b0,b:b1,back:b10,dpdown:h0.4,dpleft:h0.8,dpright:h0.2,dpup:h0.1,guide:b33,leftshoulder:b6,leftstick:b13,lefttrigger:a5,leftx:a0,lefty:a1,rightshoulder:b7,rightstick:b14,righttrigger:a4,rightx:a2,righty:a3,start:b11,x:b3,y:b4,platform:Linux,
|
||||
03000000ac0500007a05000011010000,GameSir G5,a:b0,b:b1,back:b10,dpdown:h0.4,dpleft:h0.8,dpright:h0.2,dpup:h0.1,guide:b12,leftshoulder:b6,leftstick:b13,lefttrigger:b8,leftx:a0,lefty:a1,rightshoulder:b7,rightstick:b16,righttrigger:b9,rightx:a2,righty:a3,start:b11,x:b3,y:b4,platform:Linux,
|
||||
03000000bc2000005656000011010000,GameSir T4w,a:b1,b:b2,back:b8,dpdown:h0.4,dpleft:h0.8,dpright:h0.2,dpup:h0.1,guide:b12,leftshoulder:b4,leftstick:b10,lefttrigger:b6,leftx:a0,lefty:a1,rightshoulder:b5,rightstick:b11,righttrigger:b7,rightx:a2,righty:a3,start:b9,x:b0,y:b3,platform:Linux,
|
||||
03000000ac0500001a06000011010000,GameSir-T3 2.02,a:b0,b:b1,back:b10,dpdown:h0.4,dpleft:h0.8,dpright:h0.2,dpup:h0.1,guide:b15,leftshoulder:b6,leftstick:b13,lefttrigger:a5,leftx:a0,lefty:a1,rightshoulder:b7,rightstick:b14,righttrigger:a4,rightx:a2,righty:a3,start:b11,x:b3,y:b4,platform:Linux,
|
||||
0500000047532047616d657061640000,GameStop Gamepad,a:b0,b:b1,back:b8,dpdown:h0.4,dpleft:h0.8,dpright:h0.2,dpup:h0.1,leftshoulder:b4,leftstick:b10,lefttrigger:b6,leftx:a0,lefty:a1,rightshoulder:b5,rightstick:b11,righttrigger:b7,rightx:a2,righty:a3,start:b9,x:b2,y:b3,platform:Linux,
|
||||
@@ -1334,9 +1335,12 @@ xinput,XInput Controller,a:b0,b:b1,back:b6,dpdown:h0.4,dpleft:h0.8,dpright:h0.2,
|
||||
05000000010000000100000003000000,Nintendo Wii Remote,a:b0,b:b1,back:b8,dpdown:h0.4,dpleft:h0.8,dpright:h0.2,dpup:h0.1,guide:b10,leftshoulder:b4,leftstick:b11,lefttrigger:b6,leftx:a0,lefty:a1,rightshoulder:b5,rightstick:b12,righttrigger:b7,rightx:a2,righty:a3,start:b9,x:b2,y:b3,platform:Linux,
|
||||
050000007e0500003003000001000000,Nintendo Wii U Pro Controller,a:b0,b:b1,back:b8,dpdown:b14,dpleft:b15,dpright:b16,dpup:b13,guide:b10,leftshoulder:b4,leftstick:b11,lefttrigger:b6,leftx:a0,lefty:a1,rightshoulder:b5,rightstick:b12,righttrigger:b7,rightx:a2,righty:a3,start:b9,x:b3,y:b2,platform:Linux,
|
||||
030000000d0500000308000010010000,Nostromo n45 Dual Analog,a:b0,b:b1,back:b8,dpdown:h0.4,dpleft:h0.8,dpright:h0.2,dpup:h0.1,guide:b9,leftshoulder:b4,leftstick:b12,lefttrigger:b5,leftx:a0,lefty:a1,rightshoulder:b6,rightstick:b11,righttrigger:b7,rightx:a3,righty:a2,start:b10,x:b2,y:b3,platform:Linux,
|
||||
030000007e0500001920000011810000,NSO N64 Controller,+rightx:b10,+righty:b8,-rightx:b9,-righty:b7,a:b0,b:b1,dpdown:h0.4,dpleft:h0.8,dpright:h0.2,dpup:h0.1,guide:b11,leftshoulder:b3,lefttrigger:b2,leftx:a0,lefty:a1,misc1:b12,rightshoulder:b4,righttrigger:b5,start:b6,platform:Linux,
|
||||
050000007e0500001920000001000000,NSO N64 Controller,+rightx:b8,+righty:b7,-rightx:b3,-righty:b2,a:b1,b:b0,dpdown:h0.4,dpleft:h0.8,dpright:h0.2,dpup:h0.1,guide:b12,leftshoulder:b4,lefttrigger:b6,leftx:a0,lefty:a1,misc1:b13,rightshoulder:b5,righttrigger:b10,start:b9,platform:Linux,
|
||||
050000007e0500001920000001800000,NSO N64 Controller,+rightx:b10,+righty:b8,-rightx:b9,-righty:b7,a:b0,b:b1,dpdown:h0.4,dpleft:h0.8,dpright:h0.2,dpup:h0.1,guide:b11,leftshoulder:b3,lefttrigger:b2,leftx:a0,lefty:a1,misc1:b12,rightshoulder:b4,righttrigger:b5,start:b6,platform:Linux,
|
||||
050000007e0500001720000001000000,NSO SNES Controller,a:b0,b:b1,back:b9,dpdown:h0.4,dpleft:h0.8,dpright:h0.2,dpup:h0.1,guide:b11,leftshoulder:b5,leftstick:b12,lefttrigger:b7,leftx:a0,lefty:a1,rightshoulder:b6,rightstick:b13,righttrigger:b8,rightx:a2,righty:a3,start:b10,x:b3,y:b2,platform:Linux,
|
||||
030000007e0500001720000011810000,NSO SNES Controller,a:b1,b:b0,back:b8,dpdown:h0.4,dpleft:h0.8,dpright:h0.2,dpup:h0.1,leftshoulder:b4,lefttrigger:b6,rightshoulder:b5,righttrigger:b7,start:b9,x:b3,y:b2,platform:Linux,
|
||||
050000007e0500001720000001000000,NSO SNES Controller,a:b0,b:b1,back:b9,dpdown:h0.4,dpleft:h0.8,dpright:h0.2,dpup:h0.1,leftshoulder:b5,lefttrigger:b7,rightshoulder:b6,righttrigger:b8,start:b10,x:b3,y:b2,platform:Linux,
|
||||
050000007e0500001720000001800000,NSO SNES Controller,a:b1,b:b0,back:b8,dpdown:h0.4,dpleft:h0.8,dpright:h0.2,dpup:h0.1,leftshoulder:b4,lefttrigger:b6,rightshoulder:b5,righttrigger:b7,start:b9,x:b3,y:b2,platform:Linux,
|
||||
03000000550900001072000011010000,NVIDIA Controller,a:b0,b:b1,dpdown:h0.4,dpleft:h0.8,dpright:h0.2,dpup:h0.1,guide:b13,leftshoulder:b4,leftstick:b8,lefttrigger:a5,leftx:a0,lefty:a1,rightshoulder:b5,rightstick:b9,righttrigger:a4,rightx:a2,righty:a3,start:b7,x:b2,y:b3,platform:Linux,
|
||||
03000000550900001472000011010000,NVIDIA Controller v01.04,a:b0,b:b1,back:b14,dpdown:h0.4,dpleft:h0.8,dpright:h0.2,dpup:h0.1,guide:b16,leftshoulder:b4,leftstick:b7,lefttrigger:a3,leftx:a0,lefty:a1,rightshoulder:b5,rightstick:b8,righttrigger:a4,rightx:a2,righty:a5,start:b6,x:b2,y:b3,platform:Linux,
|
||||
05000000550900001472000001000000,NVIDIA Controller v01.04,a:b0,b:b1,back:b14,dpdown:h0.4,dpleft:h0.8,dpright:h0.2,dpup:h0.1,guide:b16,leftshoulder:b4,leftstick:b7,lefttrigger:a3,leftx:a0,lefty:a1,rightshoulder:b5,rightstick:b8,righttrigger:a4,rightx:a2,righty:a5,start:b6,x:b2,y:b3,platform:Linux,
|
||||
|
||||
@@ -30,7 +30,7 @@ layout(binding=0, rgba8) uniform writeonly image2D imgDst;
|
||||
|
||||
AF3 CasLoad(ASU2 p)
|
||||
{
|
||||
return texelFetch(imgSrc, srcOffset + ivec2(p), 0).rgb;
|
||||
return texelFetch(imgSrc, srcOffset + ivec2(p), 0).rgb;
|
||||
}
|
||||
|
||||
// Lets you transform input from the load into a linear color space between 0 and 1. See ffx_cas.h
|
||||
@@ -42,23 +42,23 @@ void CasInput(inout AF1 r, inout AF1 g, inout AF1 b) {}
|
||||
layout(local_size_x=64) in;
|
||||
void main()
|
||||
{
|
||||
// Do remapping of local xy in workgroup for a more PS-like swizzle pattern.
|
||||
AU2 gxy = ARmp8x8(gl_LocalInvocationID.x)+AU2(gl_WorkGroupID.x<<4u,gl_WorkGroupID.y<<4u);
|
||||
// Do remapping of local xy in workgroup for a more PS-like swizzle pattern.
|
||||
AU2 gxy = ARmp8x8(gl_LocalInvocationID.x)+AU2(gl_WorkGroupID.x<<4u,gl_WorkGroupID.y<<4u);
|
||||
|
||||
// Filter.
|
||||
AF4 c;
|
||||
CasFilter(c.r, c.g, c.b, gxy, const0, const1, CAS_SHARPEN_ONLY);
|
||||
imageStore(imgDst, ASU2(gxy), c);
|
||||
gxy.x += 8u;
|
||||
// Filter.
|
||||
AF4 c;
|
||||
CasFilter(c.r, c.g, c.b, gxy, const0, const1, CAS_SHARPEN_ONLY);
|
||||
imageStore(imgDst, ASU2(gxy), c);
|
||||
gxy.x += 8u;
|
||||
|
||||
CasFilter(c.r, c.g, c.b, gxy, const0, const1, CAS_SHARPEN_ONLY);
|
||||
imageStore(imgDst, ASU2(gxy), c);
|
||||
gxy.y += 8u;
|
||||
CasFilter(c.r, c.g, c.b, gxy, const0, const1, CAS_SHARPEN_ONLY);
|
||||
imageStore(imgDst, ASU2(gxy), c);
|
||||
gxy.y += 8u;
|
||||
|
||||
CasFilter(c.r, c.g, c.b, gxy, const0, const1, CAS_SHARPEN_ONLY);
|
||||
imageStore(imgDst, ASU2(gxy), c);
|
||||
gxy.x -= 8u;
|
||||
CasFilter(c.r, c.g, c.b, gxy, const0, const1, CAS_SHARPEN_ONLY);
|
||||
imageStore(imgDst, ASU2(gxy), c);
|
||||
gxy.x -= 8u;
|
||||
|
||||
CasFilter(c.r, c.g, c.b, gxy, const0, const1, CAS_SHARPEN_ONLY);
|
||||
imageStore(imgDst, ASU2(gxy), c);
|
||||
CasFilter(c.r, c.g, c.b, gxy, const0, const1, CAS_SHARPEN_ONLY);
|
||||
imageStore(imgDst, ASU2(gxy), c);
|
||||
}
|
||||
|
||||
@@ -21,10 +21,10 @@ out vec4 PSin_c;
|
||||
|
||||
void vs_main()
|
||||
{
|
||||
PSin_p = vec4(POSITION, 0.5f, 1.0f);
|
||||
PSin_t = TEXCOORD0;
|
||||
PSin_c = COLOR;
|
||||
gl_Position = vec4(POSITION, 0.5f, 1.0f); // NOTE I don't know if it is possible to merge POSITION_OUT and gl_Position
|
||||
PSin_p = vec4(POSITION, 0.5f, 1.0f);
|
||||
PSin_t = TEXCOORD0;
|
||||
PSin_c = COLOR;
|
||||
gl_Position = vec4(POSITION, 0.5f, 1.0f); // NOTE I don't know if it is possible to merge POSITION_OUT and gl_Position
|
||||
}
|
||||
|
||||
#endif
|
||||
@@ -46,13 +46,13 @@ layout(location = 0) out vec4 SV_Target0;
|
||||
|
||||
vec4 sample_c()
|
||||
{
|
||||
return texture(TextureSampler, PSin_t);
|
||||
return texture(TextureSampler, PSin_t);
|
||||
}
|
||||
|
||||
#ifdef ps_copy
|
||||
void ps_copy()
|
||||
{
|
||||
SV_Target0 = sample_c();
|
||||
SV_Target0 = sample_c();
|
||||
}
|
||||
#endif
|
||||
|
||||
@@ -67,20 +67,20 @@ void ps_depth_copy()
|
||||
// Need to be careful with precision here, it can break games like Spider-Man 3 and Dogs Life
|
||||
void ps_convert_rgba8_16bits()
|
||||
{
|
||||
highp uvec4 i = uvec4(sample_c() * vec4(255.5f, 255.5f, 255.5f, 255.5f));
|
||||
highp uvec4 i = uvec4(sample_c() * vec4(255.5f, 255.5f, 255.5f, 255.5f));
|
||||
|
||||
SV_Target1 = ((i.x & 0x00F8u) >> 3) | ((i.y & 0x00F8u) << 2) | ((i.z & 0x00f8u) << 7) | ((i.w & 0x80u) << 8);
|
||||
SV_Target1 = ((i.x & 0x00F8u) >> 3) | ((i.y & 0x00F8u) << 2) | ((i.z & 0x00f8u) << 7) | ((i.w & 0x80u) << 8);
|
||||
}
|
||||
#endif
|
||||
|
||||
#ifdef ps_convert_float32_32bits
|
||||
void ps_convert_float32_32bits()
|
||||
{
|
||||
// Convert a GL_FLOAT32 depth texture into a 32 bits UINT texture
|
||||
// Convert a GL_FLOAT32 depth texture into a 32 bits UINT texture
|
||||
#if HAS_CLIP_CONTROL
|
||||
SV_Target1 = uint(exp2(32.0f) * sample_c().r);
|
||||
SV_Target1 = uint(exp2(32.0f) * sample_c().r);
|
||||
#else
|
||||
SV_Target1 = uint(exp2(24.0f) * sample_c().r);
|
||||
SV_Target1 = uint(exp2(24.0f) * sample_c().r);
|
||||
#endif
|
||||
}
|
||||
#endif
|
||||
@@ -88,150 +88,150 @@ void ps_convert_float32_32bits()
|
||||
#ifdef ps_convert_float32_rgba8
|
||||
void ps_convert_float32_rgba8()
|
||||
{
|
||||
// Convert a GL_FLOAT32 depth texture into a RGBA color texture
|
||||
// Convert a GL_FLOAT32 depth texture into a RGBA color texture
|
||||
#if HAS_CLIP_CONTROL
|
||||
uint d = uint(sample_c().r * exp2(32.0f));
|
||||
uint d = uint(sample_c().r * exp2(32.0f));
|
||||
#else
|
||||
uint d = uint(sample_c().r * exp2(24.0f));
|
||||
uint d = uint(sample_c().r * exp2(24.0f));
|
||||
#endif
|
||||
SV_Target0 = vec4(uvec4((d & 0xFFu), ((d >> 8) & 0xFFu), ((d >> 16) & 0xFFu), (d >> 24))) / vec4(255.0);
|
||||
SV_Target0 = vec4(uvec4((d & 0xFFu), ((d >> 8) & 0xFFu), ((d >> 16) & 0xFFu), (d >> 24))) / vec4(255.0);
|
||||
}
|
||||
#endif
|
||||
|
||||
#ifdef ps_convert_float16_rgb5a1
|
||||
void ps_convert_float16_rgb5a1()
|
||||
{
|
||||
// Convert a GL_FLOAT32 (only 16 lsb) depth into a RGB5A1 color texture
|
||||
// Convert a GL_FLOAT32 (only 16 lsb) depth into a RGB5A1 color texture
|
||||
#if HAS_CLIP_CONTROL
|
||||
uint d = uint(sample_c().r * exp2(32.0f));
|
||||
uint d = uint(sample_c().r * exp2(32.0f));
|
||||
#else
|
||||
uint d = uint(sample_c().r * exp2(24.0f));
|
||||
uint d = uint(sample_c().r * exp2(24.0f));
|
||||
#endif
|
||||
SV_Target0 = vec4(uvec4((d & 0x1Fu), ((d >> 5) & 0x1Fu), ((d >> 10) & 0x1Fu), (d >> 15) & 0x01u)) / vec4(32.0f, 32.0f, 32.0f, 1.0f);
|
||||
SV_Target0 = vec4(uvec4((d & 0x1Fu), ((d >> 5) & 0x1Fu), ((d >> 10) & 0x1Fu), (d >> 15) & 0x01u)) / vec4(32.0f, 32.0f, 32.0f, 1.0f);
|
||||
}
|
||||
#endif
|
||||
|
||||
float rgba8_to_depth32(vec4 unorm)
|
||||
{
|
||||
uvec4 c = uvec4(unorm * vec4(255.5f));
|
||||
uvec4 c = uvec4(unorm * vec4(255.5f));
|
||||
#if HAS_CLIP_CONTROL
|
||||
return float(c.r | (c.g << 8) | (c.b << 16) | (c.a << 24)) * exp2(-32.0f);
|
||||
return float(c.r | (c.g << 8) | (c.b << 16) | (c.a << 24)) * exp2(-32.0f);
|
||||
#else
|
||||
return float(c.r | (c.g << 8) | (c.b << 16) | (c.a << 24)) * exp2(-24.0f);
|
||||
return float(c.r | (c.g << 8) | (c.b << 16) | (c.a << 24)) * exp2(-24.0f);
|
||||
#endif
|
||||
}
|
||||
|
||||
float rgba8_to_depth24(vec4 unorm)
|
||||
{
|
||||
uvec3 c = uvec3(unorm.rgb * vec3(255.5f));
|
||||
uvec3 c = uvec3(unorm.rgb * vec3(255.5f));
|
||||
#if HAS_CLIP_CONTROL
|
||||
return float(c.r | (c.g << 8) | (c.b << 16)) * exp2(-32.0f);
|
||||
return float(c.r | (c.g << 8) | (c.b << 16)) * exp2(-32.0f);
|
||||
#else
|
||||
return float(c.r | (c.g << 8) | (c.b << 16)) * exp2(-24.0f);
|
||||
return float(c.r | (c.g << 8) | (c.b << 16)) * exp2(-24.0f);
|
||||
#endif
|
||||
}
|
||||
|
||||
float rgba8_to_depth16(vec4 unorm)
|
||||
{
|
||||
uvec2 c = uvec2(unorm.rg * vec2(255.5f));
|
||||
uvec2 c = uvec2(unorm.rg * vec2(255.5f));
|
||||
#if HAS_CLIP_CONTROL
|
||||
return float(c.r | (c.g << 8)) * exp2(-32.0f);
|
||||
return float(c.r | (c.g << 8)) * exp2(-32.0f);
|
||||
#else
|
||||
return float(c.r | (c.g << 8)) * exp2(-24.0f);
|
||||
return float(c.r | (c.g << 8)) * exp2(-24.0f);
|
||||
#endif
|
||||
}
|
||||
|
||||
float rgb5a1_to_depth16(vec4 unorm)
|
||||
{
|
||||
uvec4 c = uvec4(unorm * vec4(255.5f));
|
||||
uvec4 c = uvec4(unorm * vec4(255.5f));
|
||||
#if HAS_CLIP_CONTROL
|
||||
return float(((c.r & 0xF8u) >> 3) | ((c.g & 0xF8u) << 2) | ((c.b & 0xF8u) << 7) | ((c.a & 0x80u) << 8)) * exp2(-32.0f);
|
||||
return float(((c.r & 0xF8u) >> 3) | ((c.g & 0xF8u) << 2) | ((c.b & 0xF8u) << 7) | ((c.a & 0x80u) << 8)) * exp2(-32.0f);
|
||||
#else
|
||||
return float(((c.r & 0xF8u) >> 3) | ((c.g & 0xF8u) << 2) | ((c.b & 0xF8u) << 7) | ((c.a & 0x80u) << 8)) * exp2(-24.0f);
|
||||
return float(((c.r & 0xF8u) >> 3) | ((c.g & 0xF8u) << 2) | ((c.b & 0xF8u) << 7) | ((c.a & 0x80u) << 8)) * exp2(-24.0f);
|
||||
#endif
|
||||
}
|
||||
|
||||
#ifdef ps_convert_rgba8_float32
|
||||
void ps_convert_rgba8_float32()
|
||||
{
|
||||
// Convert an RGBA texture into a float depth texture
|
||||
gl_FragDepth = rgba8_to_depth32(sample_c());
|
||||
// Convert an RGBA texture into a float depth texture
|
||||
gl_FragDepth = rgba8_to_depth32(sample_c());
|
||||
}
|
||||
#endif
|
||||
|
||||
#ifdef ps_convert_rgba8_float24
|
||||
void ps_convert_rgba8_float24()
|
||||
{
|
||||
// Same as above but without the alpha channel (24 bits Z)
|
||||
// Same as above but without the alpha channel (24 bits Z)
|
||||
|
||||
// Convert an RGBA texture into a float depth texture
|
||||
gl_FragDepth = rgba8_to_depth24(sample_c());
|
||||
// Convert an RGBA texture into a float depth texture
|
||||
gl_FragDepth = rgba8_to_depth24(sample_c());
|
||||
}
|
||||
#endif
|
||||
|
||||
#ifdef ps_convert_rgba8_float16
|
||||
void ps_convert_rgba8_float16()
|
||||
{
|
||||
// Same as above but without the A/B channels (16 bits Z)
|
||||
// Same as above but without the A/B channels (16 bits Z)
|
||||
|
||||
// Convert an RGBA texture into a float depth texture
|
||||
gl_FragDepth = rgba8_to_depth16(sample_c());
|
||||
// Convert an RGBA texture into a float depth texture
|
||||
gl_FragDepth = rgba8_to_depth16(sample_c());
|
||||
}
|
||||
#endif
|
||||
|
||||
#ifdef ps_convert_rgb5a1_float16
|
||||
void ps_convert_rgb5a1_float16()
|
||||
{
|
||||
// Convert an RGB5A1 (saved as RGBA8) color to a 16 bit Z
|
||||
gl_FragDepth = rgb5a1_to_depth16(sample_c());
|
||||
// Convert an RGB5A1 (saved as RGBA8) color to a 16 bit Z
|
||||
gl_FragDepth = rgb5a1_to_depth16(sample_c());
|
||||
}
|
||||
#endif
|
||||
|
||||
#define SAMPLE_RGBA_DEPTH_BILN(CONVERT_FN) \
|
||||
ivec2 dims = textureSize(TextureSampler, 0); \
|
||||
vec2 top_left_f = PSin_t * vec2(dims) - 0.5f; \
|
||||
ivec2 top_left = ivec2(floor(top_left_f)); \
|
||||
ivec4 coords = clamp(ivec4(top_left, top_left + 1), ivec4(0), dims.xyxy - 1); \
|
||||
vec2 mix_vals = fract(top_left_f); \
|
||||
float depthTL = CONVERT_FN(texelFetch(TextureSampler, coords.xy, 0)); \
|
||||
float depthTR = CONVERT_FN(texelFetch(TextureSampler, coords.zy, 0)); \
|
||||
float depthBL = CONVERT_FN(texelFetch(TextureSampler, coords.xw, 0)); \
|
||||
float depthBR = CONVERT_FN(texelFetch(TextureSampler, coords.zw, 0)); \
|
||||
gl_FragDepth = mix(mix(depthTL, depthTR, mix_vals.x), mix(depthBL, depthBR, mix_vals.x), mix_vals.y);
|
||||
ivec2 dims = textureSize(TextureSampler, 0); \
|
||||
vec2 top_left_f = PSin_t * vec2(dims) - 0.5f; \
|
||||
ivec2 top_left = ivec2(floor(top_left_f)); \
|
||||
ivec4 coords = clamp(ivec4(top_left, top_left + 1), ivec4(0), dims.xyxy - 1); \
|
||||
vec2 mix_vals = fract(top_left_f); \
|
||||
float depthTL = CONVERT_FN(texelFetch(TextureSampler, coords.xy, 0)); \
|
||||
float depthTR = CONVERT_FN(texelFetch(TextureSampler, coords.zy, 0)); \
|
||||
float depthBL = CONVERT_FN(texelFetch(TextureSampler, coords.xw, 0)); \
|
||||
float depthBR = CONVERT_FN(texelFetch(TextureSampler, coords.zw, 0)); \
|
||||
gl_FragDepth = mix(mix(depthTL, depthTR, mix_vals.x), mix(depthBL, depthBR, mix_vals.x), mix_vals.y);
|
||||
|
||||
#ifdef ps_convert_rgba8_float32_biln
|
||||
void ps_convert_rgba8_float32_biln()
|
||||
{
|
||||
// Convert an RGBA texture into a float depth texture
|
||||
SAMPLE_RGBA_DEPTH_BILN(rgba8_to_depth32);
|
||||
// Convert an RGBA texture into a float depth texture
|
||||
SAMPLE_RGBA_DEPTH_BILN(rgba8_to_depth32);
|
||||
}
|
||||
#endif
|
||||
|
||||
#ifdef ps_convert_rgba8_float24_biln
|
||||
void ps_convert_rgba8_float24_biln()
|
||||
{
|
||||
// Same as above but without the alpha channel (24 bits Z)
|
||||
// Same as above but without the alpha channel (24 bits Z)
|
||||
|
||||
// Convert an RGBA texture into a float depth texture
|
||||
SAMPLE_RGBA_DEPTH_BILN(rgba8_to_depth24);
|
||||
// Convert an RGBA texture into a float depth texture
|
||||
SAMPLE_RGBA_DEPTH_BILN(rgba8_to_depth24);
|
||||
}
|
||||
#endif
|
||||
|
||||
#ifdef ps_convert_rgba8_float16_biln
|
||||
void ps_convert_rgba8_float16_biln()
|
||||
{
|
||||
// Same as above but without the A/B channels (16 bits Z)
|
||||
// Same as above but without the A/B channels (16 bits Z)
|
||||
|
||||
// Convert an RGBA texture into a float depth texture
|
||||
SAMPLE_RGBA_DEPTH_BILN(rgba8_to_depth16);
|
||||
// Convert an RGBA texture into a float depth texture
|
||||
SAMPLE_RGBA_DEPTH_BILN(rgba8_to_depth16);
|
||||
}
|
||||
#endif
|
||||
|
||||
#ifdef ps_convert_rgb5a1_float16_biln
|
||||
void ps_convert_rgb5a1_float16_biln()
|
||||
{
|
||||
// Convert an RGB5A1 (saved as RGBA8) color to a 16 bit Z
|
||||
SAMPLE_RGBA_DEPTH_BILN(rgb5a1_to_depth16);
|
||||
// Convert an RGB5A1 (saved as RGBA8) color to a 16 bit Z
|
||||
SAMPLE_RGBA_DEPTH_BILN(rgb5a1_to_depth16);
|
||||
}
|
||||
#endif
|
||||
|
||||
@@ -242,51 +242,51 @@ uniform float ScaleFactor;
|
||||
|
||||
void ps_convert_rgba_8i()
|
||||
{
|
||||
// Convert a RGBA texture into a 8 bits packed texture
|
||||
// Input column: 8x2 RGBA pixels
|
||||
// 0: 8 RGBA
|
||||
// 1: 8 RGBA
|
||||
// Output column: 16x4 Index pixels
|
||||
// 0: 8 R | 8 B
|
||||
// 1: 8 R | 8 B
|
||||
// 2: 8 G | 8 A
|
||||
// 3: 8 G | 8 A
|
||||
uvec2 pos = uvec2(gl_FragCoord.xy);
|
||||
// Convert a RGBA texture into a 8 bits packed texture
|
||||
// Input column: 8x2 RGBA pixels
|
||||
// 0: 8 RGBA
|
||||
// 1: 8 RGBA
|
||||
// Output column: 16x4 Index pixels
|
||||
// 0: 8 R | 8 B
|
||||
// 1: 8 R | 8 B
|
||||
// 2: 8 G | 8 A
|
||||
// 3: 8 G | 8 A
|
||||
uvec2 pos = uvec2(gl_FragCoord.xy);
|
||||
|
||||
// Collapse separate R G B A areas into their base pixel
|
||||
uvec2 block = (pos & ~uvec2(15u, 3u)) >> 1;
|
||||
uvec2 subblock = pos & uvec2(7u, 1u);
|
||||
uvec2 coord = block | subblock;
|
||||
// Collapse separate R G B A areas into their base pixel
|
||||
uvec2 block = (pos & ~uvec2(15u, 3u)) >> 1;
|
||||
uvec2 subblock = pos & uvec2(7u, 1u);
|
||||
uvec2 coord = block | subblock;
|
||||
|
||||
// Compensate for potentially differing page pitch.
|
||||
// Compensate for potentially differing page pitch.
|
||||
uvec2 block_xy = coord / uvec2(64u, 32u);
|
||||
uint block_num = (block_xy.y * (DBW / 128u)) + block_xy.x;
|
||||
uvec2 block_offset = uvec2((block_num % (SBW / 64u)) * 64u, (block_num / (SBW / 64u)) * 32u);
|
||||
coord = (coord % uvec2(64u, 32u)) + block_offset;
|
||||
|
||||
// Apply offset to cols 1 and 2
|
||||
uint is_col23 = pos.y & 4u;
|
||||
uint is_col13 = pos.y & 2u;
|
||||
uint is_col12 = is_col23 ^ (is_col13 << 1);
|
||||
coord.x ^= is_col12; // If cols 1 or 2, flip bit 3 of x
|
||||
// Apply offset to cols 1 and 2
|
||||
uint is_col23 = pos.y & 4u;
|
||||
uint is_col13 = pos.y & 2u;
|
||||
uint is_col12 = is_col23 ^ (is_col13 << 1);
|
||||
coord.x ^= is_col12; // If cols 1 or 2, flip bit 3 of x
|
||||
|
||||
if (floor(ScaleFactor) != ScaleFactor)
|
||||
coord = uvec2(vec2(coord) * ScaleFactor);
|
||||
else
|
||||
coord *= uvec2(ScaleFactor);
|
||||
if (floor(ScaleFactor) != ScaleFactor)
|
||||
coord = uvec2(vec2(coord) * ScaleFactor);
|
||||
else
|
||||
coord *= uvec2(ScaleFactor);
|
||||
|
||||
vec4 pixel = texelFetch(TextureSampler, ivec2(coord), 0);
|
||||
vec2 sel0 = (pos.y & 2u) == 0u ? pixel.rb : pixel.ga;
|
||||
float sel1 = (pos.x & 8u) == 0u ? sel0.x : sel0.y;
|
||||
SV_Target0 = vec4(sel1);
|
||||
vec4 pixel = texelFetch(TextureSampler, ivec2(coord), 0);
|
||||
vec2 sel0 = (pos.y & 2u) == 0u ? pixel.rb : pixel.ga;
|
||||
float sel1 = (pos.x & 8u) == 0u ? sel0.x : sel0.y;
|
||||
SV_Target0 = vec4(sel1);
|
||||
}
|
||||
#endif
|
||||
|
||||
#ifdef ps_filter_transparency
|
||||
void ps_filter_transparency()
|
||||
{
|
||||
vec4 c = sample_c();
|
||||
SV_Target0 = vec4(c.rgb, 1.0);
|
||||
vec4 c = sample_c();
|
||||
SV_Target0 = vec4(c.rgb, 1.0);
|
||||
}
|
||||
#endif
|
||||
|
||||
@@ -295,8 +295,8 @@ void ps_filter_transparency()
|
||||
#ifdef ps_datm1
|
||||
void ps_datm1()
|
||||
{
|
||||
if(sample_c().a < (127.5f / 255.0f)) // >= 0x80 pass
|
||||
discard;
|
||||
if(sample_c().a < (127.5f / 255.0f)) // >= 0x80 pass
|
||||
discard;
|
||||
}
|
||||
#endif
|
||||
|
||||
@@ -305,24 +305,24 @@ void ps_datm1()
|
||||
#ifdef ps_datm0
|
||||
void ps_datm0()
|
||||
{
|
||||
if((127.5f / 255.0f) < sample_c().a) // < 0x80 pass (== 0x80 should not pass)
|
||||
discard;
|
||||
if((127.5f / 255.0f) < sample_c().a) // < 0x80 pass (== 0x80 should not pass)
|
||||
discard;
|
||||
}
|
||||
#endif
|
||||
|
||||
#ifdef ps_hdr_init
|
||||
void ps_hdr_init()
|
||||
{
|
||||
vec4 value = sample_c();
|
||||
SV_Target0 = vec4(round(value.rgb * 255.0f) / 65535.0f, value.a);
|
||||
vec4 value = sample_c();
|
||||
SV_Target0 = vec4(round(value.rgb * 255.0f) / 65535.0f, value.a);
|
||||
}
|
||||
#endif
|
||||
|
||||
#ifdef ps_hdr_resolve
|
||||
void ps_hdr_resolve()
|
||||
{
|
||||
vec4 value = sample_c();
|
||||
SV_Target0 = vec4(vec3(uvec3(value.rgb * 65535.0f) & 255u) / 255.0f, value.a);
|
||||
vec4 value = sample_c();
|
||||
SV_Target0 = vec4(vec3(uvec3(value.rgb * 65535.0f) & 255u) / 255.0f, value.a);
|
||||
}
|
||||
#endif
|
||||
|
||||
@@ -366,51 +366,51 @@ uniform ivec2 EMOD;
|
||||
|
||||
void ps_yuv()
|
||||
{
|
||||
vec4 i = sample_c();
|
||||
vec4 o;
|
||||
vec4 i = sample_c();
|
||||
vec4 o;
|
||||
|
||||
mat3 rgb2yuv; // Value from GS manual
|
||||
rgb2yuv[0] = vec3(0.587, -0.311, -0.419);
|
||||
rgb2yuv[1] = vec3(0.114, 0.500, -0.081);
|
||||
rgb2yuv[2] = vec3(0.299, -0.169, 0.500);
|
||||
mat3 rgb2yuv; // Value from GS manual
|
||||
rgb2yuv[0] = vec3(0.587, -0.311, -0.419);
|
||||
rgb2yuv[1] = vec3(0.114, 0.500, -0.081);
|
||||
rgb2yuv[2] = vec3(0.299, -0.169, 0.500);
|
||||
|
||||
vec3 yuv = rgb2yuv * i.gbr;
|
||||
vec3 yuv = rgb2yuv * i.gbr;
|
||||
|
||||
float Y = float(0xDB)/255.0f * yuv.x + float(0x10)/255.0f;
|
||||
float Cr = float(0xE0)/255.0f * yuv.y + float(0x80)/255.0f;
|
||||
float Cb = float(0xE0)/255.0f * yuv.z + float(0x80)/255.0f;
|
||||
float Y = float(0xDB)/255.0f * yuv.x + float(0x10)/255.0f;
|
||||
float Cr = float(0xE0)/255.0f * yuv.y + float(0x80)/255.0f;
|
||||
float Cb = float(0xE0)/255.0f * yuv.z + float(0x80)/255.0f;
|
||||
|
||||
switch(EMOD.x) {
|
||||
case 0:
|
||||
o.a = i.a;
|
||||
break;
|
||||
case 1:
|
||||
o.a = Y;
|
||||
break;
|
||||
case 2:
|
||||
o.a = Y/2.0f;
|
||||
break;
|
||||
case 3:
|
||||
o.a = 0.0f;
|
||||
break;
|
||||
}
|
||||
switch(EMOD.x) {
|
||||
case 0:
|
||||
o.a = i.a;
|
||||
break;
|
||||
case 1:
|
||||
o.a = Y;
|
||||
break;
|
||||
case 2:
|
||||
o.a = Y/2.0f;
|
||||
break;
|
||||
case 3:
|
||||
o.a = 0.0f;
|
||||
break;
|
||||
}
|
||||
|
||||
switch(EMOD.y) {
|
||||
case 0:
|
||||
o.rgb = i.rgb;
|
||||
break;
|
||||
case 1:
|
||||
o.rgb = vec3(Y);
|
||||
break;
|
||||
case 2:
|
||||
o.rgb = vec3(Y, Cb, Cr);
|
||||
break;
|
||||
case 3:
|
||||
o.rgb = vec3(i.a);
|
||||
break;
|
||||
}
|
||||
switch(EMOD.y) {
|
||||
case 0:
|
||||
o.rgb = i.rgb;
|
||||
break;
|
||||
case 1:
|
||||
o.rgb = vec3(Y);
|
||||
break;
|
||||
case 2:
|
||||
o.rgb = vec3(Y, Cb, Cr);
|
||||
break;
|
||||
case 3:
|
||||
o.rgb = vec3(i.a);
|
||||
break;
|
||||
}
|
||||
|
||||
SV_Target0 = o;
|
||||
SV_Target0 = o;
|
||||
}
|
||||
#endif
|
||||
|
||||
@@ -418,16 +418,16 @@ void ps_yuv()
|
||||
|
||||
void main()
|
||||
{
|
||||
SV_Target0 = vec4(0x7FFFFFFF);
|
||||
SV_Target0 = vec4(0x7FFFFFFF);
|
||||
|
||||
#ifdef ps_stencil_image_init_0
|
||||
if((127.5f / 255.0f) < sample_c().a) // < 0x80 pass (== 0x80 should not pass)
|
||||
SV_Target0 = vec4(-1);
|
||||
#endif
|
||||
#ifdef ps_stencil_image_init_1
|
||||
if(sample_c().a < (127.5f / 255.0f)) // >= 0x80 pass
|
||||
SV_Target0 = vec4(-1);
|
||||
#endif
|
||||
#ifdef ps_stencil_image_init_0
|
||||
if((127.5f / 255.0f) < sample_c().a) // < 0x80 pass (== 0x80 should not pass)
|
||||
SV_Target0 = vec4(-1);
|
||||
#endif
|
||||
#ifdef ps_stencil_image_init_1
|
||||
if(sample_c().a < (127.5f / 255.0f)) // >= 0x80 pass
|
||||
SV_Target0 = vec4(-1);
|
||||
#endif
|
||||
}
|
||||
#endif
|
||||
|
||||
|
||||
@@ -14,17 +14,17 @@ layout(location = 0) out vec4 SV_Target0;
|
||||
|
||||
void ps_main0()
|
||||
{
|
||||
vec4 c = texture(TextureSampler, PSin_t);
|
||||
// Note: clamping will be done by fixed unit
|
||||
c.a *= 2.0f;
|
||||
SV_Target0 = c;
|
||||
vec4 c = texture(TextureSampler, PSin_t);
|
||||
// Note: clamping will be done by fixed unit
|
||||
c.a *= 2.0f;
|
||||
SV_Target0 = c;
|
||||
}
|
||||
|
||||
void ps_main1()
|
||||
{
|
||||
vec4 c = texture(TextureSampler, PSin_t);
|
||||
c.a = BGColor.a;
|
||||
SV_Target0 = c;
|
||||
vec4 c = texture(TextureSampler, PSin_t);
|
||||
c.a = BGColor.a;
|
||||
SV_Target0 = c;
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
@@ -119,10 +119,10 @@ void ps_filter_triangular() // triangular
|
||||
#ifdef ps_filter_complex
|
||||
void ps_filter_complex()
|
||||
{
|
||||
const float PI = 3.14159265359f;
|
||||
vec2 texdim = vec2(textureSize(TextureSampler, 0));
|
||||
float factor = (0.9f - 0.4f * cos(2.0f * PI * PSin_t.y * texdim.y));
|
||||
vec4 c = factor * texture(TextureSampler, vec2(PSin_t.x, (floor(PSin_t.y * texdim.y) + 0.5f) / texdim.y));
|
||||
const float PI = 3.14159265359f;
|
||||
vec2 texdim = vec2(textureSize(TextureSampler, 0));
|
||||
float factor = (0.9f - 0.4f * cos(2.0f * PI * PSin_t.y * texdim.y));
|
||||
vec4 c = factor * texture(TextureSampler, vec2(PSin_t.x, (floor(PSin_t.y * texdim.y) + 0.5f) / texdim.y));
|
||||
|
||||
SV_Target0 = c;
|
||||
}
|
||||
|
||||
@@ -24,33 +24,33 @@ layout(location = 0) out vec4 SV_Target0;
|
||||
// For all settings: 1.0 = 100% 0.5=50% 1.5 = 150%
|
||||
vec4 ContrastSaturationBrightness(vec4 color)
|
||||
{
|
||||
float brt = params.x;
|
||||
float con = params.y;
|
||||
float sat = params.z;
|
||||
float brt = params.x;
|
||||
float con = params.y;
|
||||
float sat = params.z;
|
||||
|
||||
// Increase or decrease these values to adjust r, g and b color channels separately
|
||||
const float AvgLumR = 0.5;
|
||||
const float AvgLumG = 0.5;
|
||||
const float AvgLumB = 0.5;
|
||||
// Increase or decrease these values to adjust r, g and b color channels separately
|
||||
const float AvgLumR = 0.5;
|
||||
const float AvgLumG = 0.5;
|
||||
const float AvgLumB = 0.5;
|
||||
|
||||
const vec3 LumCoeff = vec3(0.2125, 0.7154, 0.0721);
|
||||
const vec3 LumCoeff = vec3(0.2125, 0.7154, 0.0721);
|
||||
|
||||
vec3 AvgLumin = vec3(AvgLumR, AvgLumG, AvgLumB);
|
||||
vec3 brtColor = color.rgb * brt;
|
||||
float dot_intensity = dot(brtColor, LumCoeff);
|
||||
vec3 intensity = vec3(dot_intensity, dot_intensity, dot_intensity);
|
||||
vec3 satColor = mix(intensity, brtColor, sat);
|
||||
vec3 conColor = mix(AvgLumin, satColor, con);
|
||||
vec3 AvgLumin = vec3(AvgLumR, AvgLumG, AvgLumB);
|
||||
vec3 brtColor = color.rgb * brt;
|
||||
float dot_intensity = dot(brtColor, LumCoeff);
|
||||
vec3 intensity = vec3(dot_intensity, dot_intensity, dot_intensity);
|
||||
vec3 satColor = mix(intensity, brtColor, sat);
|
||||
vec3 conColor = mix(AvgLumin, satColor, con);
|
||||
|
||||
color.rgb = conColor;
|
||||
return color;
|
||||
color.rgb = conColor;
|
||||
return color;
|
||||
}
|
||||
|
||||
|
||||
void ps_main()
|
||||
{
|
||||
vec4 c = texture(TextureSampler, PSin_t);
|
||||
SV_Target0 = ContrastSaturationBrightness(c);
|
||||
vec4 c = texture(TextureSampler, PSin_t);
|
||||
SV_Target0 = ContrastSaturationBrightness(c);
|
||||
}
|
||||
|
||||
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -2,28 +2,28 @@
|
||||
|
||||
layout(std140, binding = 1) uniform cb20
|
||||
{
|
||||
vec2 VertexScale;
|
||||
vec2 VertexOffset;
|
||||
vec2 VertexScale;
|
||||
vec2 VertexOffset;
|
||||
|
||||
vec2 TextureScale;
|
||||
vec2 TextureOffset;
|
||||
vec2 TextureScale;
|
||||
vec2 TextureOffset;
|
||||
|
||||
vec2 PointSize;
|
||||
uint MaxDepth;
|
||||
uint pad_cb20;
|
||||
vec2 PointSize;
|
||||
uint MaxDepth;
|
||||
uint pad_cb20;
|
||||
};
|
||||
|
||||
#ifdef VERTEX_SHADER
|
||||
|
||||
out SHADER
|
||||
{
|
||||
vec4 t_float;
|
||||
vec4 t_int;
|
||||
#if VS_IIP != 0
|
||||
vec4 c;
|
||||
#else
|
||||
flat vec4 c;
|
||||
#endif
|
||||
vec4 t_float;
|
||||
vec4 t_int;
|
||||
#if VS_IIP != 0
|
||||
vec4 c;
|
||||
#else
|
||||
flat vec4 c;
|
||||
#endif
|
||||
} VSout;
|
||||
|
||||
const float exp_min32 = exp2(-32.0f);
|
||||
@@ -40,198 +40,198 @@ layout(location = 7) in vec4 i_f;
|
||||
|
||||
void texture_coord()
|
||||
{
|
||||
vec2 uv = vec2(i_uv) - TextureOffset;
|
||||
vec2 st = i_st - TextureOffset;
|
||||
vec2 uv = vec2(i_uv) - TextureOffset;
|
||||
vec2 st = i_st - TextureOffset;
|
||||
|
||||
// Float coordinate
|
||||
VSout.t_float.xy = st;
|
||||
VSout.t_float.w = i_q;
|
||||
// Float coordinate
|
||||
VSout.t_float.xy = st;
|
||||
VSout.t_float.w = i_q;
|
||||
|
||||
// Integer coordinate => normalized
|
||||
VSout.t_int.xy = uv * TextureScale;
|
||||
// Integer coordinate => normalized
|
||||
VSout.t_int.xy = uv * TextureScale;
|
||||
#if VS_FST
|
||||
// Integer coordinate => integral
|
||||
VSout.t_int.zw = uv;
|
||||
// Integer coordinate => integral
|
||||
VSout.t_int.zw = uv;
|
||||
#else
|
||||
// Some games uses float coordinate for post-processing effect
|
||||
VSout.t_int.zw = st / TextureScale;
|
||||
// Some games uses float coordinate for post-processing effect
|
||||
VSout.t_int.zw = st / TextureScale;
|
||||
#endif
|
||||
}
|
||||
|
||||
void vs_main()
|
||||
{
|
||||
// Clamp to max depth, gs doesn't wrap
|
||||
highp uint z = min(i_z, MaxDepth);
|
||||
// Clamp to max depth, gs doesn't wrap
|
||||
highp uint z = min(i_z, MaxDepth);
|
||||
|
||||
// pos -= 0.05 (1/320 pixel) helps avoiding rounding problems (integral part of pos is usually 5 digits, 0.05 is about as low as we can go)
|
||||
// example: ceil(afterseveralvertextransformations(y = 133)) => 134 => line 133 stays empty
|
||||
// input granularity is 1/16 pixel, anything smaller than that won't step drawing up/left by one pixel
|
||||
// example: 133.0625 (133 + 1/16) should start from line 134, ceil(133.0625 - 0.05) still above 133
|
||||
vec4 p;
|
||||
// pos -= 0.05 (1/320 pixel) helps avoiding rounding problems (integral part of pos is usually 5 digits, 0.05 is about as low as we can go)
|
||||
// example: ceil(afterseveralvertextransformations(y = 133)) => 134 => line 133 stays empty
|
||||
// input granularity is 1/16 pixel, anything smaller than that won't step drawing up/left by one pixel
|
||||
// example: 133.0625 (133 + 1/16) should start from line 134, ceil(133.0625 - 0.05) still above 133
|
||||
vec4 p;
|
||||
|
||||
p.xy = vec2(i_p) - vec2(0.05f, 0.05f);
|
||||
p.xy = p.xy * VertexScale - VertexOffset;
|
||||
p.w = 1.0f;
|
||||
p.xy = vec2(i_p) - vec2(0.05f, 0.05f);
|
||||
p.xy = p.xy * VertexScale - VertexOffset;
|
||||
p.w = 1.0f;
|
||||
|
||||
#if HAS_CLIP_CONTROL
|
||||
p.z = float(z) * exp_min32;
|
||||
p.z = float(z) * exp_min32;
|
||||
#else
|
||||
// GLES doesn't support ARB_clip_control, so remap it to -1..1. We also reduce the range from 32 bits
|
||||
// to 24 bits, which means some games with very large depth ranges will not render correctly. But,
|
||||
// for most, it's okay, and really, the best we can do.
|
||||
p.z = min(float(z) * exp2(-23.0f), 2.0f) - 1.0f;
|
||||
// GLES doesn't support ARB_clip_control, so remap it to -1..1. We also reduce the range from 32 bits
|
||||
// to 24 bits, which means some games with very large depth ranges will not render correctly. But,
|
||||
// for most, it's okay, and really, the best we can do.
|
||||
p.z = min(float(z) * exp2(-23.0f), 2.0f) - 1.0f;
|
||||
#endif
|
||||
|
||||
gl_Position = p;
|
||||
gl_Position = p;
|
||||
|
||||
texture_coord();
|
||||
texture_coord();
|
||||
|
||||
VSout.c = i_c;
|
||||
VSout.t_float.z = i_f.x; // pack for with texture
|
||||
VSout.c = i_c;
|
||||
VSout.t_float.z = i_f.x; // pack for with texture
|
||||
|
||||
#if VS_POINT_SIZE
|
||||
gl_PointSize = PointSize.x;
|
||||
#endif
|
||||
#if VS_POINT_SIZE
|
||||
gl_PointSize = PointSize.x;
|
||||
#endif
|
||||
}
|
||||
|
||||
#else // VS_EXPAND
|
||||
|
||||
struct RawVertex
|
||||
{
|
||||
vec2 ST;
|
||||
uint RGBA;
|
||||
float Q;
|
||||
uint XY;
|
||||
uint Z;
|
||||
uint UV;
|
||||
uint FOG;
|
||||
vec2 ST;
|
||||
uint RGBA;
|
||||
float Q;
|
||||
uint XY;
|
||||
uint Z;
|
||||
uint UV;
|
||||
uint FOG;
|
||||
};
|
||||
|
||||
layout(std140, binding = 2) readonly buffer VertexBuffer {
|
||||
RawVertex vertex_buffer[];
|
||||
RawVertex vertex_buffer[];
|
||||
};
|
||||
|
||||
struct ProcessedVertex
|
||||
{
|
||||
vec4 p;
|
||||
vec4 t_float;
|
||||
vec4 t_int;
|
||||
vec4 c;
|
||||
vec4 p;
|
||||
vec4 t_float;
|
||||
vec4 t_int;
|
||||
vec4 c;
|
||||
};
|
||||
|
||||
ProcessedVertex load_vertex(uint index)
|
||||
{
|
||||
#if defined(GL_ARB_shader_draw_parameters) && GL_ARB_shader_draw_parameters
|
||||
RawVertex rvtx = vertex_buffer[index + gl_BaseVertexARB];
|
||||
RawVertex rvtx = vertex_buffer[index + gl_BaseVertexARB];
|
||||
#else
|
||||
RawVertex rvtx = vertex_buffer[index];
|
||||
RawVertex rvtx = vertex_buffer[index];
|
||||
#endif
|
||||
|
||||
vec2 i_st = rvtx.ST;
|
||||
vec4 i_c = vec4(uvec4(bitfieldExtract(rvtx.RGBA, 0, 8), bitfieldExtract(rvtx.RGBA, 8, 8),
|
||||
vec2 i_st = rvtx.ST;
|
||||
vec4 i_c = vec4(uvec4(bitfieldExtract(rvtx.RGBA, 0, 8), bitfieldExtract(rvtx.RGBA, 8, 8),
|
||||
bitfieldExtract(rvtx.RGBA, 16, 8), bitfieldExtract(rvtx.RGBA, 24, 8)));
|
||||
float i_q = rvtx.Q;
|
||||
uvec2 i_p = uvec2(bitfieldExtract(rvtx.XY, 0, 16), bitfieldExtract(rvtx.XY, 16, 16));
|
||||
uint i_z = rvtx.Z;
|
||||
uvec2 i_uv = uvec2(bitfieldExtract(rvtx.UV, 0, 16), bitfieldExtract(rvtx.UV, 16, 16));
|
||||
vec4 i_f = unpackUnorm4x8(rvtx.FOG);
|
||||
float i_q = rvtx.Q;
|
||||
uvec2 i_p = uvec2(bitfieldExtract(rvtx.XY, 0, 16), bitfieldExtract(rvtx.XY, 16, 16));
|
||||
uint i_z = rvtx.Z;
|
||||
uvec2 i_uv = uvec2(bitfieldExtract(rvtx.UV, 0, 16), bitfieldExtract(rvtx.UV, 16, 16));
|
||||
vec4 i_f = unpackUnorm4x8(rvtx.FOG);
|
||||
|
||||
ProcessedVertex vtx;
|
||||
ProcessedVertex vtx;
|
||||
|
||||
uint z = min(i_z, MaxDepth);
|
||||
vtx.p.xy = vec2(i_p) - vec2(0.05f, 0.05f);
|
||||
vtx.p.xy = vtx.p.xy * VertexScale - VertexOffset;
|
||||
vtx.p.w = 1.0f;
|
||||
uint z = min(i_z, MaxDepth);
|
||||
vtx.p.xy = vec2(i_p) - vec2(0.05f, 0.05f);
|
||||
vtx.p.xy = vtx.p.xy * VertexScale - VertexOffset;
|
||||
vtx.p.w = 1.0f;
|
||||
|
||||
#if HAS_CLIP_CONTROL
|
||||
vtx.p.z = float(z) * exp_min32;
|
||||
vtx.p.z = float(z) * exp_min32;
|
||||
#else
|
||||
vtx.p.z = min(float(z) * exp2(-23.0f), 2.0f) - 1.0f;
|
||||
vtx.p.z = min(float(z) * exp2(-23.0f), 2.0f) - 1.0f;
|
||||
#endif
|
||||
|
||||
vec2 uv = vec2(i_uv) - TextureOffset;
|
||||
vec2 st = i_st - TextureOffset;
|
||||
vec2 uv = vec2(i_uv) - TextureOffset;
|
||||
vec2 st = i_st - TextureOffset;
|
||||
|
||||
vtx.t_float.xy = st;
|
||||
vtx.t_float.w = i_q;
|
||||
vtx.t_float.xy = st;
|
||||
vtx.t_float.w = i_q;
|
||||
|
||||
vtx.t_int.xy = uv * TextureScale;
|
||||
vtx.t_int.xy = uv * TextureScale;
|
||||
#if VS_FST
|
||||
vtx.t_int.zw = uv;
|
||||
vtx.t_int.zw = uv;
|
||||
#else
|
||||
vtx.t_int.zw = st / TextureScale;
|
||||
vtx.t_int.zw = st / TextureScale;
|
||||
#endif
|
||||
|
||||
vtx.c = i_c;
|
||||
vtx.t_float.z = i_f.x;
|
||||
vtx.c = i_c;
|
||||
vtx.t_float.z = i_f.x;
|
||||
|
||||
return vtx;
|
||||
return vtx;
|
||||
}
|
||||
|
||||
void main()
|
||||
{
|
||||
ProcessedVertex vtx;
|
||||
ProcessedVertex vtx;
|
||||
|
||||
#if defined(GL_ARB_shader_draw_parameters) && GL_ARB_shader_draw_parameters
|
||||
uint vid = uint(gl_VertexID - gl_BaseVertexARB);
|
||||
uint vid = uint(gl_VertexID - gl_BaseVertexARB);
|
||||
#else
|
||||
uint vid = uint(gl_VertexID);
|
||||
uint vid = uint(gl_VertexID);
|
||||
#endif
|
||||
|
||||
#if VS_EXPAND == 1 // Point
|
||||
|
||||
vtx = load_vertex(vid >> 2);
|
||||
vtx = load_vertex(vid >> 2);
|
||||
|
||||
vtx.p.x += ((vid & 1u) != 0u) ? PointSize.x : 0.0f;
|
||||
vtx.p.y += ((vid & 2u) != 0u) ? PointSize.y : 0.0f;
|
||||
vtx.p.x += ((vid & 1u) != 0u) ? PointSize.x : 0.0f;
|
||||
vtx.p.y += ((vid & 2u) != 0u) ? PointSize.y : 0.0f;
|
||||
|
||||
#elif VS_EXPAND == 2 // Line
|
||||
|
||||
uint vid_base = vid >> 2;
|
||||
bool is_bottom = (vid & 2u) != 0u;
|
||||
bool is_right = (vid & 1u) != 0u;
|
||||
uint vid_other = is_bottom ? vid_base - 1 : vid_base + 1;
|
||||
vtx = load_vertex(vid_base);
|
||||
ProcessedVertex other = load_vertex(vid_other);
|
||||
uint vid_base = vid >> 2;
|
||||
bool is_bottom = (vid & 2u) != 0u;
|
||||
bool is_right = (vid & 1u) != 0u;
|
||||
uint vid_other = is_bottom ? vid_base - 1 : vid_base + 1;
|
||||
vtx = load_vertex(vid_base);
|
||||
ProcessedVertex other = load_vertex(vid_other);
|
||||
|
||||
vec2 line_vector = normalize(vtx.p.xy - other.p.xy);
|
||||
vec2 line_normal = vec2(line_vector.y, -line_vector.x);
|
||||
vec2 line_width = (line_normal * PointSize) / 2;
|
||||
// line_normal is inverted for bottom point
|
||||
vec2 offset = ((uint(is_bottom) ^ uint(is_right)) != 0u) ? line_width : -line_width;
|
||||
vtx.p.xy += offset;
|
||||
vec2 line_vector = normalize(vtx.p.xy - other.p.xy);
|
||||
vec2 line_normal = vec2(line_vector.y, -line_vector.x);
|
||||
vec2 line_width = (line_normal * PointSize) / 2;
|
||||
// line_normal is inverted for bottom point
|
||||
vec2 offset = ((uint(is_bottom) ^ uint(is_right)) != 0u) ? line_width : -line_width;
|
||||
vtx.p.xy += offset;
|
||||
|
||||
// Lines will be run as (0 1 2) (1 2 3)
|
||||
// This means that both triangles will have a point based off the top line point as their first point
|
||||
// So we don't have to do anything for !IIP
|
||||
// Lines will be run as (0 1 2) (1 2 3)
|
||||
// This means that both triangles will have a point based off the top line point as their first point
|
||||
// So we don't have to do anything for !IIP
|
||||
|
||||
#elif VS_EXPAND == 3 // Sprite
|
||||
|
||||
// Sprite points are always in pairs
|
||||
uint vid_base = vid >> 1;
|
||||
uint vid_lt = vid_base & ~1u;
|
||||
uint vid_rb = vid_base | 1u;
|
||||
// Sprite points are always in pairs
|
||||
uint vid_base = vid >> 1;
|
||||
uint vid_lt = vid_base & ~1u;
|
||||
uint vid_rb = vid_base | 1u;
|
||||
|
||||
ProcessedVertex lt = load_vertex(vid_lt);
|
||||
ProcessedVertex rb = load_vertex(vid_rb);
|
||||
vtx = rb;
|
||||
ProcessedVertex lt = load_vertex(vid_lt);
|
||||
ProcessedVertex rb = load_vertex(vid_rb);
|
||||
vtx = rb;
|
||||
|
||||
bool is_right = ((vid & 1u) != 0u);
|
||||
vtx.p.x = is_right ? lt.p.x : vtx.p.x;
|
||||
vtx.t_float.x = is_right ? lt.t_float.x : vtx.t_float.x;
|
||||
vtx.t_int.xz = is_right ? lt.t_int.xz : vtx.t_int.xz;
|
||||
bool is_right = ((vid & 1u) != 0u);
|
||||
vtx.p.x = is_right ? lt.p.x : vtx.p.x;
|
||||
vtx.t_float.x = is_right ? lt.t_float.x : vtx.t_float.x;
|
||||
vtx.t_int.xz = is_right ? lt.t_int.xz : vtx.t_int.xz;
|
||||
|
||||
bool is_bottom = ((vid & 2u) != 0u);
|
||||
vtx.p.y = is_bottom ? lt.p.y : vtx.p.y;
|
||||
vtx.t_float.y = is_bottom ? lt.t_float.y : vtx.t_float.y;
|
||||
vtx.t_int.yw = is_bottom ? lt.t_int.yw : vtx.t_int.yw;
|
||||
bool is_bottom = ((vid & 2u) != 0u);
|
||||
vtx.p.y = is_bottom ? lt.p.y : vtx.p.y;
|
||||
vtx.t_float.y = is_bottom ? lt.t_float.y : vtx.t_float.y;
|
||||
vtx.t_int.yw = is_bottom ? lt.t_int.yw : vtx.t_int.yw;
|
||||
|
||||
#endif
|
||||
|
||||
gl_Position = vtx.p;
|
||||
VSout.t_float = vtx.t_float;
|
||||
VSout.t_int = vtx.t_int;
|
||||
VSout.c = vtx.c;
|
||||
gl_Position = vtx.p;
|
||||
VSout.t_float = vtx.t_float;
|
||||
VSout.t_int = vtx.t_int;
|
||||
VSout.c = vtx.c;
|
||||
}
|
||||
|
||||
#endif // VS_EXPAND
|
||||
|
||||
@@ -34,6 +34,7 @@ if(UNIX AND NOT APPLE)
|
||||
option(USE_LEGACY_USER_DIRECTORY "Use legacy home/PCSX2 user directory instead of XDG standard" OFF)
|
||||
option(X11_API "Enable X11 support" ON)
|
||||
option(WAYLAND_API "Enable Wayland support" ON)
|
||||
option(DBUS_API "Enable DBus support for screensaver inhibiting" ON)
|
||||
endif()
|
||||
|
||||
if(APPLE)
|
||||
|
||||
@@ -218,6 +218,14 @@ if(APPLE)
|
||||
target_link_options(common PRIVATE -fobjc-link-runtime)
|
||||
endif()
|
||||
|
||||
if(DBUS_API)
|
||||
target_compile_definitions(common PRIVATE DBUS_API)
|
||||
find_package(PkgConfig REQUIRED)
|
||||
pkg_check_modules(DBUS REQUIRED dbus-1)
|
||||
target_include_directories(common PRIVATE ${DBUS_INCLUDE_DIRS})
|
||||
target_link_libraries(common PRIVATE ${DBUS_LINK_LIBRARIES})
|
||||
endif()
|
||||
|
||||
if(USE_OPENGL)
|
||||
if(WIN32)
|
||||
target_sources(common PRIVATE
|
||||
|
||||
@@ -31,6 +31,10 @@
|
||||
#include "common/Threading.h"
|
||||
#include "common/WindowInfo.h"
|
||||
|
||||
#ifdef DBUS_API
|
||||
#include <dbus/dbus.h>
|
||||
#endif
|
||||
|
||||
// Returns 0 on failure (not supported by the operating system).
|
||||
u64 GetPhysicalMemory()
|
||||
{
|
||||
@@ -69,7 +73,74 @@ std::string GetOSVersionString()
|
||||
#endif
|
||||
}
|
||||
|
||||
#ifdef X11_API
|
||||
#ifdef DBUS_API
|
||||
|
||||
static dbus_uint32_t s_screensaver_dbus_cookie;
|
||||
bool ChangeScreenSaverStateDBus(const bool inhibit_requested, const char* program_name, const char* reason)
|
||||
{
|
||||
// "error_dbus" doesn't need to be cleared in the end with "dbus_message_unref" at least if there is
|
||||
// no error set, since calling "dbus_error_free" reinitializes it like "dbus_error_init" after freeing.
|
||||
DBusError error_dbus;
|
||||
dbus_error_init(&error_dbus);
|
||||
DBusConnection* connection = nullptr;
|
||||
DBusMessage* message = nullptr;
|
||||
DBusMessage* response = nullptr;
|
||||
// Initialized here because initializations should be before "goto" statements.
|
||||
const char* bus_method = (inhibit_requested) ? "Inhibit" : "UnInhibit";
|
||||
// "dbus_bus_get" gets a pointer to the same connection in libdbus, if exists, without creating a new connection.
|
||||
// this doesn't need to be deleted, except if there's an error then calling "dbus_connection_unref", to free it,
|
||||
// might be better so a new connection is established on the next try.
|
||||
if (!(connection = dbus_bus_get(DBUS_BUS_SESSION, &error_dbus)) || (dbus_error_is_set(&error_dbus)))
|
||||
goto cleanup;
|
||||
if (!(message = dbus_message_new_method_call("org.freedesktop.ScreenSaver", "/org/freedesktop/ScreenSaver", "org.freedesktop.ScreenSaver", bus_method)))
|
||||
goto cleanup;
|
||||
// Initialize an append iterator for the message, gets freed with the message.
|
||||
DBusMessageIter message_itr;
|
||||
dbus_message_iter_init_append(message, &message_itr);
|
||||
if (inhibit_requested)
|
||||
{
|
||||
// Append process/window name.
|
||||
if (!dbus_message_iter_append_basic(&message_itr, DBUS_TYPE_STRING, &program_name))
|
||||
goto cleanup;
|
||||
// Append reason for inhibiting the screensaver.
|
||||
if (!dbus_message_iter_append_basic(&message_itr, DBUS_TYPE_STRING, &reason))
|
||||
goto cleanup;
|
||||
}
|
||||
else
|
||||
{
|
||||
// Only Append the cookie.
|
||||
if (!dbus_message_iter_append_basic(&message_itr, DBUS_TYPE_UINT32, &s_screensaver_dbus_cookie))
|
||||
goto cleanup;
|
||||
s_screensaver_dbus_cookie = 0;
|
||||
}
|
||||
// Send message and get response.
|
||||
if (!(response = dbus_connection_send_with_reply_and_block(connection, message, DBUS_TIMEOUT_USE_DEFAULT, &error_dbus))
|
||||
|| dbus_error_is_set(&error_dbus))
|
||||
goto cleanup;
|
||||
if (inhibit_requested)
|
||||
{
|
||||
// Get the cookie from the response message.
|
||||
if (!dbus_message_get_args(response, &error_dbus, DBUS_TYPE_UINT32, &s_screensaver_dbus_cookie, DBUS_TYPE_INVALID))
|
||||
goto cleanup;
|
||||
}
|
||||
dbus_message_unref(message);
|
||||
dbus_message_unref(response);
|
||||
return true;
|
||||
cleanup:
|
||||
if (dbus_error_is_set(&error_dbus))
|
||||
dbus_error_free(&error_dbus);
|
||||
if (connection)
|
||||
dbus_connection_unref(connection);
|
||||
if (message)
|
||||
dbus_message_unref(message);
|
||||
if (response)
|
||||
dbus_message_unref(response);
|
||||
return false;
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
#if !defined(DBUS_API) && defined(X11_API)
|
||||
|
||||
static bool SetScreensaverInhibitX11(const WindowInfo& wi, bool inhibit)
|
||||
{
|
||||
@@ -88,8 +159,6 @@ static bool SetScreensaverInhibitX11(const WindowInfo& wi, bool inhibit)
|
||||
return (res == 0);
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
static bool SetScreensaverInhibit(const WindowInfo& wi, bool inhibit)
|
||||
{
|
||||
switch (wi.type)
|
||||
@@ -106,8 +175,18 @@ static bool SetScreensaverInhibit(const WindowInfo& wi, bool inhibit)
|
||||
|
||||
static std::optional<WindowInfo> s_inhibit_window_info;
|
||||
|
||||
#endif
|
||||
|
||||
bool WindowInfo::InhibitScreensaver(const WindowInfo& wi, bool inhibit)
|
||||
{
|
||||
|
||||
#ifdef DBUS_API
|
||||
|
||||
return ChangeScreenSaverStateDBus(inhibit, "PCSX2", "PCSX2 VM is running.");
|
||||
|
||||
#else
|
||||
|
||||
//ChangeScreenSaverStateDBus
|
||||
if (s_inhibit_window_info.has_value())
|
||||
{
|
||||
// Bit of extra logic here, because wx spams it and we don't want to
|
||||
@@ -132,6 +211,9 @@ bool WindowInfo::InhibitScreensaver(const WindowInfo& wi, bool inhibit)
|
||||
|
||||
s_inhibit_window_info = wi;
|
||||
return true;
|
||||
|
||||
#endif
|
||||
|
||||
}
|
||||
|
||||
bool Common::PlaySoundAsync(const char* path)
|
||||
|
||||
+173
-159
@@ -15,197 +15,211 @@
|
||||
|
||||
#include "common/Perf.h"
|
||||
#include "common/Pcsx2Defs.h"
|
||||
#ifdef __unix__
|
||||
#include <unistd.h>
|
||||
#endif
|
||||
#include "common/Assertions.h"
|
||||
#include "common/StringUtil.h"
|
||||
|
||||
#ifdef ENABLE_VTUNE
|
||||
#include "jitprofiling.h"
|
||||
#endif
|
||||
|
||||
#include <string> // std::string
|
||||
#include <cstring> // strncpy
|
||||
#include <algorithm> // std::remove_if
|
||||
#include <array>
|
||||
#include <cstring>
|
||||
|
||||
#ifdef __linux__
|
||||
#include <atomic>
|
||||
#include <ctime>
|
||||
#include <mutex>
|
||||
#include <elf.h>
|
||||
#include <unistd.h>
|
||||
#include <sys/mman.h>
|
||||
#include <sys/syscall.h>
|
||||
#endif
|
||||
|
||||
//#define ProfileWithPerf
|
||||
#define MERGE_BLOCK_RESULT
|
||||
//#define ProfileWithPerfJitDump
|
||||
|
||||
#ifdef ENABLE_VTUNE
|
||||
#ifdef _WIN32
|
||||
#if defined(ENABLE_VTUNE) && defined(_WIN32)
|
||||
#pragma comment(lib, "jitprofiling.lib")
|
||||
#endif
|
||||
#endif
|
||||
|
||||
namespace Perf
|
||||
{
|
||||
// Warning object aren't thread safe
|
||||
InfoVector any("");
|
||||
InfoVector ee("EE");
|
||||
InfoVector iop("IOP");
|
||||
InfoVector vu("VU");
|
||||
InfoVector vif("VIF");
|
||||
Group any("");
|
||||
Group ee("EE");
|
||||
Group iop("IOP");
|
||||
Group vu0("VU0");
|
||||
Group vu1("VU1");
|
||||
Group vif("VIF");
|
||||
|
||||
// Perf is only supported on linux
|
||||
#if defined(__linux__) && (defined(ProfileWithPerf) || defined(ENABLE_VTUNE))
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////
|
||||
// Implementation of the Info object
|
||||
////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
Info::Info(uptr x86, u32 size, const char* symbol)
|
||||
: m_x86(x86)
|
||||
, m_size(size)
|
||||
, m_dynamic(false)
|
||||
#if defined(__linux__) && defined(ProfileWithPerf)
|
||||
static std::FILE* s_map_file = nullptr;
|
||||
static bool s_map_file_opened = false;
|
||||
static std::mutex s_mutex;
|
||||
static void RegisterMethod(const void* ptr, size_t size, const char* symbol)
|
||||
{
|
||||
strncpy(m_symbol, symbol, sizeof(m_symbol));
|
||||
}
|
||||
std::unique_lock lock(s_mutex);
|
||||
|
||||
Info::Info(uptr x86, u32 size, const char* symbol, u32 pc)
|
||||
: m_x86(x86)
|
||||
, m_size(size)
|
||||
, m_dynamic(true)
|
||||
{
|
||||
snprintf(m_symbol, sizeof(m_symbol), "%s_0x%08x", symbol, pc);
|
||||
}
|
||||
|
||||
void Info::Print(FILE* fp)
|
||||
{
|
||||
fprintf(fp, "%x %x %s\n", m_x86, m_size, m_symbol);
|
||||
}
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////
|
||||
// Implementation of the InfoVector object
|
||||
////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
InfoVector::InfoVector(const char* prefix)
|
||||
{
|
||||
strncpy(m_prefix, prefix, sizeof(m_prefix));
|
||||
#ifdef ENABLE_VTUNE
|
||||
m_vtune_id = iJIT_GetNewMethodID();
|
||||
#else
|
||||
m_vtune_id = 0;
|
||||
#endif
|
||||
}
|
||||
|
||||
void InfoVector::print(FILE* fp)
|
||||
{
|
||||
for (auto&& it : m_v)
|
||||
it.Print(fp);
|
||||
}
|
||||
|
||||
void InfoVector::map(uptr x86, u32 size, const char* symbol)
|
||||
{
|
||||
// This function is typically used for dispatcher and recompiler.
|
||||
// Dispatchers are on a page and must always be kept.
|
||||
// Recompilers are much bigger (TODO check VIF) and are only
|
||||
// useful when MERGE_BLOCK_RESULT is defined
|
||||
#if defined(ENABLE_VTUNE) || !defined(MERGE_BLOCK_RESULT)
|
||||
u32 max_code_size = 16 * _1kb;
|
||||
#else
|
||||
u32 max_code_size = _1gb;
|
||||
#endif
|
||||
|
||||
if (size < max_code_size)
|
||||
if (!s_map_file)
|
||||
{
|
||||
m_v.emplace_back(x86, size, symbol);
|
||||
if (s_map_file_opened)
|
||||
return;
|
||||
|
||||
#ifdef ENABLE_VTUNE
|
||||
std::string name = std::string(symbol);
|
||||
|
||||
iJIT_Method_Load ml;
|
||||
|
||||
memset(&ml, 0, sizeof(ml));
|
||||
|
||||
ml.method_id = iJIT_GetNewMethodID();
|
||||
ml.method_name = (char*)name.c_str();
|
||||
ml.method_load_address = (void*)x86;
|
||||
ml.method_size = size;
|
||||
|
||||
iJIT_NotifyEvent(iJVM_EVENT_TYPE_METHOD_LOAD_FINISHED, &ml);
|
||||
|
||||
//fprintf(stderr, "mapF %s: %p size %dKB\n", ml.method_name, ml.method_load_address, ml.method_size / 1024u);
|
||||
#endif
|
||||
char file[256];
|
||||
snprintf(file, std::size(file), "/tmp/perf-%d.map", getpid());
|
||||
s_map_file = std::fopen(file, "wb");
|
||||
s_map_file_opened = true;
|
||||
if (!s_map_file)
|
||||
return;
|
||||
}
|
||||
|
||||
std::fprintf(s_map_file, "%" PRIx64 " %zx %s\n", static_cast<u64>(reinterpret_cast<uintptr_t>(ptr)), size, symbol);
|
||||
std::fflush(s_map_file);
|
||||
}
|
||||
#elif defined(__linux__) && defined(ProfileWithPerfJitDump)
|
||||
enum : u32
|
||||
{
|
||||
JIT_CODE_LOAD = 0,
|
||||
JIT_CODE_MOVE = 1,
|
||||
JIT_CODE_DEBUG_INFO = 2,
|
||||
JIT_CODE_CLOSE = 3,
|
||||
JIT_CODE_UNWINDING_INFO = 4
|
||||
};
|
||||
|
||||
#pragma pack(push, 1)
|
||||
struct JITDUMP_HEADER
|
||||
{
|
||||
u32 magic = 0x4A695444; // JiTD
|
||||
u32 version = 1;
|
||||
u32 header_size = sizeof(JITDUMP_HEADER);
|
||||
u32 elf_mach;
|
||||
u32 pad1 = 0;
|
||||
u32 pid;
|
||||
u64 timestamp;
|
||||
u64 flags = 0;
|
||||
};
|
||||
struct JITDUMP_RECORD_HEADER
|
||||
{
|
||||
u32 id;
|
||||
u32 total_size;
|
||||
u64 timestamp;
|
||||
};
|
||||
struct JITDUMP_CODE_LOAD
|
||||
{
|
||||
JITDUMP_RECORD_HEADER header;
|
||||
u32 pid;
|
||||
u32 tid;
|
||||
u64 vma;
|
||||
u64 code_addr;
|
||||
u64 code_size;
|
||||
u64 code_index;
|
||||
// name
|
||||
};
|
||||
#pragma pack(pop)
|
||||
|
||||
static u64 JitDumpTimestamp()
|
||||
{
|
||||
struct timespec ts = {};
|
||||
clock_gettime(CLOCK_MONOTONIC, &ts);
|
||||
return (static_cast<u64>(ts.tv_sec) * 1000000000ULL) + static_cast<u64>(ts.tv_nsec);
|
||||
}
|
||||
|
||||
void InfoVector::map(uptr x86, u32 size, u32 pc)
|
||||
static FILE* s_jitdump_file = nullptr;
|
||||
static bool s_jitdump_file_opened = false;
|
||||
static std::mutex s_jitdump_mutex;
|
||||
static u32 s_jitdump_record_id;
|
||||
|
||||
static void RegisterMethod(const void* ptr, size_t size, const char* symbol)
|
||||
{
|
||||
#ifndef MERGE_BLOCK_RESULT
|
||||
m_v.emplace_back(x86, size, m_prefix, pc);
|
||||
#endif
|
||||
const u32 namelen = std::strlen(symbol) + 1;
|
||||
|
||||
#ifdef ENABLE_VTUNE
|
||||
iJIT_Method_Load_V2 ml;
|
||||
std::unique_lock lock(s_jitdump_mutex);
|
||||
if (!s_jitdump_file)
|
||||
{
|
||||
if (!s_jitdump_file_opened)
|
||||
{
|
||||
char file[256];
|
||||
snprintf(file, std::size(file), "jit-%d.dump", getpid());
|
||||
s_jitdump_file = fopen(file, "w+b");
|
||||
s_jitdump_file_opened = true;
|
||||
if (!s_jitdump_file)
|
||||
return;
|
||||
}
|
||||
|
||||
memset(&ml, 0, sizeof(ml));
|
||||
void* perf_marker = mmap(nullptr, 4096, PROT_READ | PROT_EXEC, MAP_PRIVATE, fileno(s_jitdump_file), 0);
|
||||
pxAssertRel(perf_marker != MAP_FAILED, "Map perf marker");
|
||||
|
||||
#ifdef MERGE_BLOCK_RESULT
|
||||
ml.method_id = m_vtune_id;
|
||||
ml.method_name = m_prefix;
|
||||
#else
|
||||
std::string name = std::string(m_prefix) + "_" + std::to_string(pc);
|
||||
JITDUMP_HEADER jh = {};
|
||||
jh.elf_mach = EM_X86_64;
|
||||
jh.pid = getpid();
|
||||
jh.timestamp = JitDumpTimestamp();
|
||||
std::fwrite(&jh, sizeof(jh), 1, s_jitdump_file);
|
||||
}
|
||||
|
||||
JITDUMP_CODE_LOAD cl = {};
|
||||
cl.header.id = JIT_CODE_LOAD;
|
||||
cl.header.total_size = sizeof(cl) + namelen + static_cast<u32>(size);
|
||||
cl.header.timestamp = JitDumpTimestamp();
|
||||
cl.pid = getpid();
|
||||
cl.tid = syscall(SYS_gettid);
|
||||
cl.vma = 0;
|
||||
cl.code_addr = static_cast<u64>(reinterpret_cast<uintptr_t>(ptr));
|
||||
cl.code_size = static_cast<u64>(size);
|
||||
cl.code_index = s_jitdump_record_id++;
|
||||
std::fwrite(&cl, sizeof(cl), 1, s_jitdump_file);
|
||||
std::fwrite(symbol, namelen, 1, s_jitdump_file);
|
||||
std::fwrite(ptr, size, 1, s_jitdump_file);
|
||||
std::fflush(s_jitdump_file);
|
||||
}
|
||||
#elif defined(ENABLE_VTUNE)
|
||||
static void RegisterMethod(const void* ptr, size_t size, const char* symbol)
|
||||
{
|
||||
iJIT_Method_Load_V2 ml = {};
|
||||
ml.method_id = iJIT_GetNewMethodID();
|
||||
ml.method_name = (char*)name.c_str();
|
||||
#endif
|
||||
ml.method_load_address = (void*)x86;
|
||||
ml.method_size = size;
|
||||
|
||||
ml.method_name = const_cast<char*>(symbol);
|
||||
ml.method_load_address = ptr;
|
||||
ml.method_size = static_cast<unsigned int>(size);
|
||||
iJIT_NotifyEvent(iJVM_EVENT_TYPE_METHOD_LOAD_FINISHED_V2, &ml);
|
||||
|
||||
//fprintf(stderr, "mapB %s: %p size %d\n", ml.method_name, ml.method_load_address, ml.method_size);
|
||||
#endif
|
||||
}
|
||||
|
||||
void InfoVector::reset()
|
||||
{
|
||||
auto dynamic = std::remove_if(m_v.begin(), m_v.end(), [](Info i) { return i.m_dynamic; });
|
||||
m_v.erase(dynamic, m_v.end());
|
||||
}
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////
|
||||
// Global function
|
||||
////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
void dump()
|
||||
{
|
||||
char file[256];
|
||||
snprintf(file, 250, "/tmp/perf-%d.map", getpid());
|
||||
FILE* fp = fopen(file, "w");
|
||||
|
||||
any.print(fp);
|
||||
ee.print(fp);
|
||||
iop.print(fp);
|
||||
vu.print(fp);
|
||||
|
||||
if (fp)
|
||||
fclose(fp);
|
||||
}
|
||||
|
||||
void dump_and_reset()
|
||||
{
|
||||
dump();
|
||||
|
||||
any.reset();
|
||||
ee.reset();
|
||||
iop.reset();
|
||||
vu.reset();
|
||||
}
|
||||
|
||||
#else
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////
|
||||
// Dummy implementation
|
||||
////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
InfoVector::InfoVector(const char* prefix)
|
||||
: m_vtune_id(0)
|
||||
static void RegisterMethod(const void* ptr, size_t size, const char* method)
|
||||
{
|
||||
}
|
||||
void InfoVector::map(uptr x86, u32 size, const char* symbol) {}
|
||||
void InfoVector::map(uptr x86, u32 size, u32 pc) {}
|
||||
void InfoVector::reset() {}
|
||||
#endif
|
||||
|
||||
void dump() {}
|
||||
void dump_and_reset() {}
|
||||
#if (defined(__linux__) && (defined(ProfileWithPerf) || defined(ProfileWithPerfJitDump))) || defined(ENABLE_VTUNE)
|
||||
void Group::Register(const void* ptr, size_t size, const char* symbol)
|
||||
{
|
||||
char full_symbol[128];
|
||||
if (HasPrefix())
|
||||
std::snprintf(full_symbol, std::size(full_symbol), "%s_%s", m_prefix, symbol);
|
||||
else
|
||||
StringUtil::Strlcpy(full_symbol, symbol, std::size(full_symbol));
|
||||
RegisterMethod(ptr, size, full_symbol);
|
||||
}
|
||||
|
||||
void Group::RegisterPC(const void* ptr, size_t size, u32 pc)
|
||||
{
|
||||
char full_symbol[128];
|
||||
if (HasPrefix())
|
||||
std::snprintf(full_symbol, std::size(full_symbol), "%s_%08X", m_prefix, pc);
|
||||
else
|
||||
std::snprintf(full_symbol, std::size(full_symbol), "%08X", pc);
|
||||
RegisterMethod(ptr, size, full_symbol);
|
||||
}
|
||||
|
||||
void Group::RegisterKey(const void* ptr, size_t size, const char* prefix, u64 key)
|
||||
{
|
||||
char full_symbol[128];
|
||||
if (HasPrefix())
|
||||
std::snprintf(full_symbol, std::size(full_symbol), "%s_%s%016" PRIX64, m_prefix, prefix, key);
|
||||
else
|
||||
std::snprintf(full_symbol, std::size(full_symbol), "%s%016" PRIX64, prefix, key);
|
||||
RegisterMethod(ptr, size, full_symbol);
|
||||
}
|
||||
#else
|
||||
void Group::Register(const void* ptr, size_t size, const char* symbol) {}
|
||||
void Group::RegisterPC(const void* ptr, size_t size, u32 pc) {}
|
||||
void Group::RegisterKey(const void* ptr, size_t size, const char* prefix, u64 key) {}
|
||||
#endif
|
||||
} // namespace Perf
|
||||
|
||||
+13
-32
@@ -21,42 +21,23 @@
|
||||
|
||||
namespace Perf
|
||||
{
|
||||
|
||||
struct Info
|
||||
class Group
|
||||
{
|
||||
uptr m_x86;
|
||||
u32 m_size;
|
||||
char m_symbol[20];
|
||||
// The idea is to keep static zones that are set only
|
||||
// once.
|
||||
bool m_dynamic;
|
||||
|
||||
Info(uptr x86, u32 size, const char* symbol);
|
||||
Info(uptr x86, u32 size, const char* symbol, u32 pc);
|
||||
void Print(FILE* fp);
|
||||
};
|
||||
|
||||
class InfoVector
|
||||
{
|
||||
std::vector<Info> m_v;
|
||||
char m_prefix[20];
|
||||
unsigned int m_vtune_id;
|
||||
const char* m_prefix;
|
||||
|
||||
public:
|
||||
InfoVector(const char* prefix);
|
||||
constexpr Group(const char* prefix) : m_prefix(prefix) {}
|
||||
bool HasPrefix() const { return (m_prefix && m_prefix[0]); }
|
||||
|
||||
void print(FILE* fp);
|
||||
void map(uptr x86, u32 size, const char* symbol);
|
||||
void map(uptr x86, u32 size, u32 pc);
|
||||
void reset();
|
||||
void Register(const void* ptr, size_t size, const char* symbol);
|
||||
void RegisterPC(const void* ptr, size_t size, u32 pc);
|
||||
void RegisterKey(const void* ptr, size_t size, const char* prefix, u64 key);
|
||||
};
|
||||
|
||||
void dump();
|
||||
void dump_and_reset();
|
||||
|
||||
extern InfoVector any;
|
||||
extern InfoVector ee;
|
||||
extern InfoVector iop;
|
||||
extern InfoVector vu;
|
||||
extern InfoVector vif;
|
||||
extern Group any;
|
||||
extern Group ee;
|
||||
extern Group iop;
|
||||
extern Group vu0;
|
||||
extern Group vu1;
|
||||
extern Group vif;
|
||||
} // namespace Perf
|
||||
|
||||
+23
-44
@@ -26,6 +26,8 @@
|
||||
#include <array>
|
||||
#include <cstring>
|
||||
|
||||
#include "fmt/format.h"
|
||||
|
||||
#ifdef _WIN32
|
||||
#include "common/RedtapeWindows.h"
|
||||
#else
|
||||
@@ -192,83 +194,60 @@ namespace Vulkan
|
||||
|
||||
Context::GPUList Context::EnumerateGPUs(VkInstance instance)
|
||||
{
|
||||
GPUList gpus;
|
||||
|
||||
u32 gpu_count = 0;
|
||||
VkResult res = vkEnumeratePhysicalDevices(instance, &gpu_count, nullptr);
|
||||
if ((res != VK_SUCCESS && res != VK_INCOMPLETE) || gpu_count == 0)
|
||||
{
|
||||
LOG_VULKAN_ERROR(res, "vkEnumeratePhysicalDevices (1) failed: ");
|
||||
return {};
|
||||
return gpus;
|
||||
}
|
||||
|
||||
GPUList gpus;
|
||||
gpus.resize(gpu_count);
|
||||
|
||||
res = vkEnumeratePhysicalDevices(instance, &gpu_count, gpus.data());
|
||||
std::vector<VkPhysicalDevice> physical_devices(gpu_count);
|
||||
res = vkEnumeratePhysicalDevices(instance, &gpu_count, physical_devices.data());
|
||||
if (res == VK_INCOMPLETE)
|
||||
{
|
||||
Console.Warning("First vkEnumeratePhysicalDevices() call returned %zu devices, but second returned %u", gpus.size(), gpu_count);
|
||||
Console.Warning("First vkEnumeratePhysicalDevices() call returned %zu devices, but second returned %u",
|
||||
physical_devices.size(), gpu_count);
|
||||
}
|
||||
else if (res != VK_SUCCESS)
|
||||
{
|
||||
LOG_VULKAN_ERROR(res, "vkEnumeratePhysicalDevices (2) failed: ");
|
||||
return {};
|
||||
return gpus;
|
||||
}
|
||||
|
||||
// Maybe we lost a GPU?
|
||||
if (gpu_count < gpus.size())
|
||||
gpus.resize(gpu_count);
|
||||
if (gpu_count < physical_devices.size())
|
||||
physical_devices.resize(gpu_count);
|
||||
|
||||
return gpus;
|
||||
}
|
||||
|
||||
Context::GPUNameList Context::EnumerateGPUNames(VkInstance instance)
|
||||
{
|
||||
u32 gpu_count = 0;
|
||||
VkResult res = vkEnumeratePhysicalDevices(instance, &gpu_count, nullptr);
|
||||
if (res != VK_SUCCESS || gpu_count == 0)
|
||||
{
|
||||
LOG_VULKAN_ERROR(res, "vkEnumeratePhysicalDevices failed: ");
|
||||
return {};
|
||||
}
|
||||
|
||||
GPUList gpus;
|
||||
gpus.resize(gpu_count);
|
||||
|
||||
res = vkEnumeratePhysicalDevices(instance, &gpu_count, gpus.data());
|
||||
if (res != VK_SUCCESS)
|
||||
{
|
||||
LOG_VULKAN_ERROR(res, "vkEnumeratePhysicalDevices failed: ");
|
||||
return {};
|
||||
}
|
||||
|
||||
GPUNameList gpu_names;
|
||||
gpu_names.reserve(gpu_count);
|
||||
for (u32 i = 0; i < gpu_count; i++)
|
||||
gpus.reserve(physical_devices.size());
|
||||
for (VkPhysicalDevice device : physical_devices)
|
||||
{
|
||||
VkPhysicalDeviceProperties props = {};
|
||||
vkGetPhysicalDeviceProperties(gpus[i], &props);
|
||||
vkGetPhysicalDeviceProperties(device, &props);
|
||||
|
||||
std::string gpu_name(props.deviceName);
|
||||
std::string gpu_name = props.deviceName;
|
||||
|
||||
// handle duplicate adapter names
|
||||
if (std::any_of(gpu_names.begin(), gpu_names.end(),
|
||||
[&gpu_name](const std::string& other) { return (gpu_name == other); }))
|
||||
if (std::any_of(gpus.begin(), gpus.end(),
|
||||
[&gpu_name](const auto& other) { return (gpu_name == other.second); }))
|
||||
{
|
||||
std::string original_adapter_name = std::move(gpu_name);
|
||||
|
||||
u32 current_extra = 2;
|
||||
do
|
||||
{
|
||||
gpu_name = StringUtil::StdStringFromFormat("%s (%u)", original_adapter_name.c_str(), current_extra);
|
||||
gpu_name = fmt::format("{} ({})", original_adapter_name, current_extra);
|
||||
current_extra++;
|
||||
} while (std::any_of(gpu_names.begin(), gpu_names.end(),
|
||||
[&gpu_name](const std::string& other) { return (gpu_name == other); }));
|
||||
} while (std::any_of(gpus.begin(), gpus.end(),
|
||||
[&gpu_name](const auto& other) { return (gpu_name == other.second); }));
|
||||
}
|
||||
|
||||
gpu_names.push_back(std::move(gpu_name));
|
||||
gpus.emplace_back(device, std::move(gpu_name));
|
||||
}
|
||||
|
||||
return gpu_names;
|
||||
return gpus;
|
||||
}
|
||||
|
||||
bool Context::Create(VkInstance instance, VkSurfaceKHR surface, VkPhysicalDevice physical_device,
|
||||
|
||||
@@ -66,10 +66,8 @@ namespace Vulkan
|
||||
const WindowInfo& wi, bool enable_debug_utils, bool enable_validation_layer);
|
||||
|
||||
// Returns a list of Vulkan-compatible GPUs.
|
||||
using GPUList = std::vector<VkPhysicalDevice>;
|
||||
using GPUNameList = std::vector<std::string>;
|
||||
using GPUList = std::vector<std::pair<VkPhysicalDevice, std::string>>;
|
||||
static GPUList EnumerateGPUs(VkInstance instance);
|
||||
static GPUNameList EnumerateGPUNames(VkInstance instance);
|
||||
|
||||
// Creates a new context and sets it up as global.
|
||||
static bool Create(VkInstance instance, VkSurfaceKHR surface, VkPhysicalDevice physical_device,
|
||||
|
||||
@@ -319,8 +319,12 @@ bool BreakpointModel::insertBreakpointRows(int row, int count, std::vector<Break
|
||||
if (breakpoints.size() != static_cast<size_t>(count))
|
||||
return false;
|
||||
|
||||
beginInsertRows(index, row, row + count);
|
||||
beginInsertRows(index, row, row + (count - 1));
|
||||
|
||||
// After endInsertRows, Qt will try and validate our new rows
|
||||
// Because we add the breakpoints off of the UI thread, our new rows may not be visible yet
|
||||
// To prevent the (seemingly harmless?) warning emitted by enderInsertRows, add the breakpoints manually here as well
|
||||
m_breakpoints.insert(m_breakpoints.begin(), breakpoints.begin(), breakpoints.end());
|
||||
for (const auto& bp_mc : breakpoints)
|
||||
{
|
||||
if (const auto* bp = std::get_if<BreakPoint>(&bp_mc))
|
||||
|
||||
@@ -359,6 +359,7 @@ void GSclose()
|
||||
{
|
||||
CloseGSRenderer();
|
||||
CloseGSDevice(true);
|
||||
Host::ReleaseRenderWindow();
|
||||
}
|
||||
|
||||
void GSreset(bool hardware_reset)
|
||||
|
||||
@@ -0,0 +1,36 @@
|
||||
<?xml version="1.0" encoding="utf-8"?>
|
||||
<AutoVisualizer xmlns="http://schemas.microsoft.com/vstudio/debugger/natvis/2010">
|
||||
<Type Name="FastList<*>">
|
||||
<DisplayString>{{ size={m_free_indexes_stack_top} }}</DisplayString>
|
||||
<Expand>
|
||||
<Item Name="[size]" ExcludeView="simple">m_free_indexes_stack_top</Item>
|
||||
<Item Name="[capacity]" ExcludeView="simple">m_capacity - 1</Item>
|
||||
|
||||
<CustomListItems MaxItemsPerView="5000" ExcludeView="simple">
|
||||
<Variable Name="index" InitialValue="m_buffer[0].next_index" />
|
||||
|
||||
<Size>m_free_indexes_stack_top</Size>
|
||||
<Loop>
|
||||
<Item>m_buffer[index].data,na</Item>
|
||||
<Exec>index = m_buffer[index].next_index</Exec>
|
||||
</Loop>
|
||||
</CustomListItems>
|
||||
</Expand>
|
||||
</Type>
|
||||
|
||||
<Type Name="GSVector2T<*>">
|
||||
<DisplayString>{{ {x}, {y} }}</DisplayString>
|
||||
</Type>
|
||||
|
||||
<Type Name="GSVector4">
|
||||
<DisplayString>{{ {F32[0]}, {F32[1]}, {F32[2]}, {F32[3]} }}</DisplayString>
|
||||
</Type>
|
||||
<Type Name="GSVector4i">
|
||||
<DisplayString>{{ {I32[0]}, {I32[1]}, {I32[2]}, {I32[3]} }}</DisplayString>
|
||||
</Type>
|
||||
|
||||
<Type Name="GSTextureCache::Target">
|
||||
<DisplayString Condition="m_type == 0">{{ RT @ BP={m_TEX0.TBP0,X}-{m_end_block,X} BW={m_TEX0.TBW} PSM={m_TEX0.PSM,X} {m_unscaled_size.x}x{m_unscaled_size.y} {m_valid.z},{m_valid.w} }}</DisplayString>
|
||||
<DisplayString Condition="m_type == 1">{{ Depth @ BP={m_TEX0.TBP0,X}-{m_end_block,X} BW={m_TEX0.TBW} PSM={m_TEX0.PSM,X} {m_unscaled_size.x}x{m_unscaled_size.y} {m_valid.z},{m_valid.w} }}</DisplayString>
|
||||
</Type>
|
||||
</AutoVisualizer>
|
||||
+35
-45
@@ -388,6 +388,40 @@ float GSState::GetTvRefreshRate()
|
||||
__assume(0); // unreachable
|
||||
}
|
||||
|
||||
const char* GSState::GetFlushReasonString(GSFlushReason reason)
|
||||
{
|
||||
switch (reason)
|
||||
{
|
||||
case GSFlushReason::RESET:
|
||||
return "RESET";
|
||||
case GSFlushReason::CONTEXTCHANGE:
|
||||
return "CONTEXT CHANGE";
|
||||
case GSFlushReason::CLUTCHANGE:
|
||||
return "CLUT CHANGE (RELOAD REQ)";
|
||||
case GSFlushReason::GSTRANSFER:
|
||||
return "GS TRANSFER";
|
||||
case GSFlushReason::UPLOADDIRTYTEX:
|
||||
return "GS UPLOAD OVERWRITES CURRENT TEXTURE OR CLUT";
|
||||
case GSFlushReason::LOCALTOLOCALMOVE:
|
||||
return "GS LOCAL TO LOCAL OVERWRITES CURRENT TEXTURE OR CLUT";
|
||||
case GSFlushReason::DOWNLOADFIFO:
|
||||
return "DOWNLOAD FIFO";
|
||||
case GSFlushReason::SAVESTATE:
|
||||
return "SAVESTATE";
|
||||
case GSFlushReason::LOADSTATE:
|
||||
return "LOAD SAVESTATE";
|
||||
case GSFlushReason::AUTOFLUSH:
|
||||
return "AUTOFLUSH OVERLAP DETECTED";
|
||||
case GSFlushReason::VSYNC:
|
||||
return "VSYNC";
|
||||
case GSFlushReason::GSREOPEN:
|
||||
return "GS REOPEN";
|
||||
case GSFlushReason::UNKNOWN:
|
||||
default:
|
||||
return "UNKNOWN";
|
||||
}
|
||||
}
|
||||
|
||||
void GSState::DumpVertices(const std::string& filename)
|
||||
{
|
||||
std::ofstream file(filename);
|
||||
@@ -395,51 +429,7 @@ void GSState::DumpVertices(const std::string& filename)
|
||||
if (!file.is_open())
|
||||
return;
|
||||
|
||||
file << "FLUSH REASON: ";
|
||||
|
||||
switch (m_state_flush_reason)
|
||||
{
|
||||
case GSFlushReason::RESET:
|
||||
file << "RESET";
|
||||
break;
|
||||
case GSFlushReason::CONTEXTCHANGE:
|
||||
file << "CONTEXT CHANGE";
|
||||
break;
|
||||
case GSFlushReason::CLUTCHANGE:
|
||||
file << "CLUT CHANGE (RELOAD REQ)";
|
||||
break;
|
||||
case GSFlushReason::GSTRANSFER:
|
||||
file << "GS TRANSFER";
|
||||
break;
|
||||
case GSFlushReason::UPLOADDIRTYTEX:
|
||||
file << "GS UPLOAD OVERWRITES CURRENT TEXTURE OR CLUT";
|
||||
break;
|
||||
case GSFlushReason::LOCALTOLOCALMOVE:
|
||||
file << "GS LOCAL TO LOCAL OVERWRITES CURRENT TEXTURE OR CLUT";
|
||||
break;
|
||||
case GSFlushReason::DOWNLOADFIFO:
|
||||
file << "DOWNLOAD FIFO";
|
||||
break;
|
||||
case GSFlushReason::SAVESTATE:
|
||||
file << "SAVESTATE";
|
||||
break;
|
||||
case GSFlushReason::LOADSTATE:
|
||||
file << "LOAD SAVESTATE";
|
||||
break;
|
||||
case GSFlushReason::AUTOFLUSH:
|
||||
file << "AUTOFLUSH OVERLAP DETECTED";
|
||||
break;
|
||||
case GSFlushReason::VSYNC:
|
||||
file << "VSYNC";
|
||||
break;
|
||||
case GSFlushReason::GSREOPEN:
|
||||
file << "GS REOPEN";
|
||||
break;
|
||||
case GSFlushReason::UNKNOWN:
|
||||
default:
|
||||
file << "UNKNOWN";
|
||||
break;
|
||||
}
|
||||
file << "FLUSH REASON: " << GetFlushReasonString(m_state_flush_reason);
|
||||
|
||||
if (m_state_flush_reason != GSFlushReason::CONTEXTCHANGE && m_dirty_gs_regs)
|
||||
file << " AND POSSIBLE CONTEXT CHANGE";
|
||||
|
||||
@@ -870,6 +870,9 @@ public:
|
||||
/// Expands dither matrix, suitable for software renderer.
|
||||
static void ExpandDIMX(GSVector4i* dimx, const GIFRegDIMX DIMX);
|
||||
|
||||
/// Returns a string representing the flush reason.
|
||||
static const char* GetFlushReasonString(GSFlushReason reason);
|
||||
|
||||
void ResetHandlers();
|
||||
void ResetPCRTC();
|
||||
|
||||
|
||||
@@ -216,12 +216,6 @@ bool GSDevice::AcquireWindow(bool recreate_window)
|
||||
return true;
|
||||
}
|
||||
|
||||
void GSDevice::ReleaseWindow()
|
||||
{
|
||||
Host::ReleaseRenderWindow();
|
||||
m_window_info = WindowInfo();
|
||||
}
|
||||
|
||||
bool GSDevice::GetHostRefreshRate(float* refresh_rate)
|
||||
{
|
||||
if (m_window_info.surface_refresh_rate > 0.0f)
|
||||
|
||||
@@ -264,6 +264,7 @@ struct alignas(16) GSHWDrawConfig
|
||||
/// Returns true if the fixed index buffer should be used.
|
||||
__fi bool UseExpandIndexBuffer() const { return (expand == VSExpand::Point || expand == VSExpand::Sprite); }
|
||||
};
|
||||
static_assert(sizeof(VSSelector) == 1, "VSSelector is a single byte");
|
||||
#pragma pack(pop)
|
||||
#pragma pack(push, 4)
|
||||
struct PSSelector
|
||||
@@ -382,6 +383,7 @@ struct alignas(16) GSHWDrawConfig
|
||||
no_color = no_color1 = 1;
|
||||
}
|
||||
};
|
||||
static_assert(sizeof(PSSelector) == 12, "PSSelector is 12 bytes");
|
||||
#pragma pack(pop)
|
||||
struct PSSelectorHash
|
||||
{
|
||||
@@ -780,7 +782,6 @@ protected:
|
||||
FeatureSupport m_features;
|
||||
|
||||
bool AcquireWindow(bool recreate_window);
|
||||
void ReleaseWindow();
|
||||
|
||||
virtual GSTexture* CreateSurface(GSTexture::Type type, int width, int height, int levels, GSTexture::Format format) = 0;
|
||||
GSTexture* FetchSurface(GSTexture::Type type, int width, int height, int levels, GSTexture::Format format, bool clear, bool prefer_reuse);
|
||||
|
||||
@@ -1,20 +0,0 @@
|
||||
<?xml version="1.0" encoding="utf-8"?>
|
||||
<AutoVisualizer xmlns="http://schemas.microsoft.com/vstudio/debugger/natvis/2010">
|
||||
<Type Name="FastList<*>">
|
||||
<DisplayString>{{ size={m_free_indexes_stack_top} }}</DisplayString>
|
||||
<Expand>
|
||||
<Item Name="[size]" ExcludeView="simple">m_free_indexes_stack_top</Item>
|
||||
<Item Name="[capacity]" ExcludeView="simple">m_capacity - 1</Item>
|
||||
|
||||
<CustomListItems MaxItemsPerView="5000" ExcludeView="simple">
|
||||
<Variable Name="index" InitialValue="m_buffer[0].next_index" />
|
||||
|
||||
<Size>m_free_indexes_stack_top</Size>
|
||||
<Loop>
|
||||
<Item>m_buffer[index].data,na</Item>
|
||||
<Exec>index = m_buffer[index].next_index</Exec>
|
||||
</Loop>
|
||||
</CustomListItems>
|
||||
</Expand>
|
||||
</Type>
|
||||
</AutoVisualizer>
|
||||
@@ -493,7 +493,6 @@ void GSDevice11::Destroy()
|
||||
{
|
||||
GSDevice::Destroy();
|
||||
DestroySwapChain();
|
||||
ReleaseWindow();
|
||||
DestroyTimestampQueries();
|
||||
|
||||
m_convert = {};
|
||||
@@ -769,7 +768,6 @@ bool GSDevice11::UpdateWindow()
|
||||
if (m_window_info.type != WindowInfo::Type::Surfaceless && !CreateSwapChain())
|
||||
{
|
||||
Console.WriteLn("Failed to create swap chain on updated window");
|
||||
ReleaseWindow();
|
||||
return false;
|
||||
}
|
||||
|
||||
|
||||
@@ -218,7 +218,6 @@ void GSDevice12::Destroy()
|
||||
ExecuteCommandList(true);
|
||||
DestroyResources();
|
||||
DestroySwapChain();
|
||||
ReleaseWindow();
|
||||
g_d3d12_context->Destroy();
|
||||
}
|
||||
}
|
||||
@@ -432,7 +431,6 @@ bool GSDevice12::UpdateWindow()
|
||||
if (m_window_info.type != WindowInfo::Type::Surfaceless && !CreateSwapChain())
|
||||
{
|
||||
Console.WriteLn("Failed to create swap chain on updated window");
|
||||
ReleaseWindow();
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -3217,7 +3215,6 @@ void GSDevice12::RenderHW(GSHWDrawConfig& config)
|
||||
{
|
||||
draw_rt = m_current_render_target;
|
||||
m_pipeline_selector.rt = true;
|
||||
m_pipeline_selector.cms.wrgba = 0;
|
||||
}
|
||||
}
|
||||
else if (!draw_ds && m_current_depth_target && config.tex != m_current_depth_target &&
|
||||
@@ -3225,8 +3222,6 @@ void GSDevice12::RenderHW(GSHWDrawConfig& config)
|
||||
{
|
||||
draw_ds = m_current_depth_target;
|
||||
m_pipeline_selector.ds = true;
|
||||
m_pipeline_selector.dss.ztst = ZTST_ALWAYS;
|
||||
m_pipeline_selector.dss.zwe = false;
|
||||
}
|
||||
|
||||
OMSetRenderTargets(draw_rt, draw_ds, config.scissor);
|
||||
|
||||
@@ -51,15 +51,16 @@ public:
|
||||
u32 key;
|
||||
};
|
||||
|
||||
GSHWDrawConfig::BlendState bs;
|
||||
GSHWDrawConfig::VSSelector vs;
|
||||
GSHWDrawConfig::DepthStencilSelector dss;
|
||||
GSHWDrawConfig::ColorMaskSelector cms;
|
||||
GSHWDrawConfig::BlendState bs;
|
||||
u8 pad;
|
||||
|
||||
__fi bool operator==(const PipelineSelector& p) const { return (memcmp(this, &p, sizeof(p)) == 0); }
|
||||
__fi bool operator!=(const PipelineSelector& p) const { return (memcmp(this, &p, sizeof(p)) != 0); }
|
||||
__fi bool operator==(const PipelineSelector& p) const { return BitEqual(*this, p); }
|
||||
__fi bool operator!=(const PipelineSelector& p) const { return !BitEqual(*this, p); }
|
||||
|
||||
__fi PipelineSelector() { memset(this, 0, sizeof(*this)); }
|
||||
__fi PipelineSelector() { std::memset(this, 0, sizeof(*this)); }
|
||||
};
|
||||
static_assert(sizeof(PipelineSelector) == 24, "Pipeline selector is 24 bytes");
|
||||
|
||||
|
||||
@@ -1111,7 +1111,7 @@ bool GSHwHack::OI_HauntingGround(GSRendererHW& r, GSTexture* rt, GSTexture* ds,
|
||||
{
|
||||
// Haunting Ground clears two targets by doing a 256x448 direct colour write at 0x3000, covering a target at 0x3380.
|
||||
// This currently isn't handled in our HLE clears, so we need to manually remove the other target.
|
||||
if (rt && !ds && !t && r.IsConstantDirectWriteMemClear(true))
|
||||
if (rt && !ds && !t && r.IsConstantDirectWriteMemClear())
|
||||
{
|
||||
GL_CACHE("GSHwHack::OI_HauntingGround()");
|
||||
g_texture_cache->InvalidateVideoMemTargets(GSTextureCache::RenderTarget, RFRAME.Block(), RFRAME.FBW, RFRAME.PSM, r.m_r);
|
||||
|
||||
@@ -621,7 +621,7 @@ GSVector4 GSRendererHW::RealignTargetTextureCoordinate(const GSTextureCache::Sou
|
||||
GSVector4i GSRendererHW::ComputeBoundingBox(const GSVector2i& rtsize, float rtscale)
|
||||
{
|
||||
const GSVector4 offset = GSVector4(-1.0f, 1.0f); // Round value
|
||||
const GSVector4 box = m_vt.m_min.p.xyxy(m_vt.m_max.p) + offset.xxyy();
|
||||
const GSVector4 box = m_vt.m_min.p.upld(m_vt.m_max.p) + offset.xxyy();
|
||||
return GSVector4i(box * GSVector4(rtscale)).rintersect(GSVector4i(0, 0, rtsize.x, rtsize.y));
|
||||
}
|
||||
|
||||
@@ -653,9 +653,9 @@ void GSRendererHW::MergeSprite(GSTextureCache::Source* tex)
|
||||
}
|
||||
|
||||
#if 0
|
||||
GSVector4 delta_p = m_vt.m_max.p - m_vt.m_min.p;
|
||||
GSVector4 delta_t = m_vt.m_max.t - m_vt.m_min.t;
|
||||
bool is_blit = PrimitiveOverlap() == PRIM_OVERLAP_NO;
|
||||
const GSVector4 delta_p = m_vt.m_max.p - m_vt.m_min.p;
|
||||
const GSVector4 delta_t = m_vt.m_max.t - m_vt.m_min.t;
|
||||
const bool is_blit = PrimitiveOverlap() == PRIM_OVERLAP_NO;
|
||||
GL_INS("PP SAMPLER: Dp %f %f Dt %f %f. Is blit %d, is paving %d, count %d", delta_p.x, delta_p.y, delta_t.x, delta_t.y, is_blit, is_paving, m_vertex.tail);
|
||||
#endif
|
||||
|
||||
@@ -767,7 +767,7 @@ bool GSRendererHW::IsPossibleChannelShuffle() const
|
||||
// WRC 4 does channel shuffles in vertical strips. So check for page alignment.
|
||||
// Texture TBW should also be twice the framebuffer FBW, because the page is twice as wide.
|
||||
if (m_cached_ctx.TEX0.TBW == (m_cached_ctx.FRAME.FBW * 2) &&
|
||||
GSLocalMemory::IsPageAligned(m_cached_ctx.FRAME.PSM, GSVector4i(m_vt.m_min.p.xyxy(m_vt.m_max.p))))
|
||||
GSLocalMemory::IsPageAligned(m_cached_ctx.FRAME.PSM, GSVector4i(m_vt.m_min.p.upld(m_vt.m_max.p))))
|
||||
{
|
||||
return true;
|
||||
}
|
||||
@@ -802,11 +802,11 @@ bool GSRendererHW::IsSplitTextureShuffle()
|
||||
// For texture shuffles, the U will be offset by 8.
|
||||
const GSLocalMemory::psm_t& frame_psm = GSLocalMemory::m_psm[m_cached_ctx.FRAME.PSM];
|
||||
|
||||
const GSVector4i pos_rc = GSVector4i(m_vt.m_min.p.upld(m_vt.m_max.p));
|
||||
const GSVector4i pos_rc = GSVector4i(m_vt.m_min.p.upld(m_vt.m_max.p + GSVector4::cxpr(0.5f)));
|
||||
const GSVector4i tex_rc = GSVector4i(m_vt.m_min.t.upld(m_vt.m_max.t));
|
||||
|
||||
// Width/height should match.
|
||||
if (pos_rc.width() != tex_rc.width() || pos_rc.height() != tex_rc.height())
|
||||
if (std::abs(pos_rc.width() - tex_rc.width()) > 8 || pos_rc.height() != tex_rc.height())
|
||||
return false;
|
||||
|
||||
// X might be offset by up to -8/+8, but either the position or UV should be aligned.
|
||||
@@ -865,7 +865,7 @@ bool GSRendererHW::IsSplitTextureShuffle()
|
||||
GSVector4i GSRendererHW::GetSplitTextureShuffleDrawRect() const
|
||||
{
|
||||
const GSLocalMemory::psm_t& frame_psm = GSLocalMemory::m_psm[m_cached_ctx.FRAME.PSM];
|
||||
GSVector4i r = GSVector4i(m_vt.m_min.p.xyxy(m_vt.m_max.p)).rintersect(GSVector4i(m_context->scissor.in));
|
||||
GSVector4i r = GSVector4i(m_vt.m_min.p.upld(m_vt.m_max.p + GSVector4::cxpr(0.5f))).rintersect(GSVector4i(m_context->scissor.in));
|
||||
|
||||
// Some games (e.g. Crash Twinsanity) adjust both FBP and TBP0, so the rectangle will be half the size
|
||||
// of the actual shuffle. Others leave the FBP alone, but only adjust TBP0, and offset the draw rectangle
|
||||
@@ -1433,6 +1433,9 @@ void GSRendererHW::Draw()
|
||||
}
|
||||
|
||||
GL_PUSH("HW Draw %d (Context %u)", s_n, PRIM->CTXT);
|
||||
GL_INS("FLUSH REASON: %s%s", GetFlushReasonString(m_state_flush_reason),
|
||||
(m_state_flush_reason != GSFlushReason::CONTEXTCHANGE && m_dirty_gs_regs) ? " AND POSSIBLE CONTEXT CHANGE" :
|
||||
"");
|
||||
|
||||
// When the format is 24bit (Z or C), DATE ceases to function.
|
||||
// It was believed that in 24bit mode all pixels pass because alpha doesn't exist
|
||||
@@ -1570,7 +1573,7 @@ void GSRendererHW::Draw()
|
||||
}
|
||||
|
||||
// The rectangle of the draw rounded up.
|
||||
const GSVector4 rect = m_vt.m_min.p.xyxy(m_vt.m_max.p) + GSVector4(0.0f, 0.0f, 0.5f, 0.5f);
|
||||
const GSVector4 rect = m_vt.m_min.p.upld(m_vt.m_max.p + GSVector4::cxpr(0.5f));
|
||||
m_r = GSVector4i(rect).rintersect(GSVector4i(context->scissor.in));
|
||||
|
||||
if (!m_channel_shuffle && m_cached_ctx.FRAME.Block() == m_cached_ctx.TEX0.TBP0 &&
|
||||
@@ -1623,9 +1626,11 @@ void GSRendererHW::Draw()
|
||||
cleanup_draw();
|
||||
};
|
||||
|
||||
const bool is_possible_mem_clear = IsConstantDirectWriteMemClear();
|
||||
|
||||
if (!GSConfig.UserHacks_DisableSafeFeatures)
|
||||
{
|
||||
if (IsConstantDirectWriteMemClear(true))
|
||||
if (is_possible_mem_clear)
|
||||
{
|
||||
// Likely doing a huge single page width clear, which never goes well. (Superman)
|
||||
// Burnout 3 does a 32x1024 double width clear on its reflection targets.
|
||||
@@ -1662,6 +1667,8 @@ void GSRendererHW::Draw()
|
||||
|
||||
if (is_zero_clear && OI_GsMemClear() && clear_height_valid)
|
||||
{
|
||||
GL_INS("Clear draw with mem clear and valid clear height, invalidating.");
|
||||
|
||||
g_texture_cache->InvalidateVideoMem(context->offset.fb, m_r, false, true);
|
||||
g_texture_cache->InvalidateVideoMemType(GSTextureCache::RenderTarget, m_cached_ctx.FRAME.Block());
|
||||
|
||||
@@ -1830,15 +1837,15 @@ void GSRendererHW::Draw()
|
||||
// Normally we would use 1024 here to match the clear above, but The Godfather does a 1023x1023 draw instead
|
||||
// (very close to 1024x1024, but apparently the GS rounds down..). So, catch that here, we don't want to
|
||||
// create that target, because the clear isn't black, it'll hang around and never get invalidated.
|
||||
const bool is_square = (t_size.y == t_size.x) && m_r.w >= 1023 &&
|
||||
((m_index.tail == 2 && m_vt.m_primclass == GS_SPRITE_CLASS) || (m_index.tail == 6 && m_vt.m_primclass == GS_TRIANGLE_CLASS));
|
||||
const bool is_clear = IsConstantDirectWriteMemClear(false) && is_square;
|
||||
const bool is_square = (t_size.y == t_size.x) && m_r.w >= 1023 && PrimitiveCoversWithoutGaps();
|
||||
const bool is_clear = is_possible_mem_clear && is_square;
|
||||
rt = g_texture_cache->LookupTarget(FRAME_TEX0, t_size, target_scale, GSTextureCache::RenderTarget, true,
|
||||
fm, false, is_clear, force_preload);
|
||||
|
||||
// Draw skipped because it was a clear and there was no target.
|
||||
if (!rt)
|
||||
{
|
||||
GL_INS("Clear draw with no target, skipping.");
|
||||
cleanup_cancelled_draw();
|
||||
OI_GsMemClear();
|
||||
return;
|
||||
@@ -2025,8 +2032,7 @@ void GSRendererHW::Draw()
|
||||
|
||||
// Deferred update of TEX0. We don't want to change it when we're doing a shuffle/clear, because it
|
||||
// may increase the buffer width, or change PSM, which breaks P8 conversion amongst other things.
|
||||
const bool is_mem_clear = IsConstantDirectWriteMemClear(false);
|
||||
const bool can_update_size = !is_mem_clear && !m_texture_shuffle && !m_channel_shuffle;
|
||||
const bool can_update_size = !is_possible_mem_clear && !m_texture_shuffle && !m_channel_shuffle;
|
||||
if (!m_texture_shuffle && !m_channel_shuffle)
|
||||
{
|
||||
if (rt)
|
||||
@@ -2203,13 +2209,8 @@ void GSRendererHW::Draw()
|
||||
return;
|
||||
}
|
||||
|
||||
if (!GSConfig.UserHacks_DisableSafeFeatures)
|
||||
{
|
||||
if (IsConstantDirectWriteMemClear(true) && IsBlendedOrOpaque())
|
||||
{
|
||||
OI_DoubleHalfClear(rt, ds);
|
||||
}
|
||||
}
|
||||
if (!GSConfig.UserHacks_DisableSafeFeatures && is_possible_mem_clear && IsBlendedOrOpaque())
|
||||
OI_DoubleHalfClear(rt, ds);
|
||||
|
||||
// A couple of hack to avoid upscaling issue. So far it seems to impacts mostly sprite
|
||||
// Note: first hack corrects both position and texture coordinate
|
||||
@@ -2511,9 +2512,9 @@ void GSRendererHW::SetupIA(float target_scale, float sx, float sy)
|
||||
m_conf.nindices = m_index.tail;
|
||||
}
|
||||
|
||||
void GSRendererHW::EmulateZbuffer()
|
||||
void GSRendererHW::EmulateZbuffer(const GSTextureCache::Target* ds)
|
||||
{
|
||||
if (m_cached_ctx.TEST.ZTE)
|
||||
if (ds && m_cached_ctx.TEST.ZTE)
|
||||
{
|
||||
m_conf.depth.ztst = m_cached_ctx.TEST.ZTST;
|
||||
// AA1: Z is not written on lines since coverage is always less than 0x80.
|
||||
@@ -3314,7 +3315,7 @@ void GSRendererHW::EmulateBlending(bool& DATE_PRIMID, bool& DATE_BARRIER, bool&
|
||||
|
||||
// For stat to optimize accurate option
|
||||
#if 0
|
||||
GL_INS("BLEND_INFO: %u/%u/%u/%u. Clamp:%u. Prim:%d number %u (drawlist %u) (sw %d)",
|
||||
GL_INS("BLEND_INFO: %u/%u/%u/%u. Clamp:%u. Prim:%d number %u (drawlist %zu) (sw %d)",
|
||||
m_conf.ps.blend_a, m_conf.ps.blend_b, m_conf.ps.blend_c, m_conf.ps.blend_d,
|
||||
m_env.COLCLAMP.CLAMP, m_vt.m_primclass, m_vertex.next, m_drawlist.size(), sw_blending);
|
||||
#endif
|
||||
@@ -4090,7 +4091,7 @@ bool GSRendererHW::CanUseTexIsFB(const GSTextureCache::Target* rt, const GSTextu
|
||||
|
||||
// Make sure that we're not sampling away from the area we're rendering.
|
||||
// We need to take the absolute here, because Beyond Good and Evil undithers itself using a -1,-1 offset.
|
||||
const GSVector4 diff(m_vt.m_min.p.xyxy(m_vt.m_max.p) - m_vt.m_min.t.xyxy(m_vt.m_max.t));
|
||||
const GSVector4 diff(m_vt.m_min.p.upld(m_vt.m_max.p) - m_vt.m_min.t.upld(m_vt.m_max.t));
|
||||
GL_CACHE("Coord diff: %f,%f", diff.x, diff.y);
|
||||
if ((diff.abs() < GSVector4(1.0f)).alltrue())
|
||||
{
|
||||
@@ -4160,8 +4161,8 @@ void GSRendererHW::ResetStates()
|
||||
__ri void GSRendererHW::DrawPrims(GSTextureCache::Target* rt, GSTextureCache::Target* ds, GSTextureCache::Source* tex, const TextureMinMaxResult& tmm)
|
||||
{
|
||||
#ifdef ENABLE_OGL_DEBUG
|
||||
const GSVector4i area_out = GSVector4i(m_vt.m_min.p.xyxy(m_vt.m_max.p)).rintersect(GSVector4i(m_context->scissor.in));
|
||||
const GSVector4i area_in = GSVector4i(m_vt.m_min.t.xyxy(m_vt.m_max.t));
|
||||
const GSVector4i area_out = GSVector4i(m_vt.m_min.p.upld(m_vt.m_max.p)).rintersect(GSVector4i(m_context->scissor.in));
|
||||
const GSVector4i area_in = GSVector4i(m_vt.m_min.t.upld(m_vt.m_max.t));
|
||||
|
||||
GL_PUSH("GL Draw from (area %d,%d => %d,%d) in (area %d,%d => %d,%d)",
|
||||
area_in.x, area_in.y, area_in.z, area_in.w,
|
||||
@@ -4187,7 +4188,7 @@ __ri void GSRendererHW::DrawPrims(GSTextureCache::Target* rt, GSTextureCache::Ta
|
||||
m_conf.ds = ds ? ds->m_texture : nullptr;
|
||||
|
||||
// Z setup has to come before channel shuffle
|
||||
EmulateZbuffer();
|
||||
EmulateZbuffer(ds);
|
||||
|
||||
// HLE implementation of the channel selection effect
|
||||
//
|
||||
@@ -4319,7 +4320,10 @@ __ri void GSRendererHW::DrawPrims(GSTextureCache::Target* rt, GSTextureCache::Ta
|
||||
// No point outputting colours if we're just writing depth.
|
||||
// We might still need the framebuffer for DATE, though.
|
||||
if (!rt || m_conf.colormask.wrgba == 0)
|
||||
{
|
||||
m_conf.ps.DisableColorOutput();
|
||||
m_conf.colormask.wrgba = 0;
|
||||
}
|
||||
|
||||
if (m_conf.ps.scanmsk & 2)
|
||||
DATE_PRIMID = false; // to have discard in the shader work correctly
|
||||
@@ -5053,10 +5057,10 @@ bool GSRendererHW::OI_GsMemClear()
|
||||
&& (m_cached_ctx.FRAME.PSM & 0xF) == (m_cached_ctx.ZBUF.PSM & 0xF) && m_vt.m_eq.z == 1 && m_vertex.buff[1].XYZ.Z == m_vertex.buff[1].RGBAQ.U32[0];
|
||||
|
||||
// Limit it further to a full screen 0 write
|
||||
if (((m_vertex.next == 2) || ZisFrame) && m_vt.m_eq.rgba == 0xFFFF)
|
||||
if ((PrimitiveCoversWithoutGaps() || ZisFrame) && m_vt.m_eq.rgba == 0xFFFF)
|
||||
{
|
||||
const GSOffset& off = m_context->offset.fb;
|
||||
GSVector4i r = GSVector4i(m_vt.m_min.p.xyxy(m_vt.m_max.p)).rintersect(GSVector4i(m_context->scissor.in));
|
||||
GSVector4i r = GSVector4i(m_vt.m_min.p.upld(m_vt.m_max.p)).rintersect(GSVector4i(m_context->scissor.in));
|
||||
|
||||
if (r.width() == 32 && ZisFrame)
|
||||
r.z += 32;
|
||||
@@ -5192,13 +5196,43 @@ bool GSRendererHW::IsBlendedOrOpaque()
|
||||
return (!PRIM->ABE || IsOpaque() || m_context->ALPHA.IsCdOutput());
|
||||
}
|
||||
|
||||
bool GSRendererHW::IsConstantDirectWriteMemClear(bool include_zero)
|
||||
bool GSRendererHW::PrimitiveCoversWithoutGaps() const
|
||||
{
|
||||
// Draw shouldn't be offset.
|
||||
if ((m_r.eq32(GSVector4i::zero())).mask() & 0xff00)
|
||||
return false;
|
||||
|
||||
// This is potentially wrong for fans/strips...
|
||||
if (m_vt.m_primclass == GS_TRIANGLE_CLASS)
|
||||
return (m_index.tail == 6);
|
||||
else if (m_vt.m_primclass != GS_SPRITE_CLASS)
|
||||
return false;
|
||||
|
||||
// Simple case: one sprite.
|
||||
if (m_index.tail == 2)
|
||||
return true;
|
||||
|
||||
// Borrowed from MergeSprite().
|
||||
const GSVertex* v = &m_vertex.buff[0];
|
||||
const int first_dpX = v[1].XYZ.X - v[0].XYZ.X;
|
||||
for (u32 i = 0; i < m_vertex.next; i += 2)
|
||||
{
|
||||
const int dpX = v[i + 1].XYZ.X - v[i].XYZ.X;
|
||||
if (dpX != first_dpX)
|
||||
return false;
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
bool GSRendererHW::IsConstantDirectWriteMemClear()
|
||||
{
|
||||
const bool direct_draw = (m_vt.m_primclass == GS_SPRITE_CLASS) || (m_index.tail == 6 && m_vt.m_primclass == GS_TRIANGLE_CLASS);
|
||||
// Constant Direct Write without texture/test/blending (aka a GS mem clear)
|
||||
if (direct_draw && !PRIM->TME // Direct write
|
||||
&& !(m_draw_env->SCANMSK.MSK & 2)
|
||||
&& !m_cached_ctx.TEST.ATE // no alpha test
|
||||
&& !m_cached_ctx.TEST.DATE // no destination alpha test
|
||||
&& (!m_cached_ctx.TEST.ZTE || m_cached_ctx.TEST.ZTST == ZTST_ALWAYS) // no depth test
|
||||
&& (m_vt.m_eq.rgba == 0xFFFF || m_vertex.next == 2) // constant color write
|
||||
&& m_r.x == 0 && m_r.y == 0) // Likely full buffer write
|
||||
|
||||
@@ -52,8 +52,9 @@ private:
|
||||
float alpha1(int L, int X0, int X1);
|
||||
void SwSpriteRender();
|
||||
bool CanUseSwSpriteRender();
|
||||
bool IsConstantDirectWriteMemClear(bool include_zero);
|
||||
bool IsConstantDirectWriteMemClear();
|
||||
bool IsBlendedOrOpaque();
|
||||
bool PrimitiveCoversWithoutGaps() const;
|
||||
|
||||
enum class CLUTDrawTestResult
|
||||
{
|
||||
@@ -86,7 +87,7 @@ private:
|
||||
bool& target_region, GSVector2i& unscaled_size, float& scale, GSTexture*& src_copy);
|
||||
bool CanUseTexIsFB(const GSTextureCache::Target* rt, const GSTextureCache::Source* tex) const;
|
||||
|
||||
void EmulateZbuffer();
|
||||
void EmulateZbuffer(const GSTextureCache::Target* ds);
|
||||
void EmulateATST(float& AREF, GSHWDrawConfig::PSSelector& ps, bool pass_2);
|
||||
|
||||
void SetTCOffset();
|
||||
|
||||
@@ -550,7 +550,10 @@ GSTextureCache::Source* GSTextureCache::LookupDepthSource(const GIFRegTEX0& TEX0
|
||||
|
||||
const SourceRegion region = SourceRegion::Create(TEX0, CLAMP);
|
||||
const GSLocalMemory::psm_t& psm_s = GSLocalMemory::m_psm[TEX0.PSM];
|
||||
Source* src = FindSourceInMap(TEX0, TEXA, psm_s, nullptr, nullptr, GSVector2i(0, 0), region,
|
||||
// Yes, this can get called with color PSMs that have palettes
|
||||
const u32* const clut = g_gs_renderer->m_mem.m_clut;
|
||||
GSTexture* const gpu_clut = (psm_s.pal > 0) ? g_gs_renderer->m_mem.m_clut.GetGPUTexture() : nullptr;
|
||||
Source* src = FindSourceInMap(TEX0, TEXA, psm_s, clut, gpu_clut, GSVector2i(0, 0), region,
|
||||
region.IsFixedTEX0(TEX0), m_src.m_map[TEX0.TBP0 >> 5]);
|
||||
if (src)
|
||||
{
|
||||
@@ -2436,15 +2439,17 @@ bool GSTextureCache::Move(u32 SBP, u32 SBW, u32 SPSM, int sx, int sy, u32 DBP, u
|
||||
|
||||
// DX11/12 is a bit lame and can't partial copy depth targets. We could do this with a blit instead,
|
||||
// but so far haven't seen anything which needs it.
|
||||
const GSLocalMemory::psm_t& spsm_s = GSLocalMemory::m_psm[SPSM];
|
||||
const GSLocalMemory::psm_t& dpsm_s = GSLocalMemory::m_psm[DPSM];
|
||||
if (GSConfig.Renderer == GSRendererType::DX11 || GSConfig.Renderer == GSRendererType::DX12)
|
||||
{
|
||||
if (GSLocalMemory::m_psm[SPSM].depth || GSLocalMemory::m_psm[DPSM].depth)
|
||||
if (spsm_s.depth || dpsm_s.depth)
|
||||
return false;
|
||||
}
|
||||
|
||||
// Look for an exact match on the targets.
|
||||
GSTextureCache::Target* src = GetExactTarget(SBP, SBW, SPSM);
|
||||
GSTextureCache::Target* dst = GetExactTarget(DBP, DBW, DPSM);
|
||||
GSTextureCache::Target* src = GetExactTarget(SBP, SBW, spsm_s.depth ? DepthStencil : RenderTarget);
|
||||
GSTextureCache::Target* dst = GetExactTarget(DBP, DBW, dpsm_s.depth ? DepthStencil : RenderTarget);
|
||||
|
||||
// Beware of the case where a game might create a larger texture by moving a bunch of chunks around.
|
||||
// We use dx/dy == 0 and the TBW check as a safeguard to make sure these go through to local memory.
|
||||
@@ -2517,7 +2522,8 @@ bool GSTextureCache::Move(u32 SBP, u32 SBW, u32 SPSM, int sx, int sy, u32 DBP, u
|
||||
// Make sure the copy doesn't go out of bounds (it shouldn't).
|
||||
if ((scaled_dx + scaled_w) > dst->m_texture->GetWidth() || (scaled_dy + scaled_h) > dst->m_texture->GetHeight())
|
||||
return false;
|
||||
GL_CACHE("HW Move 0x%x to 0x%x <%d,%d->%d,%d> -> <%d,%d->%d,%d>", SBP, DBP, sx, sy, sx + w, sy, h, dx, dy, dx + w, dy + h);
|
||||
GL_CACHE("HW Move 0x%x[BW:%u PSM:%s] to 0x%x[BW:%u PSM:%s] <%d,%d->%d,%d> -> <%d,%d->%d,%d>", SBP, SBW,
|
||||
psm_str(SPSM), DBP, DBW, psm_str(DPSM), sx, sy, sx + w, sy + h, dx, dy, dx + w, dy + h);
|
||||
g_gs_device->CopyRect(src->m_texture, dst->m_texture,
|
||||
GSVector4i(scaled_sx, scaled_sy, scaled_sx + scaled_w, scaled_sy + scaled_h),
|
||||
scaled_dx, scaled_dy);
|
||||
@@ -2652,14 +2658,17 @@ bool GSTextureCache::ShuffleMove(u32 BP, u32 BW, u32 PSM, int sx, int sy, int dx
|
||||
return true;
|
||||
}
|
||||
|
||||
GSTextureCache::Target* GSTextureCache::GetExactTarget(u32 BP, u32 BW, u32 PSM) const
|
||||
GSTextureCache::Target* GSTextureCache::GetExactTarget(u32 BP, u32 BW, int type)
|
||||
{
|
||||
auto& rts = m_dst[GSLocalMemory::m_psm[PSM].depth ? DepthStencil : RenderTarget];
|
||||
auto& rts = m_dst[type];
|
||||
for (auto it = rts.begin(); it != rts.end(); ++it) // Iterate targets from MRU to LRU.
|
||||
{
|
||||
Target* t = *it;
|
||||
if (t->m_TEX0.TBP0 == BP && t->m_TEX0.TBW == BW && t->m_TEX0.PSM == PSM)
|
||||
if (t->m_TEX0.TBP0 == BP && t->m_TEX0.TBW == BW)
|
||||
{
|
||||
rts.MoveFront(it.Index());
|
||||
return t;
|
||||
}
|
||||
}
|
||||
|
||||
return nullptr;
|
||||
|
||||
@@ -444,8 +444,8 @@ public:
|
||||
bool is_frame = false, bool is_clear = false, bool preload = GSConfig.PreloadFrameWithGSData);
|
||||
Target* LookupDisplayTarget(GIFRegTEX0 TEX0, const GSVector2i& size, float scale);
|
||||
|
||||
/// Looks up a target in the cache, and only returns it if the BP/BW/PSM match exactly.
|
||||
Target* GetExactTarget(u32 BP, u32 BW, u32 PSM) const;
|
||||
/// Looks up a target in the cache, and only returns it if the BP/BW match exactly.
|
||||
Target* GetExactTarget(u32 BP, u32 BW, int type);
|
||||
Target* GetTargetWithSharedBits(u32 BP, u32 PSM) const;
|
||||
|
||||
GSVector2i GetTargetSize(u32 bp, u32 fbw, u32 psm, s32 min_width, s32 min_height);
|
||||
|
||||
@@ -1,7 +0,0 @@
|
||||
<?xml version="1.0" encoding="utf-8"?>
|
||||
<AutoVisualizer xmlns="http://schemas.microsoft.com/vstudio/debugger/natvis/2010">
|
||||
<Type Name="GSTextureCache::Target">
|
||||
<DisplayString Condition="m_type == 0">{{ RT @ BP={m_TEX0.TBP0,X}-{m_end_block,X} BW={m_TEX0.TBW} PSM={m_TEX0.PSM,X} {m_unscaled_size.x}x{m_unscaled_size.y} {m_valid.z},{m_valid.w} }}</DisplayString>
|
||||
<DisplayString Condition="m_type == 1">{{ Depth @ BP={m_TEX0.TBP0,X}-{m_end_block,X} BW={m_TEX0.TBW} PSM={m_TEX0.PSM,X} {m_unscaled_size.x}x{m_unscaled_size.y} {m_valid.z},{m_valid.w} }}</DisplayString>
|
||||
</Type>
|
||||
</AutoVisualizer>
|
||||
@@ -252,10 +252,12 @@ void GSDeviceMTL::DrawCommandBufferFinished(u64 draw, id<MTLCommandBuffer> buffe
|
||||
|
||||
void GSDeviceMTL::FlushEncoders()
|
||||
{
|
||||
if (!m_current_render_cmdbuf)
|
||||
return;
|
||||
EndRenderPass();
|
||||
Sync(m_vertex_upload_buf);
|
||||
bool needs_submit = m_current_render_cmdbuf;
|
||||
if (needs_submit)
|
||||
{
|
||||
EndRenderPass();
|
||||
Sync(m_vertex_upload_buf);
|
||||
}
|
||||
if (m_dev.features.unified_memory)
|
||||
{
|
||||
ASSERT(!m_vertex_upload_cmdbuf && "Should never be used!");
|
||||
@@ -274,6 +276,8 @@ void GSDeviceMTL::FlushEncoders()
|
||||
m_texture_upload_encoder = nil;
|
||||
m_texture_upload_cmdbuf = nil;
|
||||
}
|
||||
if (!needs_submit)
|
||||
return;
|
||||
if (m_late_texture_upload_encoder)
|
||||
{
|
||||
[m_late_texture_upload_encoder endEncoding];
|
||||
@@ -1170,7 +1174,6 @@ void GSDeviceMTL::Destroy()
|
||||
|
||||
GSDevice::Destroy();
|
||||
GSDeviceMTL::DestroySurface();
|
||||
ReleaseWindow();
|
||||
m_queue = nullptr;
|
||||
m_dev.Reset();
|
||||
}}
|
||||
|
||||
@@ -551,8 +551,6 @@ void GSDeviceOGL::Destroy()
|
||||
|
||||
m_gl_context->DoneCurrent();
|
||||
m_gl_context.reset();
|
||||
|
||||
ReleaseWindow();
|
||||
}
|
||||
}
|
||||
|
||||
@@ -673,7 +671,6 @@ bool GSDeviceOGL::UpdateWindow()
|
||||
if (!m_gl_context->ChangeSurface(m_window_info))
|
||||
{
|
||||
Console.Error("Failed to change surface");
|
||||
ReleaseWindow();
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -2396,7 +2393,7 @@ void GSDeviceOGL::RenderHW(GSHWDrawConfig& config)
|
||||
psel.vs = config.vs;
|
||||
psel.ps.key_hi = config.ps.key_hi;
|
||||
psel.ps.key_lo = config.ps.key_lo;
|
||||
psel.pad = 0;
|
||||
std::memset(psel.pad, 0, sizeof(psel.pad));
|
||||
|
||||
SetupPipeline(psel);
|
||||
|
||||
@@ -2447,22 +2444,19 @@ void GSDeviceOGL::RenderHW(GSHWDrawConfig& config)
|
||||
// avoid changing framebuffer just to switch from rt+depth to rt and vice versa
|
||||
GSTexture* draw_rt = hdr_rt ? hdr_rt : config.rt;
|
||||
GSTexture* draw_ds = config.ds;
|
||||
OMColorMaskSelector draw_colormask = config.colormask;
|
||||
if (!draw_rt && GLState::rt && GLState::ds == draw_ds && GLState::rt->GetSize() == draw_ds->GetSize())
|
||||
if (!draw_rt && GLState::rt && GLState::ds == draw_ds && config.tex != GLState::rt &&
|
||||
GLState::rt->GetSize() == draw_ds->GetSize())
|
||||
{
|
||||
draw_rt = GLState::rt;
|
||||
draw_colormask.wrgba = 0;
|
||||
}
|
||||
else if (!draw_ds && GLState::ds && GLState::rt == draw_rt && GLState::ds->GetSize() == draw_rt->GetSize())
|
||||
else if (!draw_ds && GLState::ds && GLState::rt == draw_rt && config.tex != GLState::ds &&
|
||||
GLState::ds->GetSize() == draw_rt->GetSize())
|
||||
{
|
||||
// should already be always-pass.
|
||||
draw_ds = GLState::ds;
|
||||
config.depth.ztst = ZTST_ALWAYS;
|
||||
config.depth.zwe = false;
|
||||
}
|
||||
|
||||
OMSetRenderTargets(draw_rt, draw_ds, &config.scissor);
|
||||
OMSetColorMaskState(draw_colormask);
|
||||
OMSetColorMaskState(config.colormask);
|
||||
SetupOM(config.depth);
|
||||
|
||||
SendHWDraw(config, psel.ps.IsFeedbackLoop());
|
||||
|
||||
@@ -130,10 +130,10 @@ public:
|
||||
{
|
||||
PSSelector ps;
|
||||
VSSelector vs;
|
||||
u16 pad;
|
||||
u8 pad[3];
|
||||
|
||||
__fi bool operator==(const ProgramSelector& p) const { return (std::memcmp(this, &p, sizeof(*this)) == 0); }
|
||||
__fi bool operator!=(const ProgramSelector& p) const { return (std::memcmp(this, &p, sizeof(*this)) != 0); }
|
||||
__fi bool operator==(const ProgramSelector& p) const { return BitEqual(*this, p); }
|
||||
__fi bool operator!=(const ProgramSelector& p) const { return !BitEqual(*this, p); }
|
||||
};
|
||||
static_assert(sizeof(ProgramSelector) == 16, "Program selector is 16 bytes");
|
||||
|
||||
|
||||
@@ -17,6 +17,7 @@
|
||||
#include "GSDrawScanlineCodeGenerator.all.h"
|
||||
#include "GS/Renderers/Common/GSFunctionMap.h"
|
||||
#include "GSVertexSW.h"
|
||||
#include "common/Perf.h"
|
||||
|
||||
MULTI_ISA_UNSHARED_IMPL;
|
||||
using namespace Xbyak;
|
||||
@@ -590,6 +591,8 @@ L("exit");
|
||||
if (isYmm)
|
||||
vzeroupper();
|
||||
ret();
|
||||
|
||||
Perf::any.RegisterKey(actual.getCode(), actual.getSize(), "GSDrawScanline_", m_sel.key);
|
||||
}
|
||||
|
||||
/// Inputs: a0=pixels, a1=left, a2[x64]=top, a3[x64]=v
|
||||
|
||||
@@ -16,6 +16,7 @@
|
||||
#include "PrecompiledHeader.h"
|
||||
#include "GSSetupPrimCodeGenerator.all.h"
|
||||
#include "GSVertexSW.h"
|
||||
#include "common/Perf.h"
|
||||
|
||||
MULTI_ISA_UNSHARED_IMPL;
|
||||
using namespace Xbyak;
|
||||
@@ -147,6 +148,8 @@ void GSSetupPrimCodeGenerator2::Generate()
|
||||
if (isYmm)
|
||||
vzeroupper();
|
||||
ret();
|
||||
|
||||
Perf::any.RegisterKey(actual.getCode(), actual.getSize(), "GSSetupPrim_", m_sel.key);
|
||||
}
|
||||
|
||||
void GSSetupPrimCodeGenerator2::Depth_XMM()
|
||||
|
||||
@@ -86,6 +86,15 @@ GSDeviceVK::~GSDeviceVK()
|
||||
pxAssert(!g_vulkan_context);
|
||||
}
|
||||
|
||||
static void GPUListToAdapterNames(std::vector<std::string>* dest, VkInstance instance)
|
||||
{
|
||||
Vulkan::Context::GPUList gpus = Vulkan::Context::EnumerateGPUs(instance);
|
||||
dest->clear();
|
||||
dest->reserve(gpus.size());
|
||||
for (auto& [gpu, name] : gpus)
|
||||
dest->push_back(std::move(name));
|
||||
}
|
||||
|
||||
void GSDeviceVK::GetAdaptersAndFullscreenModes(
|
||||
std::vector<std::string>* adapters, std::vector<std::string>* fullscreen_modes)
|
||||
{
|
||||
@@ -94,7 +103,7 @@ void GSDeviceVK::GetAdaptersAndFullscreenModes(
|
||||
if (g_vulkan_context)
|
||||
{
|
||||
if (adapters)
|
||||
*adapters = Vulkan::Context::EnumerateGPUNames(g_vulkan_context->GetVulkanInstance());
|
||||
GPUListToAdapterNames(adapters, g_vulkan_context->GetVulkanInstance());
|
||||
|
||||
if (fullscreen_modes)
|
||||
{
|
||||
@@ -111,7 +120,7 @@ void GSDeviceVK::GetAdaptersAndFullscreenModes(
|
||||
if (instance != VK_NULL_HANDLE)
|
||||
{
|
||||
if (Vulkan::LoadVulkanInstanceFunctions(instance))
|
||||
*adapters = Vulkan::Context::EnumerateGPUNames(instance);
|
||||
GPUListToAdapterNames(adapters, instance);
|
||||
|
||||
vkDestroyInstance(instance, nullptr);
|
||||
}
|
||||
@@ -257,7 +266,6 @@ void GSDeviceVK::Destroy()
|
||||
|
||||
g_vulkan_context->WaitForGPUIdle();
|
||||
m_swap_chain.reset();
|
||||
ReleaseWindow();
|
||||
|
||||
Vulkan::ShaderCache::Destroy();
|
||||
Vulkan::Context::Destroy();
|
||||
@@ -295,7 +303,6 @@ bool GSDeviceVK::UpdateWindow()
|
||||
if (surface == VK_NULL_HANDLE)
|
||||
{
|
||||
Console.Error("Failed to create new surface for swap chain");
|
||||
ReleaseWindow();
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -304,7 +311,6 @@ bool GSDeviceVK::UpdateWindow()
|
||||
{
|
||||
Console.Error("Failed to create swap chain");
|
||||
Vulkan::SwapChain::DestroyVulkanSurface(g_vulkan_context->GetVulkanInstance(), &m_window_info, surface);
|
||||
ReleaseWindow();
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -572,8 +578,6 @@ bool GSDeviceVK::CreateDeviceAndSwapChain()
|
||||
if (!AcquireWindow(true))
|
||||
return false;
|
||||
|
||||
ScopedGuard window_cleanup = [this]() { ReleaseWindow(); };
|
||||
|
||||
VkInstance instance =
|
||||
Vulkan::Context::CreateVulkanInstance(m_window_info, enable_debug_utils, enable_validation_layer);
|
||||
if (instance == VK_NULL_HANDLE)
|
||||
@@ -611,26 +615,25 @@ bool GSDeviceVK::CreateDeviceAndSwapChain()
|
||||
}
|
||||
|
||||
u32 gpu_index = 0;
|
||||
Vulkan::Context::GPUNameList gpu_names = Vulkan::Context::EnumerateGPUNames(instance);
|
||||
if (!GSConfig.Adapter.empty())
|
||||
{
|
||||
for (; gpu_index < static_cast<u32>(gpu_names.size()); gpu_index++)
|
||||
for (; gpu_index < static_cast<u32>(gpus.size()); gpu_index++)
|
||||
{
|
||||
Console.WriteLn(fmt::format("GPU {}: {}", gpu_index, gpu_names[gpu_index]));
|
||||
if (gpu_names[gpu_index] == GSConfig.Adapter)
|
||||
Console.WriteLn(fmt::format("GPU {}: {}", gpu_index, gpus[gpu_index].second));
|
||||
if (gpus[gpu_index].second == GSConfig.Adapter)
|
||||
break;
|
||||
}
|
||||
|
||||
if (gpu_index == static_cast<u32>(gpu_names.size()))
|
||||
if (gpu_index == static_cast<u32>(gpus.size()))
|
||||
{
|
||||
Console.Warning(
|
||||
fmt::format("Requested GPU '{}' not found, using first ({})", GSConfig.Adapter, gpu_names[0]));
|
||||
fmt::format("Requested GPU '{}' not found, using first ({})", GSConfig.Adapter, gpus[0].second));
|
||||
gpu_index = 0;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
Console.WriteLn("No GPU requested, using first (%s)", gpu_names[0].c_str());
|
||||
Console.WriteLn(fmt::format("No GPU requested, using first ({})", gpus[0].second));
|
||||
}
|
||||
|
||||
VkSurfaceKHR surface = VK_NULL_HANDLE;
|
||||
@@ -640,13 +643,13 @@ bool GSDeviceVK::CreateDeviceAndSwapChain()
|
||||
};
|
||||
if (m_window_info.type != WindowInfo::Type::Surfaceless)
|
||||
{
|
||||
surface = Vulkan::SwapChain::CreateVulkanSurface(instance, gpus[gpu_index], &m_window_info);
|
||||
surface = Vulkan::SwapChain::CreateVulkanSurface(instance, gpus[gpu_index].first, &m_window_info);
|
||||
if (surface == VK_NULL_HANDLE)
|
||||
return false;
|
||||
}
|
||||
|
||||
if (!Vulkan::Context::Create(instance, surface, gpus[gpu_index], enable_debug_utils, enable_validation_layer,
|
||||
!GSConfig.DisableThreadedPresentation))
|
||||
if (!Vulkan::Context::Create(instance, surface, gpus[gpu_index].first, !GSConfig.DisableThreadedPresentation,
|
||||
enable_debug_utils, enable_validation_layer))
|
||||
{
|
||||
Console.Error("Failed to create Vulkan context");
|
||||
return false;
|
||||
@@ -668,7 +671,6 @@ bool GSDeviceVK::CreateDeviceAndSwapChain()
|
||||
|
||||
surface_cleanup.Cancel();
|
||||
instance_cleanup.Cancel();
|
||||
window_cleanup.Cancel();
|
||||
library_cleanup.Cancel();
|
||||
|
||||
// Render a frame as soon as possible to clear out whatever was previously being displayed.
|
||||
@@ -690,12 +692,14 @@ bool GSDeviceVK::CheckFeatures()
|
||||
m_features.framebuffer_fetch = g_vulkan_context->GetOptionalExtensions().vk_ext_rasterization_order_attachment_access && !GSConfig.DisableFramebufferFetch;
|
||||
m_features.texture_barrier = GSConfig.OverrideTextureBarriers != 0;
|
||||
m_features.broken_point_sampler = isAMD;
|
||||
// Usually, geometry shader indicates primid support
|
||||
// However on Metal (MoltenVK), geometry shader is never available, but primid sometimes is
|
||||
#ifdef __APPLE__
|
||||
// On Metal (MoltenVK), primid is sometimes available, but broken on some older GPUs and MacOS versions.
|
||||
// Officially, it's available on GPUs that support barycentric coordinates (Newer AMD and Apple)
|
||||
// Unofficially, it seems to work on older Intel GPUs (but breaks other things on newer Intel GPUs, see GSMTLDeviceInfo.mm for details)
|
||||
// We'll only enable for the officially supported GPUs here. We'll leave in the option of force-enabling it with OverrideGeometryShaders though.
|
||||
m_features.primitive_id = features.geometryShader || g_vulkan_context->GetOptionalExtensions().vk_khr_fragment_shader_barycentric;
|
||||
m_features.primitive_id = g_vulkan_context->GetOptionalExtensions().vk_khr_fragment_shader_barycentric;
|
||||
#else
|
||||
m_features.primitive_id = true;
|
||||
#endif
|
||||
m_features.prefer_new_textures = true;
|
||||
m_features.provoking_vertex_last = g_vulkan_context->GetOptionalExtensions().vk_ext_provoking_vertex;
|
||||
m_features.dual_source_blend = features.dualSrcBlend && !GSConfig.DisableDualSourceBlend;
|
||||
@@ -3782,15 +3786,12 @@ void GSDeviceVK::RenderHW(GSHWDrawConfig& config)
|
||||
{
|
||||
draw_rt = m_current_render_target;
|
||||
m_pipeline_selector.rt = true;
|
||||
m_pipeline_selector.cms.wrgba = 0;
|
||||
}
|
||||
else if (!draw_ds && m_current_depth_target && config.tex != m_current_depth_target &&
|
||||
m_current_depth_target->GetSize() == draw_rt->GetSize())
|
||||
{
|
||||
draw_ds = m_current_depth_target;
|
||||
m_pipeline_selector.ds = true;
|
||||
m_pipeline_selector.dss.ztst = ZTST_ALWAYS;
|
||||
m_pipeline_selector.dss.zwe = false;
|
||||
}
|
||||
|
||||
// Prefer keeping feedback loop enabled, that way we're not constantly restarting render passes
|
||||
|
||||
@@ -57,15 +57,16 @@ public:
|
||||
u32 key;
|
||||
};
|
||||
|
||||
GSHWDrawConfig::BlendState bs;
|
||||
GSHWDrawConfig::VSSelector vs;
|
||||
GSHWDrawConfig::DepthStencilSelector dss;
|
||||
GSHWDrawConfig::ColorMaskSelector cms;
|
||||
GSHWDrawConfig::BlendState bs;
|
||||
u8 pad;
|
||||
|
||||
__fi bool operator==(const PipelineSelector& p) const { return (memcmp(this, &p, sizeof(p)) == 0); }
|
||||
__fi bool operator!=(const PipelineSelector& p) const { return (memcmp(this, &p, sizeof(p)) != 0); }
|
||||
__fi bool operator==(const PipelineSelector& p) const { return BitEqual(*this, p); }
|
||||
__fi bool operator!=(const PipelineSelector& p) const { return !BitEqual(*this, p); }
|
||||
|
||||
__fi PipelineSelector() { memset(this, 0, sizeof(*this)); }
|
||||
__fi PipelineSelector() { std::memset(this, 0, sizeof(*this)); }
|
||||
|
||||
__fi bool IsRTFeedbackLoop() const { return ((feedback_loop_flags & FeedbackLoopFlag_ReadAndWriteRT) != 0); }
|
||||
__fi bool IsTestingAndSamplingDepth() const { return ((feedback_loop_flags & FeedbackLoopFlag_ReadDS) != 0); }
|
||||
|
||||
@@ -56,7 +56,7 @@ void intBreakpoint(bool memcheck)
|
||||
|
||||
CBreakPoints::SetBreakpointTriggered(true);
|
||||
VMManager::SetPaused(true);
|
||||
throw Exception::ExitCpuExecute();
|
||||
Cpu->ExitExecution();
|
||||
}
|
||||
|
||||
void intMemcheck(u32 op, u32 bits, bool store)
|
||||
|
||||
+1
-1
@@ -36,7 +36,7 @@ enum class FreezeAction
|
||||
// [SAVEVERSION+]
|
||||
// This informs the auto updater that the users savestates will be invalidated.
|
||||
|
||||
static const u32 g_SaveVersion = (0x9A35 << 16) | 0x0000;
|
||||
static const u32 g_SaveVersion = (0x9A36 << 16) | 0x0000;
|
||||
|
||||
|
||||
// the freezing data between submodules and core
|
||||
|
||||
@@ -15,4 +15,4 @@
|
||||
|
||||
/// Version number for GS and other shaders. Increment whenever any of the contents of the
|
||||
/// shaders change, to invalidate the cache.
|
||||
static constexpr u32 SHADER_CACHE_VERSION = 24;
|
||||
static constexpr u32 SHADER_CACHE_VERSION = 25;
|
||||
|
||||
@@ -157,7 +157,6 @@ struct alignas(16) VURegs
|
||||
u32 ebit;
|
||||
u32 pending_q;
|
||||
u32 pending_p;
|
||||
u32 blockhasmbit;
|
||||
|
||||
alignas(16) u32 micro_macflags[4];
|
||||
alignas(16) u32 micro_clipflags[4];
|
||||
|
||||
@@ -54,7 +54,6 @@ static void _vu0Exec(VURegs* VU)
|
||||
if (ptr[1] & 0x20000000 && VU == &VU0) // M flag
|
||||
{
|
||||
VU->flags |= VUFLAG_MFLAGSET;
|
||||
VU0.blockhasmbit = true;
|
||||
// Console.WriteLn("fixme: M flag set");
|
||||
}
|
||||
if (ptr[1] & 0x10000000) // D flag
|
||||
@@ -185,8 +184,6 @@ static void _vu0Exec(VURegs* VU)
|
||||
{
|
||||
VU->VI[REG_TPC].UL = VU->branchpc;
|
||||
|
||||
VU->blockhasmbit = false;
|
||||
|
||||
if (VU->takedelaybranch)
|
||||
{
|
||||
DevCon.Warning("VU0 - Branch/Jump in Delay Slot");
|
||||
@@ -205,8 +202,6 @@ static void _vu0Exec(VURegs* VU)
|
||||
_vuFlushAll(VU);
|
||||
VU0.VI[REG_VPU_STAT].UL &= ~0x1; /* E flag */
|
||||
vif0Regs.stat.VEW = false;
|
||||
|
||||
VU->blockhasmbit = false;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -109,7 +109,6 @@ void SaveStateBase::vuMicroFreeze()
|
||||
Freeze(VU0.ebit);
|
||||
Freeze(VU0.pending_q);
|
||||
Freeze(VU0.pending_p);
|
||||
Freeze(VU0.blockhasmbit);
|
||||
Freeze(VU0.micro_macflags);
|
||||
Freeze(VU0.micro_clipflags);
|
||||
Freeze(VU0.micro_statusflags);
|
||||
@@ -149,7 +148,6 @@ void SaveStateBase::vuMicroFreeze()
|
||||
Freeze(VU1.ebit);
|
||||
Freeze(VU1.pending_q);
|
||||
Freeze(VU1.pending_p);
|
||||
Freeze(VU1.blockhasmbit);
|
||||
Freeze(VU1.micro_macflags);
|
||||
Freeze(VU1.micro_clipflags);
|
||||
Freeze(VU1.micro_statusflags);
|
||||
|
||||
@@ -307,14 +307,6 @@ RecompiledCodeReserve::~RecompiledCodeReserve()
|
||||
Release();
|
||||
}
|
||||
|
||||
void RecompiledCodeReserve::_registerProfiler()
|
||||
{
|
||||
if (m_profiler_name.empty() || !IsOk())
|
||||
return;
|
||||
|
||||
Perf::any.map((uptr)m_baseptr, m_size, m_profiler_name.c_str());
|
||||
}
|
||||
|
||||
void RecompiledCodeReserve::Assign(VirtualMemoryManagerPtr allocator, size_t offset, size_t size)
|
||||
{
|
||||
// Anything passed to the memory allocator must be page aligned.
|
||||
@@ -329,7 +321,6 @@ void RecompiledCodeReserve::Assign(VirtualMemoryManagerPtr allocator, size_t off
|
||||
}
|
||||
|
||||
VirtualMemoryReserve::Assign(std::move(allocator), base, size);
|
||||
_registerProfiler();
|
||||
}
|
||||
|
||||
void RecompiledCodeReserve::Reset()
|
||||
@@ -353,13 +344,3 @@ void RecompiledCodeReserve::ForbidModification()
|
||||
{
|
||||
HostSys::MemProtect(m_baseptr, m_size, PageProtectionMode().Read().Execute());
|
||||
}
|
||||
|
||||
// Sets the abbreviated name used by the profiler. Name should be under 10 characters long.
|
||||
// After a name has been set, a profiler source will be automatically registered and cleared
|
||||
// in accordance with changes in the reserve area.
|
||||
RecompiledCodeReserve& RecompiledCodeReserve::SetProfilerName(std::string name)
|
||||
{
|
||||
m_profiler_name = std::move(name);
|
||||
_registerProfiler();
|
||||
return *this;
|
||||
}
|
||||
|
||||
@@ -154,9 +154,6 @@ class RecompiledCodeReserve : public VirtualMemoryReserve
|
||||
{
|
||||
typedef VirtualMemoryReserve _parent;
|
||||
|
||||
protected:
|
||||
std::string m_profiler_name;
|
||||
|
||||
public:
|
||||
RecompiledCodeReserve(std::string name);
|
||||
~RecompiledCodeReserve();
|
||||
@@ -164,14 +161,9 @@ public:
|
||||
void Assign(VirtualMemoryManagerPtr allocator, size_t offset, size_t size);
|
||||
void Reset();
|
||||
|
||||
RecompiledCodeReserve& SetProfilerName(std::string name);
|
||||
|
||||
void ForbidModification();
|
||||
void AllowModification();
|
||||
|
||||
operator u8*() { return m_baseptr; }
|
||||
operator const u8*() const { return m_baseptr; }
|
||||
|
||||
protected:
|
||||
void _registerProfiler();
|
||||
};
|
||||
@@ -88,7 +88,6 @@
|
||||
<None Include="..\bin\resources\shaders\vulkan\interlace.glsl" />
|
||||
<None Include="..\bin\resources\shaders\vulkan\merge.glsl" />
|
||||
<None Include="..\bin\resources\shaders\vulkan\tfx.glsl" />
|
||||
<None Include="..\bin\resources\shaders\opengl\common_header.glsl" />
|
||||
<None Include="..\bin\resources\shaders\opengl\convert.glsl" />
|
||||
<None Include="..\bin\resources\shaders\opengl\interlace.glsl" />
|
||||
<None Include="..\bin\resources\shaders\opengl\merge.glsl" />
|
||||
@@ -853,8 +852,7 @@
|
||||
</ProjectReference>
|
||||
</ItemGroup>
|
||||
<ItemGroup>
|
||||
<Natvis Include="GS\Renderers\Common\GSFastList.natvis" />
|
||||
<Natvis Include="GS\Renderers\HW\GSTextureCache.natvis" />
|
||||
<Natvis Include="GS\GS.natvis" />
|
||||
</ItemGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.targets" />
|
||||
<ImportGroup Label="ExtensionTargets" />
|
||||
|
||||
@@ -336,9 +336,6 @@
|
||||
<None Include="..\bin\resources\shaders\opengl\shadeboost.glsl">
|
||||
<Filter>System\Ps2\GS\Shaders\OpenGL</Filter>
|
||||
</None>
|
||||
<None Include="..\bin\resources\shaders\opengl\common_header.glsl">
|
||||
<Filter>System\Ps2\GS\Shaders\OpenGL</Filter>
|
||||
</None>
|
||||
<None Include="..\bin\resources\shaders\common\fxaa.fx">
|
||||
<Filter>System\Ps2\GS\Shaders\Common</Filter>
|
||||
</None>
|
||||
@@ -2283,11 +2280,8 @@
|
||||
</CustomBuildStep>
|
||||
</ItemGroup>
|
||||
<ItemGroup>
|
||||
<Natvis Include="GS\Renderers\Common\GSFastList.natvis">
|
||||
<Natvis Include="GS\GS.natvis">
|
||||
<Filter>System\Ps2\GS</Filter>
|
||||
</Natvis>
|
||||
<Natvis Include="GS\Renderers\HW\GSTextureCache.natvis">
|
||||
<Filter>System\Ps2\GS\Renderers\Hardware</Filter>
|
||||
</Natvis>
|
||||
</ItemGroup>
|
||||
</Project>
|
||||
+14
-9
@@ -283,7 +283,7 @@ static void _DynGen_Dispatchers()
|
||||
|
||||
recBlocks.SetJITCompile(iopJITCompile);
|
||||
|
||||
Perf::any.map((uptr)&iopRecDispatchers, 4096, "IOP Dispatcher");
|
||||
Perf::any.Register((void*)iopRecDispatchers, 4096, "IOP Dispatcher");
|
||||
}
|
||||
|
||||
////////////////////////////////////////////////////
|
||||
@@ -896,7 +896,6 @@ static void recReserve()
|
||||
return;
|
||||
|
||||
recMem = new RecompiledCodeReserve("R3000A Recompiler Cache");
|
||||
recMem->SetProfilerName("IOPrec");
|
||||
recMem->Assign(GetVmMemory().CodeMemory(), HostMemoryMap::IOPrecOffset, 32 * _1mb);
|
||||
}
|
||||
|
||||
@@ -940,8 +939,6 @@ void recResetIOP()
|
||||
{
|
||||
DevCon.WriteLn("iR3000A Recompiler reset.");
|
||||
|
||||
Perf::iop.reset();
|
||||
|
||||
recAlloc();
|
||||
recMem->Reset();
|
||||
|
||||
@@ -1005,9 +1002,6 @@ static void recShutdown()
|
||||
|
||||
safe_free(s_pInstCache);
|
||||
s_nInstCacheSize = 0;
|
||||
|
||||
// FIXME Warning thread unsafe
|
||||
Perf::dump();
|
||||
}
|
||||
|
||||
static void iopClearRecLUT(BASEBLOCK* base, int count)
|
||||
@@ -1388,7 +1382,18 @@ static void psxRecMemcheck(u32 op, u32 bits, bool store)
|
||||
if (checks[i].result & MEMCHECK_LOG)
|
||||
{
|
||||
xMOV(edx, store);
|
||||
xFastCall((void*)psxDynarecMemLogcheck, ecx, edx);
|
||||
|
||||
// Refer to the EE recompiler for an explaination
|
||||
if(!(checks[i].result & MEMCHECK_BREAK))
|
||||
{
|
||||
xPUSH(eax); xPUSH(ebx); xPUSH(ecx); xPUSH(edx);
|
||||
xFastCall((void*)psxDynarecMemLogcheck, ecx, edx);
|
||||
xPOP(edx); xPOP(ecx); xPOP(ebx); xPOP(eax);
|
||||
}
|
||||
else
|
||||
{
|
||||
xFastCall((void*)psxDynarecMemLogcheck, ecx, edx);
|
||||
}
|
||||
}
|
||||
if (checks[i].result & MEMCHECK_BREAK)
|
||||
{
|
||||
@@ -1768,7 +1773,7 @@ StartRecomp:
|
||||
pxAssert(xGetPtr() - recPtr < _64kb);
|
||||
s_pCurBlockEx->x86size = xGetPtr() - recPtr;
|
||||
|
||||
Perf::iop.map(s_pCurBlockEx->fnptr, s_pCurBlockEx->x86size, s_pCurBlockEx->startpc);
|
||||
Perf::iop.RegisterPC((void*)s_pCurBlockEx->fnptr, s_pCurBlockEx->x86size, s_pCurBlockEx->startpc);
|
||||
|
||||
recPtr = xGetPtr();
|
||||
|
||||
|
||||
+114
-19
@@ -514,7 +514,7 @@ static void _DynGen_Dispatchers()
|
||||
|
||||
recBlocks.SetJITCompile(JITCompile);
|
||||
|
||||
Perf::any.map((uptr)&eeRecDispatchers, 4096, "EE Dispatcher");
|
||||
Perf::any.Register((void*)eeRecDispatchers, 4096, "EE Dispatcher");
|
||||
}
|
||||
|
||||
|
||||
@@ -533,7 +533,6 @@ static void recReserve()
|
||||
return;
|
||||
|
||||
recMem = new RecompiledCodeReserve("R5900 Recompiler Cache");
|
||||
recMem->SetProfilerName("EErec");
|
||||
recMem->Assign(GetVmMemory().CodeMemory(), HostMemoryMap::EErecOffset, 64 * _1mb);
|
||||
}
|
||||
|
||||
@@ -616,8 +615,6 @@ static void recResetRaw()
|
||||
{
|
||||
Console.WriteLn(Color_StrongBlack, "EE/iR5900-32 Recompiler Reset");
|
||||
|
||||
Perf::ee.reset();
|
||||
|
||||
EE::Profiler.Reset();
|
||||
|
||||
recAlloc();
|
||||
@@ -655,9 +652,6 @@ static void recShutdown()
|
||||
|
||||
safe_free(s_pInstCache);
|
||||
s_nInstCacheSize = 0;
|
||||
|
||||
// FIXME Warning thread unsafe
|
||||
Perf::dump();
|
||||
}
|
||||
|
||||
void recStep()
|
||||
@@ -743,9 +737,6 @@ static void recExecute()
|
||||
|
||||
eeCpuExecuting = false;
|
||||
|
||||
// FIXME Warning thread unsafe
|
||||
Perf::dump();
|
||||
|
||||
EE::Profiler.Print();
|
||||
}
|
||||
|
||||
@@ -1636,10 +1627,23 @@ void recMemcheck(u32 op, u32 bits, bool store)
|
||||
if (checks[i].result & MEMCHECK_LOG)
|
||||
{
|
||||
xMOV(edx, store);
|
||||
xFastCall((void*)dynarecMemLogcheck, ecx, edx);
|
||||
// Preserve ecx (address) and edx (address+size) because we aren't breaking
|
||||
// out of this loops iteration and dynarecMemLogcheck will clobber them
|
||||
// Also keep 16 byte stack alignment
|
||||
if(!(checks[i].result & MEMCHECK_BREAK))
|
||||
{
|
||||
xPUSH(eax); xPUSH(ebx); xPUSH(ecx); xPUSH(edx);
|
||||
xFastCall((void*)dynarecMemLogcheck, ecx, edx);
|
||||
xPOP(edx); xPOP(ecx); xPOP(ebx); xPOP(eax);
|
||||
}
|
||||
else
|
||||
{
|
||||
xFastCall((void*)dynarecMemLogcheck, ecx, edx);
|
||||
}
|
||||
}
|
||||
if (checks[i].result & MEMCHECK_BREAK)
|
||||
{
|
||||
// Don't need to preserve edx and ecx, we don't return
|
||||
xFastCall((void*)dynarecMemcheck);
|
||||
}
|
||||
|
||||
@@ -2129,7 +2133,7 @@ static void memory_protect_recompiled_code(u32 startpc, u32 size)
|
||||
}
|
||||
|
||||
// Skip MPEG Game-Fix
|
||||
bool skipMPEG_By_Pattern(u32 sPC)
|
||||
static bool skipMPEG_By_Pattern(u32 sPC)
|
||||
{
|
||||
|
||||
if (!CHECK_SKIPMPEGHACK)
|
||||
@@ -2158,6 +2162,56 @@ bool skipMPEG_By_Pattern(u32 sPC)
|
||||
return 0;
|
||||
}
|
||||
|
||||
static bool recSkipTimeoutLoop(s32 reg, bool is_timeout_loop)
|
||||
{
|
||||
if (!EmuConfig.Speedhacks.WaitLoop || !is_timeout_loop)
|
||||
return false;
|
||||
|
||||
DevCon.WriteLn("[EE] Skipping timeout loop at 0x%08X -> 0x%08X", s_pCurBlockEx->startpc, s_nEndBlock);
|
||||
|
||||
// basically, if the time it takes the loop to run is shorter than the
|
||||
// time to the next event, then we want to skip ahead to the event, but
|
||||
// update v0 to reflect how long the loop would have run for.
|
||||
|
||||
// if (cycle >= nextEventCycle) { jump to dispatcher, we're running late }
|
||||
// new_cycles = min(v0 * 8, nextEventCycle)
|
||||
// new_v0 = (new_cycles - cycles) / 8
|
||||
// if new_v0 > 0 { jump to dispatcher because loop exited early }
|
||||
// else new_v0 is 0, so exit loop
|
||||
|
||||
xMOV(ebx, ptr32[&cpuRegs.cycle]); // ebx = cycle
|
||||
xMOV(ecx, ptr32[&cpuRegs.nextEventCycle]); // ecx = nextEventCycle
|
||||
xCMP(ebx, ecx);
|
||||
//xJAE((void*)DispatcherEvent); // jump to dispatcher if event immediately
|
||||
|
||||
// TODO: In the case where nextEventCycle < cycle because it's overflowed, tack 8
|
||||
// cycles onto the event count, so hopefully it'll wrap around. This is pretty
|
||||
// gross, but until we switch to 64-bit counters, not many better options.
|
||||
xForwardJB8 not_dispatcher;
|
||||
xADD(ebx, 8);
|
||||
xMOV(ptr32[&cpuRegs.cycle], ebx);
|
||||
xJMP((void*)DispatcherEvent);
|
||||
not_dispatcher.SetTarget();
|
||||
|
||||
xMOV(edx, ptr32[&cpuRegs.GPR.r[reg].UL[0]]); // eax = v0
|
||||
xLEA(rax, ptrNative[rdx * 8 + rbx]); // edx = v0 * 8 + cycle
|
||||
xCMP(rcx, rax);
|
||||
xCMOVB(rax, rcx); // eax = new_cycles = min(v8 * 8, nextEventCycle)
|
||||
xMOV(ptr32[&cpuRegs.cycle], eax); // writeback new_cycles
|
||||
xSUB(eax, ebx); // new_cycles -= cycle
|
||||
xSHR(eax, 3); // compute new v0 value
|
||||
xSUB(edx, eax); // v0 -= cycle_diff
|
||||
xMOV(ptr32[&cpuRegs.GPR.r[reg].UL[0]], edx); // write back new value of v0
|
||||
xJNZ((void*)DispatcherEvent); // jump to dispatcher if new v0 is not zero (i.e. an event)
|
||||
xMOV(ptr32[&cpuRegs.pc], s_nEndBlock); // otherwise end of loop
|
||||
recBlocks.Link(HWADDR(s_nEndBlock), xJcc32());
|
||||
|
||||
g_branch = 1;
|
||||
pc = s_nEndBlock;
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
static void recRecompile(const u32 startpc)
|
||||
{
|
||||
u32 i = 0;
|
||||
@@ -2268,6 +2322,26 @@ static void recRecompile(const u32 startpc)
|
||||
s_nEndBlock = 0xffffffff;
|
||||
s_branchTo = -1;
|
||||
|
||||
// Timeout loop speedhack.
|
||||
// God of War 2 and other games (e.g. NFS series) have these timeout loops which just spin for a few thousand
|
||||
// iterations, usually after kicking something which results in an IRQ, but instead of cancelling the loop,
|
||||
// they just let it finish anyway. Such loops look like:
|
||||
//
|
||||
// 00186D6C addiu v0,v0, -0x1
|
||||
// 00186D70 nop
|
||||
// 00186D74 nop
|
||||
// 00186D78 nop
|
||||
// 00186D7C nop
|
||||
// 00186D80 bne v0, zero, ->$0x00186D6C
|
||||
// 00186D84 nop
|
||||
//
|
||||
// Skipping them entirely seems to have no negative effects, but we skip cycles based on the incoming value
|
||||
// if the register being decremented, which appears to vary. So far I haven't seen any which increment instead
|
||||
// of decrementing, so we'll limit the test to that to be safe.
|
||||
//
|
||||
s32 timeout_reg = -1;
|
||||
bool is_timeout_loop = true;
|
||||
|
||||
// compile breakpoints as individual blocks
|
||||
int n1 = isBreakpointNeeded(i);
|
||||
int n2 = isMemcheckNeeded(i);
|
||||
@@ -2311,6 +2385,28 @@ static void recRecompile(const u32 startpc)
|
||||
//HUH ? PSM ? whut ? THIS IS VIRTUAL ACCESS GOD DAMMIT
|
||||
cpuRegs.code = *(int*)PSM(i);
|
||||
|
||||
if (is_timeout_loop)
|
||||
{
|
||||
if ((cpuRegs.code >> 26) == 8 || (cpuRegs.code >> 26) == 9)
|
||||
{
|
||||
// addi/addiu
|
||||
if (timeout_reg >= 0 || _Rs_ != _Rt_ || _Imm_ >= 0)
|
||||
is_timeout_loop = false;
|
||||
else
|
||||
timeout_reg = _Rs_;
|
||||
}
|
||||
else if ((cpuRegs.code >> 26) == 5)
|
||||
{
|
||||
// bne
|
||||
if (timeout_reg != _Rs_ || _Rt_ != 0 || memRead32(i + 4) != 0)
|
||||
is_timeout_loop = false;
|
||||
}
|
||||
else if (cpuRegs.code != 0)
|
||||
{
|
||||
is_timeout_loop = false;
|
||||
}
|
||||
}
|
||||
|
||||
switch (cpuRegs.code >> 26)
|
||||
{
|
||||
case 0: // special
|
||||
@@ -2402,11 +2498,6 @@ StartRecomp:
|
||||
// (excepting registers initialised with constants or memory loads) or use any instructions
|
||||
// which alter the machine state apart from registers, it will do the same thing on every
|
||||
// iteration.
|
||||
// TODO: special handling for counting loops. God of war wastes time in a loop which just
|
||||
// counts to some large number and does nothing else, many other games use a counter as a
|
||||
// timeout on a register read. AFAICS the only way to optimise this for non-const cases
|
||||
// without a significant loss in cycle accuracy is with a division, but games would probably
|
||||
// be happy with time wasting loops completing in 0 cycles and timeouts waiting forever.
|
||||
s_nBlockFF = false;
|
||||
if (s_branchTo == startpc)
|
||||
{
|
||||
@@ -2485,6 +2576,10 @@ StartRecomp:
|
||||
}
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
is_timeout_loop = false;
|
||||
}
|
||||
|
||||
// rec info //
|
||||
bool has_cop2_instructions = false;
|
||||
@@ -2547,7 +2642,7 @@ StartRecomp:
|
||||
memory_protect_recompiled_code(startpc, (s_nEndBlock - startpc) >> 2);
|
||||
|
||||
// Skip Recompilation if sceMpegIsEnd Pattern detected
|
||||
bool doRecompilation = !skipMPEG_By_Pattern(startpc);
|
||||
bool doRecompilation = !skipMPEG_By_Pattern(startpc) && !recSkipTimeoutLoop(timeout_reg, is_timeout_loop);
|
||||
|
||||
if (doRecompilation)
|
||||
{
|
||||
@@ -2678,7 +2773,7 @@ StartRecomp:
|
||||
iDumpBlock(s_pCurBlockEx->startpc, s_pCurBlockEx->size*4, s_pCurBlockEx->fnptr, s_pCurBlockEx->x86size);
|
||||
}
|
||||
#endif
|
||||
Perf::ee.map(s_pCurBlockEx->fnptr, s_pCurBlockEx->x86size, s_pCurBlockEx->startpc);
|
||||
Perf::ee.RegisterPC((void*)s_pCurBlockEx->fnptr, s_pCurBlockEx->x86size, s_pCurBlockEx->startpc);
|
||||
|
||||
recPtr = xGetPtr();
|
||||
|
||||
|
||||
@@ -388,7 +388,7 @@ void vtlb_dynarec_init()
|
||||
|
||||
HostSys::MemProtectStatic(m_IndirectDispatchers, PageAccess_ExecOnly());
|
||||
|
||||
Perf::any.map((uptr)m_IndirectDispatchers, __pagesize, "TLB Dispatcher");
|
||||
Perf::any.Register(m_IndirectDispatchers, __pagesize, "TLB Dispatcher");
|
||||
}
|
||||
|
||||
//////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
@@ -31,7 +31,6 @@ alignas(__pagesize) static u8 vu1_RecDispatchers[mVUdispCacheSize];
|
||||
void mVUreserveCache(microVU& mVU)
|
||||
{
|
||||
mVU.cache_reserve = new RecompiledCodeReserve(StringUtil::StdStringFromFormat("Micro VU%u Recompiler Cache", mVU.index));
|
||||
mVU.cache_reserve->SetProfilerName(StringUtil::StdStringFromFormat("mVU%urec", mVU.index));
|
||||
|
||||
const size_t alloc_offset = mVU.index ? HostMemoryMap::mVU0recOffset : HostMemoryMap::mVU1recOffset;
|
||||
mVU.cache_reserve->Assign(GetVmMemory().CodeMemory(), alloc_offset, mVU.cacheSize * _1mb);
|
||||
@@ -128,11 +127,6 @@ void mVUreset(microVU& mVU, bool resetReserve)
|
||||
}
|
||||
|
||||
HostSys::MemProtect(mVU.dispCache, mVUdispCacheSize, PageAccess_ExecOnly());
|
||||
|
||||
if (mVU.index)
|
||||
Perf::any.map((uptr)&mVU.dispCache, mVUdispCacheSize, "mVU1 Dispatcher");
|
||||
else
|
||||
Perf::any.map((uptr)&mVU.dispCache, mVUdispCacheSize, "mVU0 Dispatcher");
|
||||
}
|
||||
|
||||
// Free Allocated Resources
|
||||
|
||||
+1
-1
@@ -164,7 +164,7 @@ public:
|
||||
{
|
||||
u32 viCRC = 0, vfCRC = 0, crc = 0, z = sizeof(microRegInfo) / 4;
|
||||
for (u32 j = 0; j < 4; j++) viCRC -= ((u32*)linkI->block.pState.VI)[j];
|
||||
for (u32 j = 0; j < 32; j++) vfCRC -= linkI->block.pState.VF[j].reg;
|
||||
for (u32 j = 0; j < 32; j++) vfCRC -= linkI->block.pState.VF[j].x + (linkI->block.pState.VF[j].y << 8) + (linkI->block.pState.VF[j].z << 16) + (linkI->block.pState.VF[j].w << 24);
|
||||
for (u32 j = 0; j < z; j++) crc -= ((u32*)&linkI->block.pState)[j];
|
||||
DevCon.WriteLn(Color_Green,
|
||||
"[%04x][Block #%d][crc=%08x][q=%02d][p=%02d][xgkick=%d][vi15=%04x][vi15v=%d][viBackup=%02d]"
|
||||
|
||||
@@ -314,9 +314,9 @@ __ri void eBitWarning(mV)
|
||||
//------------------------------------------------------------------
|
||||
// Cycles / Pipeline State / Early Exit from Execution
|
||||
//------------------------------------------------------------------
|
||||
__fi void optimizeReg(u8& rState) { rState = (rState == 1) ? 0 : rState; }
|
||||
__fi void calcCycles(u8& reg, u8 x) { reg = ((reg > x) ? (reg - x) : 0); }
|
||||
__fi void tCycles(u8& dest, u8& src) { dest = std::max(dest, src); }
|
||||
__fi u8 optimizeReg(u8 rState) { return (rState == 1) ? 0 : rState; }
|
||||
__fi u8 calcCycles(u8 reg, u8 x) { return ((reg > x) ? (reg - x) : 0); }
|
||||
__fi u8 tCycles(u8 dest, u8 src) { return std::max(dest, src); }
|
||||
__fi void incP(mV) { mVU.p ^= 1; }
|
||||
__fi void incQ(mV) { mVU.q ^= 1; }
|
||||
|
||||
@@ -328,17 +328,17 @@ void mVUoptimizePipeState(mV)
|
||||
{
|
||||
for (int i = 0; i < 32; i++)
|
||||
{
|
||||
optimizeReg(mVUregs.VF[i].x);
|
||||
optimizeReg(mVUregs.VF[i].y);
|
||||
optimizeReg(mVUregs.VF[i].z);
|
||||
optimizeReg(mVUregs.VF[i].w);
|
||||
mVUregs.VF[i].x = optimizeReg(mVUregs.VF[i].x);
|
||||
mVUregs.VF[i].y = optimizeReg(mVUregs.VF[i].y);
|
||||
mVUregs.VF[i].z = optimizeReg(mVUregs.VF[i].z);
|
||||
mVUregs.VF[i].w = optimizeReg(mVUregs.VF[i].w);
|
||||
}
|
||||
for (int i = 0; i < 16; i++)
|
||||
{
|
||||
optimizeReg(mVUregs.VI[i]);
|
||||
mVUregs.VI[i] = optimizeReg(mVUregs.VI[i]);
|
||||
}
|
||||
if (mVUregs.q) { optimizeReg(mVUregs.q); if (!mVUregs.q) { incQ(mVU); } }
|
||||
if (mVUregs.p) { optimizeReg(mVUregs.p); if (!mVUregs.p) { incP(mVU); } }
|
||||
if (mVUregs.q) { mVUregs.q = optimizeReg(mVUregs.q); if (!mVUregs.q) { incQ(mVU); } }
|
||||
if (mVUregs.p) { mVUregs.p = optimizeReg(mVUregs.p); if (!mVUregs.p) { incP(mVU); } }
|
||||
mVUregs.r = 0; // There are no stalls on the R-reg, so its Safe to discard info
|
||||
}
|
||||
|
||||
@@ -348,21 +348,21 @@ void mVUincCycles(mV, int x)
|
||||
// VF[0] is a constant value (0.0 0.0 0.0 1.0)
|
||||
for (int z = 31; z > 0; z--)
|
||||
{
|
||||
calcCycles(mVUregs.VF[z].x, x);
|
||||
calcCycles(mVUregs.VF[z].y, x);
|
||||
calcCycles(mVUregs.VF[z].z, x);
|
||||
calcCycles(mVUregs.VF[z].w, x);
|
||||
mVUregs.VF[z].x = calcCycles(mVUregs.VF[z].x, x);
|
||||
mVUregs.VF[z].y = calcCycles(mVUregs.VF[z].y, x);
|
||||
mVUregs.VF[z].z = calcCycles(mVUregs.VF[z].z, x);
|
||||
mVUregs.VF[z].w = calcCycles(mVUregs.VF[z].w, x);
|
||||
}
|
||||
// VI[0] is a constant value (0)
|
||||
for (int z = 15; z > 0; z--)
|
||||
{
|
||||
calcCycles(mVUregs.VI[z], x);
|
||||
mVUregs.VI[z] = calcCycles(mVUregs.VI[z], x);
|
||||
}
|
||||
if (mVUregs.q)
|
||||
{
|
||||
if (mVUregs.q > 4)
|
||||
{
|
||||
calcCycles(mVUregs.q, x);
|
||||
mVUregs.q = calcCycles(mVUregs.q, x);
|
||||
if (mVUregs.q <= 4)
|
||||
{
|
||||
mVUinfo.doDivFlag = 1;
|
||||
@@ -370,27 +370,27 @@ void mVUincCycles(mV, int x)
|
||||
}
|
||||
else
|
||||
{
|
||||
calcCycles(mVUregs.q, x);
|
||||
mVUregs.q = calcCycles(mVUregs.q, x);
|
||||
}
|
||||
if (!mVUregs.q)
|
||||
incQ(mVU);
|
||||
}
|
||||
if (mVUregs.p)
|
||||
{
|
||||
calcCycles(mVUregs.p, x);
|
||||
mVUregs.p = calcCycles(mVUregs.p, x);
|
||||
if (!mVUregs.p || mVUregsTemp.p)
|
||||
incP(mVU);
|
||||
}
|
||||
if (mVUregs.xgkick)
|
||||
{
|
||||
calcCycles(mVUregs.xgkick, x);
|
||||
mVUregs.xgkick = calcCycles(mVUregs.xgkick, x);
|
||||
if (!mVUregs.xgkick)
|
||||
{
|
||||
mVUinfo.doXGKICK = 1;
|
||||
mVUinfo.XGKICKPC = xPC;
|
||||
}
|
||||
}
|
||||
calcCycles(mVUregs.r, x);
|
||||
mVUregs.r = calcCycles(mVUregs.r, x);
|
||||
}
|
||||
|
||||
// Helps check if upper/lower ops read/write to same regs...
|
||||
@@ -430,21 +430,21 @@ void mVUsetCycles(mV)
|
||||
cmpVFregs(mVUlow.VF_write, mVUup.VF_read[1], mVUinfo.backupVF);
|
||||
}
|
||||
|
||||
tCycles(mVUregs.VF[mVUregsTemp.VFreg[0]].x, mVUregsTemp.VF[0].x);
|
||||
tCycles(mVUregs.VF[mVUregsTemp.VFreg[0]].y, mVUregsTemp.VF[0].y);
|
||||
tCycles(mVUregs.VF[mVUregsTemp.VFreg[0]].z, mVUregsTemp.VF[0].z);
|
||||
tCycles(mVUregs.VF[mVUregsTemp.VFreg[0]].w, mVUregsTemp.VF[0].w);
|
||||
mVUregs.VF[mVUregsTemp.VFreg[0]].x = tCycles(mVUregs.VF[mVUregsTemp.VFreg[0]].x, mVUregsTemp.VF[0].x);
|
||||
mVUregs.VF[mVUregsTemp.VFreg[0]].y = tCycles(mVUregs.VF[mVUregsTemp.VFreg[0]].y, mVUregsTemp.VF[0].y);
|
||||
mVUregs.VF[mVUregsTemp.VFreg[0]].z = tCycles(mVUregs.VF[mVUregsTemp.VFreg[0]].z, mVUregsTemp.VF[0].z);
|
||||
mVUregs.VF[mVUregsTemp.VFreg[0]].w = tCycles(mVUregs.VF[mVUregsTemp.VFreg[0]].w, mVUregsTemp.VF[0].w);
|
||||
|
||||
tCycles(mVUregs.VF[mVUregsTemp.VFreg[1]].x, mVUregsTemp.VF[1].x);
|
||||
tCycles(mVUregs.VF[mVUregsTemp.VFreg[1]].y, mVUregsTemp.VF[1].y);
|
||||
tCycles(mVUregs.VF[mVUregsTemp.VFreg[1]].z, mVUregsTemp.VF[1].z);
|
||||
tCycles(mVUregs.VF[mVUregsTemp.VFreg[1]].w, mVUregsTemp.VF[1].w);
|
||||
mVUregs.VF[mVUregsTemp.VFreg[1]].x = tCycles(mVUregs.VF[mVUregsTemp.VFreg[1]].x, mVUregsTemp.VF[1].x);
|
||||
mVUregs.VF[mVUregsTemp.VFreg[1]].y = tCycles(mVUregs.VF[mVUregsTemp.VFreg[1]].y, mVUregsTemp.VF[1].y);
|
||||
mVUregs.VF[mVUregsTemp.VFreg[1]].z = tCycles(mVUregs.VF[mVUregsTemp.VFreg[1]].z, mVUregsTemp.VF[1].z);
|
||||
mVUregs.VF[mVUregsTemp.VFreg[1]].w = tCycles(mVUregs.VF[mVUregsTemp.VFreg[1]].w, mVUregsTemp.VF[1].w);
|
||||
|
||||
tCycles(mVUregs.VI[mVUregsTemp.VIreg], mVUregsTemp.VI);
|
||||
tCycles(mVUregs.q, mVUregsTemp.q);
|
||||
tCycles(mVUregs.p, mVUregsTemp.p);
|
||||
tCycles(mVUregs.r, mVUregsTemp.r);
|
||||
tCycles(mVUregs.xgkick, mVUregsTemp.xgkick);
|
||||
mVUregs.VI[mVUregsTemp.VIreg] = tCycles(mVUregs.VI[mVUregsTemp.VIreg], mVUregsTemp.VI);
|
||||
mVUregs.q = tCycles(mVUregs.q, mVUregsTemp.q);
|
||||
mVUregs.p = tCycles(mVUregs.p, mVUregsTemp.p);
|
||||
mVUregs.r = tCycles(mVUregs.r, mVUregsTemp.r);
|
||||
mVUregs.xgkick = tCycles(mVUregs.xgkick, mVUregsTemp.xgkick);
|
||||
}
|
||||
|
||||
// Prints Start/End PC of blocks executed, for debugging...
|
||||
@@ -470,8 +470,7 @@ void mVUtestCycles(microVU& mVU, microFlagCycles& mFC)
|
||||
else
|
||||
xSUB(eax, 1); // Running ahead, make sure cycles left are above 0
|
||||
|
||||
xCMP(eax, 0);
|
||||
xForwardJGE32 skip;
|
||||
xForwardJNS32 skip;
|
||||
|
||||
u8* writeback = x86Ptr;
|
||||
xLoadFarAddr(rax, x86Ptr);
|
||||
@@ -556,7 +555,6 @@ __fi void mVUinitFirstPass(microVU& mVU, uptr pState, u8* thisPtr)
|
||||
mVUregs.blockType = 0;
|
||||
mVUregs.viBackUp = 0;
|
||||
mVUregs.flagInfo = 0;
|
||||
mVUregs.mbitinblock = false;
|
||||
mVUsFlagHack = CHECK_VU_FLAGHACK;
|
||||
mVUinitConstValues(mVU);
|
||||
}
|
||||
@@ -727,7 +725,6 @@ void* mVUcompile(microVU& mVU, u32 startPC, uptr pState)
|
||||
|
||||
if ((curI & _Mbit_) && isVU0)
|
||||
{
|
||||
mVUregs.mbitinblock = true;
|
||||
if (xPC > 0)
|
||||
{
|
||||
incPC(-2);
|
||||
@@ -850,7 +847,6 @@ void* mVUcompile(microVU& mVU, u32 startPC, uptr pState)
|
||||
// Fix up vi15 const info for propagation through blocks
|
||||
mVUregs.vi15 = (doConstProp && mVUconstReg[15].isValid) ? (u16)mVUconstReg[15].regValue : 0;
|
||||
mVUregs.vi15v = (doConstProp && mVUconstReg[15].isValid) ? 1 : 0;
|
||||
xMOV(ptr32[&mVU.regs().blockhasmbit], mVUregs.mbitinblock);
|
||||
mVUsetFlags(mVU, mFC); // Sets Up Flag instances
|
||||
mVUoptimizePipeState(mVU); // Optimize the End Pipeline State for nicer Block Linking
|
||||
mVUdebugPrintBlocks(mVU, false); // Prints Start/End PC of blocks executed, for debugging...
|
||||
@@ -997,7 +993,13 @@ void* mVUcompile(microVU& mVU, u32 startPC, uptr pState)
|
||||
|
||||
perf_and_return:
|
||||
|
||||
Perf::vu.map((uptr)thisPtr, x86Ptr - thisPtr, startPC);
|
||||
if (mVU.regs().start_pc == startPC)
|
||||
{
|
||||
if (mVU.index)
|
||||
Perf::vu1.RegisterPC(thisPtr, static_cast<u32>(x86Ptr - thisPtr), startPC);
|
||||
else
|
||||
Perf::vu0.RegisterPC(thisPtr, static_cast<u32>(x86Ptr - thisPtr), startPC);
|
||||
}
|
||||
|
||||
return thisPtr;
|
||||
}
|
||||
|
||||
@@ -94,6 +94,9 @@ void mVUdispatcherAB(mV)
|
||||
|
||||
pxAssertDev(xGetPtr() < (mVU.dispCache + mVUdispCacheSize),
|
||||
"microVU: Dispatcher generation exceeded reserved cache area!");
|
||||
|
||||
Perf::any.Register(mVU.startFunct, static_cast<u32>(xGetPtr() - mVU.startFunct),
|
||||
mVU.index ? "VU1StartFunc" : "VU0StartFunc");
|
||||
}
|
||||
|
||||
// Generates the code for resuming/exit xgkick
|
||||
@@ -134,6 +137,9 @@ void mVUdispatcherCD(mV)
|
||||
|
||||
pxAssertDev(xGetPtr() < (mVU.dispCache + mVUdispCacheSize),
|
||||
"microVU: Dispatcher generation exceeded reserved cache area!");
|
||||
|
||||
Perf::any.Register(mVU.startFunctXG, static_cast<u32>(xGetPtr() - mVU.startFunctXG),
|
||||
mVU.index ? "VU1StartFuncXG" : "VU0StartFuncXG");
|
||||
}
|
||||
|
||||
void mvuGenerateWaitMTVU(mV)
|
||||
@@ -211,6 +217,9 @@ void mvuGenerateWaitMTVU(mV)
|
||||
|
||||
pxAssertDev(xGetPtr() < (mVU.dispCache + mVUdispCacheSize),
|
||||
"microVU: Dispatcher generation exceeded reserved cache area!");
|
||||
|
||||
Perf::any.Register(mVU.waitMTVU, static_cast<u32>(xGetPtr() - mVU.waitMTVU),
|
||||
mVU.index ? "VU1WaitMTVU" : "VU0WaitMTVU");
|
||||
}
|
||||
|
||||
void mvuGenerateCopyPipelineState(mV)
|
||||
@@ -223,14 +232,10 @@ void mvuGenerateCopyPipelineState(mV)
|
||||
xVMOVAPS(ymm0, ptr[rax]);
|
||||
xVMOVAPS(ymm1, ptr[rax + 32u]);
|
||||
xVMOVAPS(ymm2, ptr[rax + 64u]);
|
||||
xVMOVAPS(ymm3, ptr[rax + 96u]);
|
||||
xVMOVAPS(ymm4, ptr[rax + 128u]);
|
||||
|
||||
xVMOVUPS(ptr[reinterpret_cast<u8*>(&mVU.prog.lpState)], ymm0);
|
||||
xVMOVUPS(ptr[reinterpret_cast<u8*>(&mVU.prog.lpState) + 32u], ymm1);
|
||||
xVMOVUPS(ptr[reinterpret_cast<u8*>(&mVU.prog.lpState) + 64u], ymm2);
|
||||
xVMOVUPS(ptr[reinterpret_cast<u8*>(&mVU.prog.lpState) + 96u], ymm3);
|
||||
xVMOVUPS(ptr[reinterpret_cast<u8*>(&mVU.prog.lpState) + 128u], ymm4);
|
||||
|
||||
xVZEROUPPER();
|
||||
}
|
||||
@@ -242,10 +247,6 @@ void mvuGenerateCopyPipelineState(mV)
|
||||
xMOVAPS(xmm3, ptr[rax + 48u]);
|
||||
xMOVAPS(xmm4, ptr[rax + 64u]);
|
||||
xMOVAPS(xmm5, ptr[rax + 80u]);
|
||||
xMOVAPS(xmm6, ptr[rax + 96u]);
|
||||
xMOVAPS(xmm7, ptr[rax + 112u]);
|
||||
xMOVAPS(xmm8, ptr[rax + 128u]);
|
||||
xMOVAPS(xmm9, ptr[rax + 144u]);
|
||||
|
||||
xMOVUPS(ptr[reinterpret_cast<u8*>(&mVU.prog.lpState)], xmm0);
|
||||
xMOVUPS(ptr[reinterpret_cast<u8*>(&mVU.prog.lpState) + 16u], xmm1);
|
||||
@@ -253,16 +254,15 @@ void mvuGenerateCopyPipelineState(mV)
|
||||
xMOVUPS(ptr[reinterpret_cast<u8*>(&mVU.prog.lpState) + 48u], xmm3);
|
||||
xMOVUPS(ptr[reinterpret_cast<u8*>(&mVU.prog.lpState) + 64u], xmm4);
|
||||
xMOVUPS(ptr[reinterpret_cast<u8*>(&mVU.prog.lpState) + 80u], xmm5);
|
||||
xMOVUPS(ptr[reinterpret_cast<u8*>(&mVU.prog.lpState) + 96u], xmm6);
|
||||
xMOVUPS(ptr[reinterpret_cast<u8*>(&mVU.prog.lpState) + 112u], xmm7);
|
||||
xMOVUPS(ptr[reinterpret_cast<u8*>(&mVU.prog.lpState) + 128u], xmm8);
|
||||
xMOVUPS(ptr[reinterpret_cast<u8*>(&mVU.prog.lpState) + 144u], xmm9);
|
||||
}
|
||||
|
||||
xRET();
|
||||
|
||||
pxAssertDev(xGetPtr() < (mVU.dispCache + mVUdispCacheSize),
|
||||
"microVU: Dispatcher generation exceeded reserved cache area!");
|
||||
|
||||
Perf::any.Register(mVU.copyPLState, static_cast<u32>(xGetPtr() - mVU.copyPLState),
|
||||
mVU.index ? "VU1CopyPLState" : "VU0CopyPLState");
|
||||
}
|
||||
|
||||
//------------------------------------------------------------------
|
||||
|
||||
+19
-23
@@ -17,16 +17,12 @@
|
||||
#include "microVU.h"
|
||||
#include <array>
|
||||
|
||||
union regInfo
|
||||
struct regCycleInfo
|
||||
{
|
||||
u32 reg;
|
||||
struct
|
||||
{
|
||||
u8 x;
|
||||
u8 y;
|
||||
u8 z;
|
||||
u8 w;
|
||||
};
|
||||
u8 x : 4;
|
||||
u8 y : 4;
|
||||
u8 z : 4;
|
||||
u8 w : 4;
|
||||
};
|
||||
|
||||
// microRegInfo is carefully ordered for faster compares. The "important" information is
|
||||
@@ -57,24 +53,24 @@ union alignas(16) microRegInfo
|
||||
};
|
||||
|
||||
u32 xgkickcycles;
|
||||
u8 mbitinblock;
|
||||
u8 unused;
|
||||
u8 vi15v; // 'vi15' constant is valid
|
||||
u16 vi15; // Constant Prop Info for vi15
|
||||
|
||||
struct
|
||||
{
|
||||
u8 VI[16];
|
||||
regInfo VF[32];
|
||||
regCycleInfo VF[32];
|
||||
};
|
||||
};
|
||||
|
||||
u128 full128[160 / sizeof(u128)];
|
||||
u64 full64[160 / sizeof(u64)];
|
||||
u32 full32[160 / sizeof(u32)];
|
||||
u128 full128[96 / sizeof(u128)];
|
||||
u64 full64[96 / sizeof(u64)];
|
||||
u32 full32[96 / sizeof(u32)];
|
||||
};
|
||||
|
||||
// Note: mVUcustomSearch needs to be updated if this is changed
|
||||
static_assert(sizeof(microRegInfo) == 160, "microRegInfo was not 160 bytes");
|
||||
static_assert(sizeof(microRegInfo) == 96, "microRegInfo was not 96 bytes");
|
||||
|
||||
struct microProgram;
|
||||
struct microJumpCache
|
||||
@@ -94,14 +90,14 @@ struct alignas(16) microBlock
|
||||
|
||||
struct microTempRegInfo
|
||||
{
|
||||
regInfo VF[2]; // Holds cycle info for Fd, VF[0] = Upper Instruction, VF[1] = Lower Instruction
|
||||
u8 VFreg[2]; // Index of the VF reg
|
||||
u8 VI; // Holds cycle info for Id
|
||||
u8 VIreg; // Index of the VI reg
|
||||
u8 q; // Holds cycle info for Q reg
|
||||
u8 p; // Holds cycle info for P reg
|
||||
u8 r; // Holds cycle info for R reg (Will never cause stalls, but useful to know if R is modified)
|
||||
u8 xgkick; // Holds the cycle info for XGkick
|
||||
regCycleInfo VF[2]; // Holds cycle info for Fd, VF[0] = Upper Instruction, VF[1] = Lower Instruction
|
||||
u8 VFreg[2]; // Index of the VF reg
|
||||
u8 VI; // Holds cycle info for Id
|
||||
u8 VIreg; // Index of the VI reg
|
||||
u8 q; // Holds cycle info for Q reg
|
||||
u8 p; // Holds cycle info for P reg
|
||||
u8 r; // Holds cycle info for R reg (Will never cause stalls, but useful to know if R is modified)
|
||||
u8 xgkick; // Holds the cycle info for XGkick
|
||||
};
|
||||
|
||||
struct microVFreg
|
||||
|
||||
@@ -644,22 +644,8 @@ void mVUcustomSearch()
|
||||
xMOVAPS (xmm2, ptr32[arg1reg + 0x50]);
|
||||
xPCMP.EQD(xmm2, ptr32[arg2reg + 0x50]);
|
||||
xPAND (xmm1, xmm2);
|
||||
xPAND (xmm0, xmm1);
|
||||
|
||||
xMOVAPS (xmm2, ptr32[arg1reg + 0x60]);
|
||||
xPCMP.EQD(xmm2, ptr32[arg2reg + 0x60]);
|
||||
xMOVAPS (xmm3, ptr32[arg1reg + 0x70]);
|
||||
xPCMP.EQD(xmm3, ptr32[arg2reg + 0x70]);
|
||||
xPAND (xmm2, xmm3);
|
||||
|
||||
xMOVAPS (xmm3, ptr32[arg1reg + 0x80]);
|
||||
xPCMP.EQD(xmm3, ptr32[arg2reg + 0x80]);
|
||||
xMOVAPS (xmm4, ptr32[arg1reg + 0x90]);
|
||||
xPCMP.EQD(xmm4, ptr32[arg2reg + 0x90]);
|
||||
xPAND (xmm3, xmm4);
|
||||
|
||||
xPAND (xmm0, xmm1);
|
||||
xPAND (xmm2, xmm3);
|
||||
xPAND (xmm0, xmm2);
|
||||
xMOVMSKPS(eax, xmm0);
|
||||
xXOR(eax, 0xf);
|
||||
|
||||
@@ -675,20 +661,11 @@ void mVUcustomSearch()
|
||||
xForwardJNZ8 exitPoint;
|
||||
|
||||
xVMOVUPS(ymm0, ptr[arg1reg + 0x20]);
|
||||
xVPCMP.EQD(ymm0, ymm0, ptr[arg2reg + 0x20]);
|
||||
|
||||
xVMOVUPS(ymm1, ptr[arg1reg + 0x40]);
|
||||
xVPCMP.EQD(ymm0, ymm0, ptr[arg2reg + 0x20]);
|
||||
xVPCMP.EQD(ymm1, ymm1, ptr[arg2reg + 0x40]);
|
||||
|
||||
xVMOVUPS(ymm2, ptr[arg1reg + 0x60]);
|
||||
xVPCMP.EQD(ymm2, ymm2, ptr[arg2reg + 0x60]);
|
||||
xVPAND(ymm0, ymm0, ymm1);
|
||||
|
||||
xVMOVUPS(ymm3, ptr[arg1reg + 0x80]);
|
||||
xVPCMP.EQD(ymm3, ymm3, ptr[arg2reg + 0x80]);
|
||||
xVPAND(ymm2, ymm2, ymm3);
|
||||
xVPAND(ymm0, ymm0, ymm2);
|
||||
|
||||
xVPMOVMSKB(eax, ymm0);
|
||||
xNOT(eax);
|
||||
|
||||
|
||||
@@ -30,6 +30,7 @@ typedef void (*nVifrecCall)(uptr dest, uptr src);
|
||||
#include "newVif_HashBucket.h"
|
||||
|
||||
extern void mVUmergeRegs(const xRegisterSSE& dest, const xRegisterSSE& src, int xyzw, bool modXYZW = 0);
|
||||
extern void mVUsaveReg(const xRegisterSSE& reg, xAddressVoid ptr, int xyzw, bool modXYZW);
|
||||
extern void _nVifUnpack (int idx, const u8* data, uint mode, bool isFill);
|
||||
extern void dVifReserve (int idx);
|
||||
extern void dVifReset (int idx);
|
||||
|
||||
@@ -95,6 +95,7 @@ __fi void VifUnpackSSE_Dynarec::SetMasks(int cS) const
|
||||
xMOVAPS(xmmRow, ptr128[&vif.MaskRow]);
|
||||
MSKPATH3_LOG("Moving row");
|
||||
}
|
||||
|
||||
if (m3 && doMask)
|
||||
{
|
||||
MSKPATH3_LOG("Merging Cols");
|
||||
@@ -111,7 +112,7 @@ void VifUnpackSSE_Dynarec::doMaskWrite(const xRegisterSSE& regX) const
|
||||
{
|
||||
pxAssertDev(regX.Id <= 1, "Reg Overflow! XMM2 thru XMM6 are reserved for masking.");
|
||||
|
||||
int cc = std::min(vCL, 3);
|
||||
const int cc = std::min(vCL, 3);
|
||||
u32 m0 = (vB.mask >> (cc * 8)) & 0xff; //The actual mask example 0xE4 (protect, col, row, clear)
|
||||
u32 m3 = ((m0 & 0xaa) >> 1) & ~m0; //all the upper bits (cols shifted right) cancelling out any write protects 0x10
|
||||
u32 m2 = (m0 & 0x55) & (~m0 >> 1); // all the lower bits (rows)cancelling out any write protects 0x04
|
||||
@@ -123,17 +124,14 @@ void VifUnpackSSE_Dynarec::doMaskWrite(const xRegisterSSE& regX) const
|
||||
|
||||
if (doMask && m2) // Merge MaskRow
|
||||
{
|
||||
mergeVectors(regX, xmmRow, xmmTemp, m2);
|
||||
mVUmergeRegs(regX, xmmRow, m2);
|
||||
}
|
||||
|
||||
if (doMask && m3) // Merge MaskCol
|
||||
{
|
||||
mergeVectors(regX, xRegisterSSE(xmmCol0.Id + cc), xmmTemp, m3);
|
||||
}
|
||||
if (doMask && m4) // Merge Write Protect
|
||||
{
|
||||
xMOVAPS(xmmTemp, ptr[dstIndirect]);
|
||||
mergeVectors(regX, xmmTemp, xmmTemp, m4);
|
||||
mVUmergeRegs(regX, xRegisterSSE(xmmCol0.Id + cc), m3);
|
||||
}
|
||||
|
||||
if (doMode)
|
||||
{
|
||||
u32 m5 = ~(m2 | m3 | m4) & 0xf;
|
||||
@@ -146,14 +144,14 @@ void VifUnpackSSE_Dynarec::doMaskWrite(const xRegisterSSE& regX) const
|
||||
xPXOR(xmmTemp, xmmTemp);
|
||||
if (doMode == 3)
|
||||
{
|
||||
mergeVectors(xmmRow, regX, xmmTemp, m5);
|
||||
mVUmergeRegs(xmmRow, regX, m5);
|
||||
}
|
||||
else
|
||||
{
|
||||
mergeVectors(xmmTemp, xmmRow, xmmTemp, m5);
|
||||
mVUmergeRegs(xmmTemp, xmmRow, m5);
|
||||
xPADD.D(regX, xmmTemp);
|
||||
if (doMode == 2)
|
||||
mergeVectors(xmmRow, regX, xmmTemp, m5);
|
||||
mVUmergeRegs(xmmRow, regX, m5);
|
||||
}
|
||||
}
|
||||
else
|
||||
@@ -170,7 +168,11 @@ void VifUnpackSSE_Dynarec::doMaskWrite(const xRegisterSSE& regX) const
|
||||
}
|
||||
}
|
||||
}
|
||||
xMOVAPS(ptr32[dstIndirect], regX);
|
||||
|
||||
if (doMask && m4) // Merge Write Protect
|
||||
mVUsaveReg(regX, ptr32[dstIndirect], m4 ^ 0xf, false);
|
||||
else
|
||||
xMOVAPS(ptr32[dstIndirect], regX);
|
||||
}
|
||||
|
||||
void VifUnpackSSE_Dynarec::writeBackRow() const
|
||||
@@ -276,16 +278,17 @@ void VifUnpackSSE_Dynarec::CompileRoutine()
|
||||
// Value passed determines # of col regs we need to load
|
||||
SetMasks(isFill ? blockSize : cycleSize);
|
||||
|
||||
// Need a zero register for V2_32/V3 unpacks.
|
||||
if ((upkNum >= 8 && upkNum <= 10) || upkNum == 4)
|
||||
xXOR.PS(zeroReg, zeroReg);
|
||||
|
||||
while (vNum)
|
||||
{
|
||||
|
||||
|
||||
ShiftDisplacementWindow(dstIndirect, arg1reg);
|
||||
|
||||
if (UnpkNoOfIterations == 0)
|
||||
ShiftDisplacementWindow(srcIndirect, arg2reg); //Don't need to do this otherwise as we arent reading the source.
|
||||
|
||||
|
||||
if (vCL < cycleSize)
|
||||
{
|
||||
ModUnpack(upkNum, false);
|
||||
@@ -303,9 +306,14 @@ void VifUnpackSSE_Dynarec::CompileRoutine()
|
||||
}
|
||||
else if (isFill)
|
||||
{
|
||||
//Filling doesn't need anything fancy, it's pretty much a normal write, just doesnt increment the source.
|
||||
//DevCon.WriteLn("filling mode!");
|
||||
xUnpack(upkNum);
|
||||
// Filling doesn't need anything fancy, it's pretty much a normal write, just doesnt increment the source.
|
||||
// If all vectors read a row or column or are masked, we don't need to process the source at all.
|
||||
const int cc = std::min(vCL, 3);
|
||||
u32 m0 = (vB.mask >> (cc * 8)) & 0xff;
|
||||
m0 = (m0 >> 1) | m0;
|
||||
|
||||
if ((m0 & 0x55) != 0x55)
|
||||
xUnpack(upkNum);
|
||||
xMovDest();
|
||||
|
||||
dstIndirect += 16;
|
||||
@@ -323,6 +331,7 @@ void VifUnpackSSE_Dynarec::CompileRoutine()
|
||||
|
||||
if (doMode >= 2)
|
||||
writeBackRow();
|
||||
|
||||
xRET();
|
||||
}
|
||||
|
||||
@@ -361,7 +370,7 @@ _vifT __fi nVifBlock* dVifCompile(nVifBlock& block, bool isFill)
|
||||
|
||||
VifUnpackSSE_Dynarec(v, block).CompileRoutine();
|
||||
|
||||
Perf::vif.map((uptr)v.recWritePtr, xGetPtr() - v.recWritePtr, block.upkType /* FIXME ideally a key*/);
|
||||
Perf::vif.RegisterPC(v.recWritePtr, xGetPtr() - v.recWritePtr, block.upkType /* FIXME ideally a key*/);
|
||||
v.recWritePtr = xGetPtr();
|
||||
|
||||
return █
|
||||
|
||||
@@ -15,6 +15,7 @@
|
||||
|
||||
#include "PrecompiledHeader.h"
|
||||
#include "newVif_UnpackSSE.h"
|
||||
#include "common/Perf.h"
|
||||
#include "fmt/core.h"
|
||||
|
||||
#define xMOV8(regX, loc) xMOVSSZX(regX, loc)
|
||||
@@ -23,23 +24,9 @@
|
||||
#define xMOV64(regX, loc) xMOVUPS (regX, loc)
|
||||
#define xMOV128(regX, loc) xMOVUPS (regX, loc)
|
||||
|
||||
alignas(16) static const u32 SSEXYZWMask[4][4] =
|
||||
{
|
||||
{0xffffffff, 0xffffffff, 0xffffffff, 0x00000000},
|
||||
{0xffffffff, 0xffffffff, 0x00000000, 0xffffffff},
|
||||
{0xffffffff, 0x00000000, 0xffffffff, 0xffffffff},
|
||||
{0x00000000, 0xffffffff, 0xffffffff, 0xffffffff}
|
||||
};
|
||||
|
||||
//alignas(__pagesize) static u8 nVifUpkExec[__pagesize*4];
|
||||
static RecompiledCodeReserve* nVifUpkExec = NULL;
|
||||
|
||||
// Merges xmm vectors without modifying source reg
|
||||
void mergeVectors(xRegisterSSE dest, xRegisterSSE src, xRegisterSSE temp, int xyzw)
|
||||
{
|
||||
mVUmergeRegs(dest, src, xyzw);
|
||||
}
|
||||
|
||||
// =====================================================================================================
|
||||
// VifUnpackSSE_Base Section
|
||||
// =====================================================================================================
|
||||
@@ -51,6 +38,7 @@ VifUnpackSSE_Base::VifUnpackSSE_Base()
|
||||
, IsAligned(0)
|
||||
, dstIndirect(arg1reg)
|
||||
, srcIndirect(arg2reg)
|
||||
, zeroReg(xmm2)
|
||||
, workReg(xmm1)
|
||||
, destReg(xmm0)
|
||||
{
|
||||
@@ -82,7 +70,6 @@ void VifUnpackSSE_Base::xPMOVXX16(const xRegisterSSE& regX) const
|
||||
|
||||
void VifUnpackSSE_Base::xUPK_S_32() const
|
||||
{
|
||||
|
||||
switch (UnpkLoopIteration)
|
||||
{
|
||||
case 0:
|
||||
@@ -103,7 +90,6 @@ void VifUnpackSSE_Base::xUPK_S_32() const
|
||||
|
||||
void VifUnpackSSE_Base::xUPK_S_16() const
|
||||
{
|
||||
|
||||
switch (UnpkLoopIteration)
|
||||
{
|
||||
case 0:
|
||||
@@ -124,7 +110,6 @@ void VifUnpackSSE_Base::xUPK_S_16() const
|
||||
|
||||
void VifUnpackSSE_Base::xUPK_S_8() const
|
||||
{
|
||||
|
||||
switch (UnpkLoopIteration)
|
||||
{
|
||||
case 0:
|
||||
@@ -150,25 +135,23 @@ void VifUnpackSSE_Base::xUPK_S_8() const
|
||||
|
||||
void VifUnpackSSE_Base::xUPK_V2_32() const
|
||||
{
|
||||
|
||||
if (UnpkLoopIteration == 0)
|
||||
{
|
||||
xMOV128(workReg, ptr32[srcIndirect]);
|
||||
xPSHUF.D(destReg, workReg, 0x44); //v1v0v1v0
|
||||
if (IsAligned)
|
||||
xAND.PS(destReg, ptr128[SSEXYZWMask[0]]); //zero last word - tested on ps2
|
||||
xBLEND.PS(destReg, zeroReg, 0x8); //zero last word - tested on ps2
|
||||
}
|
||||
else
|
||||
{
|
||||
xPSHUF.D(destReg, workReg, 0xEE); //v3v2v3v2
|
||||
if (IsAligned)
|
||||
xAND.PS(destReg, ptr128[SSEXYZWMask[0]]); //zero last word - tested on ps2
|
||||
xBLEND.PS(destReg, zeroReg, 0x8); //zero last word - tested on ps2
|
||||
}
|
||||
}
|
||||
|
||||
void VifUnpackSSE_Base::xUPK_V2_16() const
|
||||
{
|
||||
|
||||
if (UnpkLoopIteration == 0)
|
||||
{
|
||||
xPMOVXX16(workReg);
|
||||
@@ -182,7 +165,6 @@ void VifUnpackSSE_Base::xUPK_V2_16() const
|
||||
|
||||
void VifUnpackSSE_Base::xUPK_V2_8() const
|
||||
{
|
||||
|
||||
if (UnpkLoopIteration == 0)
|
||||
{
|
||||
xPMOVXX8(workReg);
|
||||
@@ -196,15 +178,13 @@ void VifUnpackSSE_Base::xUPK_V2_8() const
|
||||
|
||||
void VifUnpackSSE_Base::xUPK_V3_32() const
|
||||
{
|
||||
|
||||
xMOV128(destReg, ptr128[srcIndirect]);
|
||||
if (UnpkLoopIteration != IsAligned)
|
||||
xAND.PS(destReg, ptr128[SSEXYZWMask[0]]);
|
||||
xBLEND.PS(destReg, zeroReg, 0x8); //zero last word - tested on ps2
|
||||
}
|
||||
|
||||
void VifUnpackSSE_Base::xUPK_V3_16() const
|
||||
{
|
||||
|
||||
xPMOVXX16(destReg);
|
||||
|
||||
//With V3-16, it takes the first vector from the next position as the W vector
|
||||
@@ -214,17 +194,14 @@ void VifUnpackSSE_Base::xUPK_V3_16() const
|
||||
int result = (((UnpkLoopIteration / 4) + 1 + (4 - IsAligned)) & 0x3);
|
||||
|
||||
if ((UnpkLoopIteration & 0x1) == 0 && result == 0)
|
||||
{
|
||||
xAND.PS(destReg, ptr128[SSEXYZWMask[0]]); //zero last word on QW boundary if whole 32bit word is used - tested on ps2
|
||||
}
|
||||
xBLEND.PS(destReg, zeroReg, 0x8); //zero last word - tested on ps2
|
||||
}
|
||||
|
||||
void VifUnpackSSE_Base::xUPK_V3_8() const
|
||||
{
|
||||
|
||||
xPMOVXX8(destReg);
|
||||
if (UnpkLoopIteration != IsAligned)
|
||||
xAND.PS(destReg, ptr128[SSEXYZWMask[0]]);
|
||||
xBLEND.PS(destReg, zeroReg, 0x8); //zero last word - tested on ps2
|
||||
}
|
||||
|
||||
void VifUnpackSSE_Base::xUPK_V4_32() const
|
||||
@@ -244,7 +221,6 @@ void VifUnpackSSE_Base::xUPK_V4_8() const
|
||||
|
||||
void VifUnpackSSE_Base::xUPK_V4_5() const
|
||||
{
|
||||
|
||||
xMOV16 (workReg, ptr32[srcIndirect]);
|
||||
xPSHUF.D (workReg, workReg, _v0);
|
||||
xPSLL.D (workReg, 3); // ABG|R5.000
|
||||
@@ -283,7 +259,6 @@ void VifUnpackSSE_Base::xUnpack(int upknum) const
|
||||
case 14: xUPK_V4_8(); break;
|
||||
case 15: xUPK_V4_5(); break;
|
||||
|
||||
|
||||
case 3:
|
||||
case 7:
|
||||
case 11:
|
||||
@@ -318,7 +293,6 @@ void VifUnpackSSE_Simple::doMaskWrite(const xRegisterSSE& regX) const
|
||||
// ecx = dest, edx = src
|
||||
static void nVifGen(int usn, int mask, int curCycle)
|
||||
{
|
||||
|
||||
int usnpart = usn * 2 * 16;
|
||||
int maskpart = mask * 16;
|
||||
|
||||
@@ -346,7 +320,6 @@ void VifUnpackSSE_Init()
|
||||
DevCon.WriteLn("Generating SSE-optimized unpacking functions for VIF interpreters...");
|
||||
|
||||
nVifUpkExec = new RecompiledCodeReserve("VIF SSE-optimized Unpacking Functions");
|
||||
nVifUpkExec->SetProfilerName("iVIF-SSE");
|
||||
nVifUpkExec->Assign(GetVmMemory().CodeMemory(), HostMemoryMap::VIFUnpackRecOffset, _1mb);
|
||||
xSetPtr(*nVifUpkExec);
|
||||
|
||||
@@ -365,6 +338,8 @@ void VifUnpackSSE_Init()
|
||||
nVifUpkExec->GetPtr(),
|
||||
(uint)(xGetPtr() - nVifUpkExec->GetPtr())
|
||||
);
|
||||
|
||||
Perf::any.Register(nVifUpkExec->GetPtr(), xGetPtr() - nVifUpkExec->GetPtr(), "VIF Unpack");
|
||||
}
|
||||
|
||||
void VifUnpackSSE_Destroy()
|
||||
|
||||
@@ -23,8 +23,6 @@
|
||||
|
||||
using namespace x86Emitter;
|
||||
|
||||
extern void mergeVectors(xRegisterSSE dest, xRegisterSSE src, xRegisterSSE temp, int xyzw);
|
||||
|
||||
// --------------------------------------------------------------------------------------
|
||||
// VifUnpackSSE_Base
|
||||
// --------------------------------------------------------------------------------------
|
||||
@@ -41,6 +39,7 @@ public:
|
||||
protected:
|
||||
xAddressVoid dstIndirect;
|
||||
xAddressVoid srcIndirect;
|
||||
xRegisterSSE zeroReg;
|
||||
xRegisterSSE workReg;
|
||||
xRegisterSSE destReg;
|
||||
|
||||
|
||||
Reference in New Issue
Block a user