mirror of
https://github.com/hrydgard/ppsspp.git
synced 2026-10-06 19:16:21 +02:00
Improve fast sse matrix multiply
This commit is contained in:
1 parent
524fceea5c
commit
2b6ed975e2
2 files changed
+8
-6
No files matched your search
@@ -8,12 +8,12 @@
|
||||
|
||||
void fast_matrix_mul_4x4_sse(float *dest, const float *a, const float *b) {
|
||||
int i;
|
||||
for (i = 0; i < 16; i += 4) {
|
||||
__m128 a_col_1 = _mm_loadu_ps(a);
|
||||
__m128 a_col_2 = _mm_loadu_ps(&a[4]);
|
||||
__m128 a_col_3 = _mm_loadu_ps(&a[8]);
|
||||
__m128 a_col_4 = _mm_loadu_ps(&a[12]);
|
||||
__m128 a_col_1 = _mm_loadu_ps(a);
|
||||
__m128 a_col_2 = _mm_loadu_ps(&a[4]);
|
||||
__m128 a_col_3 = _mm_loadu_ps(&a[8]);
|
||||
__m128 a_col_4 = _mm_loadu_ps(&a[12]);
|
||||
|
||||
for (i = 0; i < 16; i += 4) {
|
||||
__m128 r_col = _mm_mul_ps(a_col_1, _mm_set1_ps(b[i]));
|
||||
r_col = _mm_add_ps(r_col, _mm_mul_ps(a_col_2, _mm_set1_ps(b[i + 1])));
|
||||
r_col = _mm_add_ps(r_col, _mm_mul_ps(a_col_3, _mm_set1_ps(b[i + 2])));
|
||||
|
||||
+3
-1
@@ -754,7 +754,9 @@
|
||||
<ClCompile Include="math\math_util.cpp" />
|
||||
<ClCompile Include="math\fast\fast_math.c" />
|
||||
<ClCompile Include="math\fast\fast_matrix.c" />
|
||||
<ClCompile Include="math\fast\fast_matrix_sse.c" />
|
||||
<ClCompile Include="math\fast\fast_matrix_sse.c">
|
||||
<AssemblerOutput Condition="'$(Configuration)|$(Platform)'=='Release|x64'">AssemblyAndSourceCode</AssemblerOutput>
|
||||
</ClCompile>
|
||||
<ClCompile Include="midi\midi_input.cpp" />
|
||||
<ClCompile Include="net\http_client.cpp" />
|
||||
<ClCompile Include="net\resolve.cpp" />
|
||||
|
||||
Reference in new issue
Block a user