Improve fast sse matrix multiply

This commit is contained in:
Henrik Rydgard committed 2014-12-06 00:28:29 +01:00
1 parent 524fceea5c
commit 2b6ed975e2
2 files changed
+8 -6

No files matched your search

+5 -5
View File
@@ -8,12 +8,12 @@
void fast_matrix_mul_4x4_sse(float *dest, const float *a, const float *b) {
int i;
for (i = 0; i < 16; i += 4) {
__m128 a_col_1 = _mm_loadu_ps(a);
__m128 a_col_2 = _mm_loadu_ps(&a[4]);
__m128 a_col_3 = _mm_loadu_ps(&a[8]);
__m128 a_col_4 = _mm_loadu_ps(&a[12]);
__m128 a_col_1 = _mm_loadu_ps(a);
__m128 a_col_2 = _mm_loadu_ps(&a[4]);
__m128 a_col_3 = _mm_loadu_ps(&a[8]);
__m128 a_col_4 = _mm_loadu_ps(&a[12]);
for (i = 0; i < 16; i += 4) {
__m128 r_col = _mm_mul_ps(a_col_1, _mm_set1_ps(b[i]));
r_col = _mm_add_ps(r_col, _mm_mul_ps(a_col_2, _mm_set1_ps(b[i + 1])));
r_col = _mm_add_ps(r_col, _mm_mul_ps(a_col_3, _mm_set1_ps(b[i + 2])));
+3 -1
View File
@@ -754,7 +754,9 @@
<ClCompile Include="math\math_util.cpp" />
<ClCompile Include="math\fast\fast_math.c" />
<ClCompile Include="math\fast\fast_matrix.c" />
<ClCompile Include="math\fast\fast_matrix_sse.c" />
<ClCompile Include="math\fast\fast_matrix_sse.c">
<AssemblerOutput Condition="'$(Configuration)|$(Platform)'=='Release|x64'">AssemblyAndSourceCode</AssemblerOutput>
</ClCompile>
<ClCompile Include="midi\midi_input.cpp" />
<ClCompile Include="net\http_client.cpp" />
<ClCompile Include="net\resolve.cpp" />