From 2b6ed975e28de79d7dc34f225238eae631199afc Mon Sep 17 00:00:00 2001 From: Henrik Rydgard Date: Sat, 6 Dec 2014 00:28:11 +0100 Subject: [PATCH] Improve fast sse matrix multiply --- math/fast/fast_matrix_sse.c | 10 +++++----- native.vcxproj | 4 +++- 2 files changed, 8 insertions(+), 6 deletions(-) diff --git a/math/fast/fast_matrix_sse.c b/math/fast/fast_matrix_sse.c index 166f71abce..9de79ce760 100644 --- a/math/fast/fast_matrix_sse.c +++ b/math/fast/fast_matrix_sse.c @@ -8,12 +8,12 @@ void fast_matrix_mul_4x4_sse(float *dest, const float *a, const float *b) { int i; - for (i = 0; i < 16; i += 4) { - __m128 a_col_1 = _mm_loadu_ps(a); - __m128 a_col_2 = _mm_loadu_ps(&a[4]); - __m128 a_col_3 = _mm_loadu_ps(&a[8]); - __m128 a_col_4 = _mm_loadu_ps(&a[12]); + __m128 a_col_1 = _mm_loadu_ps(a); + __m128 a_col_2 = _mm_loadu_ps(&a[4]); + __m128 a_col_3 = _mm_loadu_ps(&a[8]); + __m128 a_col_4 = _mm_loadu_ps(&a[12]); + for (i = 0; i < 16; i += 4) { __m128 r_col = _mm_mul_ps(a_col_1, _mm_set1_ps(b[i])); r_col = _mm_add_ps(r_col, _mm_mul_ps(a_col_2, _mm_set1_ps(b[i + 1]))); r_col = _mm_add_ps(r_col, _mm_mul_ps(a_col_3, _mm_set1_ps(b[i + 2]))); diff --git a/native.vcxproj b/native.vcxproj index 59eaa83023..da563a52cd 100644 --- a/native.vcxproj +++ b/native.vcxproj @@ -754,7 +754,9 @@ - + + AssemblyAndSourceCode +