svd3x3_sse.cpp 4.0 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108
  1. // This file is part of libigl, a simple c++ geometry processing library.
  2. //
  3. // Copyright (C) 2013 Alec Jacobson <alecjacobson@gmail.com>
  4. //
  5. // This Source Code Form is subject to the terms of the Mozilla Public License
  6. // v. 2.0. If a copy of the MPL was not distributed with this file, You can
  7. // obtain one at http://mozilla.org/MPL/2.0/.
  8. #ifdef __SSE__
  9. #include "svd3x3_sse.h"
  10. #include <cmath>
  11. #include <algorithm>
  12. #undef USE_SCALAR_IMPLEMENTATION
  13. #define USE_SSE_IMPLEMENTATION
  14. #undef USE_AVX_IMPLEMENTATION
  15. #define COMPUTE_U_AS_MATRIX
  16. #define COMPUTE_V_AS_MATRIX
  17. #include "Singular_Value_Decomposition_Preamble.hpp"
  18. // disable runtime asserts on xor eax,eax type of stuff (doesn't always work,
  19. // disable explicitly in compiler settings)
  20. #pragma runtime_checks( "u", off )
  21. template<typename T>
  22. IGL_INLINE void igl::svd3x3_sse(
  23. const Eigen::Matrix<T, 3*4, 3>& A,
  24. Eigen::Matrix<T, 3*4, 3> &U,
  25. Eigen::Matrix<T, 3*4, 1> &S,
  26. Eigen::Matrix<T, 3*4, 3>&V)
  27. {
  28. // this code assumes USE_SSE_IMPLEMENTATION is defined
  29. float Ashuffle[9][4], Ushuffle[9][4], Vshuffle[9][4], Sshuffle[3][4];
  30. for (int i=0; i<3; i++)
  31. {
  32. for (int j=0; j<3; j++)
  33. {
  34. for (int k=0; k<4; k++)
  35. {
  36. Ashuffle[i + j*3][k] = A(i + 3*k, j);
  37. }
  38. }
  39. }
  40. #include "Singular_Value_Decomposition_Kernel_Declarations.hpp"
  41. ENABLE_SSE_IMPLEMENTATION(Va11=_mm_loadu_ps(Ashuffle[0]);)
  42. ENABLE_SSE_IMPLEMENTATION(Va21=_mm_loadu_ps(Ashuffle[1]);)
  43. ENABLE_SSE_IMPLEMENTATION(Va31=_mm_loadu_ps(Ashuffle[2]);)
  44. ENABLE_SSE_IMPLEMENTATION(Va12=_mm_loadu_ps(Ashuffle[3]);)
  45. ENABLE_SSE_IMPLEMENTATION(Va22=_mm_loadu_ps(Ashuffle[4]);)
  46. ENABLE_SSE_IMPLEMENTATION(Va32=_mm_loadu_ps(Ashuffle[5]);)
  47. ENABLE_SSE_IMPLEMENTATION(Va13=_mm_loadu_ps(Ashuffle[6]);)
  48. ENABLE_SSE_IMPLEMENTATION(Va23=_mm_loadu_ps(Ashuffle[7]);)
  49. ENABLE_SSE_IMPLEMENTATION(Va33=_mm_loadu_ps(Ashuffle[8]);)
  50. #include "Singular_Value_Decomposition_Main_Kernel_Body.hpp"
  51. ENABLE_SSE_IMPLEMENTATION(_mm_storeu_ps(Ushuffle[0],Vu11);)
  52. ENABLE_SSE_IMPLEMENTATION(_mm_storeu_ps(Ushuffle[1],Vu21);)
  53. ENABLE_SSE_IMPLEMENTATION(_mm_storeu_ps(Ushuffle[2],Vu31);)
  54. ENABLE_SSE_IMPLEMENTATION(_mm_storeu_ps(Ushuffle[3],Vu12);)
  55. ENABLE_SSE_IMPLEMENTATION(_mm_storeu_ps(Ushuffle[4],Vu22);)
  56. ENABLE_SSE_IMPLEMENTATION(_mm_storeu_ps(Ushuffle[5],Vu32);)
  57. ENABLE_SSE_IMPLEMENTATION(_mm_storeu_ps(Ushuffle[6],Vu13);)
  58. ENABLE_SSE_IMPLEMENTATION(_mm_storeu_ps(Ushuffle[7],Vu23);)
  59. ENABLE_SSE_IMPLEMENTATION(_mm_storeu_ps(Ushuffle[8],Vu33);)
  60. ENABLE_SSE_IMPLEMENTATION(_mm_storeu_ps(Vshuffle[0],Vv11);)
  61. ENABLE_SSE_IMPLEMENTATION(_mm_storeu_ps(Vshuffle[1],Vv21);)
  62. ENABLE_SSE_IMPLEMENTATION(_mm_storeu_ps(Vshuffle[2],Vv31);)
  63. ENABLE_SSE_IMPLEMENTATION(_mm_storeu_ps(Vshuffle[3],Vv12);)
  64. ENABLE_SSE_IMPLEMENTATION(_mm_storeu_ps(Vshuffle[4],Vv22);)
  65. ENABLE_SSE_IMPLEMENTATION(_mm_storeu_ps(Vshuffle[5],Vv32);)
  66. ENABLE_SSE_IMPLEMENTATION(_mm_storeu_ps(Vshuffle[6],Vv13);)
  67. ENABLE_SSE_IMPLEMENTATION(_mm_storeu_ps(Vshuffle[7],Vv23);)
  68. ENABLE_SSE_IMPLEMENTATION(_mm_storeu_ps(Vshuffle[8],Vv33);)
  69. ENABLE_SSE_IMPLEMENTATION(_mm_storeu_ps(Sshuffle[0],Va11);)
  70. ENABLE_SSE_IMPLEMENTATION(_mm_storeu_ps(Sshuffle[1],Va22);)
  71. ENABLE_SSE_IMPLEMENTATION(_mm_storeu_ps(Sshuffle[2],Va33);)
  72. for (int i=0; i<3; i++)
  73. {
  74. for (int j=0; j<3; j++)
  75. {
  76. for (int k=0; k<4; k++)
  77. {
  78. U(i + 3*k, j) = Ushuffle[i + j*3][k];
  79. V(i + 3*k, j) = Vshuffle[i + j*3][k];
  80. }
  81. }
  82. }
  83. for (int i=0; i<3; i++)
  84. {
  85. for (int k=0; k<4; k++)
  86. {
  87. S(i + 3*k, 0) = Sshuffle[i][k];
  88. }
  89. }
  90. }
  91. #pragma runtime_checks( "u", restore )
  92. // forced instantiation
  93. template void igl::svd3x3_sse(const Eigen::Matrix<float, 3*4, 3>& A, Eigen::Matrix<float, 3*4, 3> &U, Eigen::Matrix<float, 3*4, 1> &S, Eigen::Matrix<float, 3*4, 3>&V);
  94. //// doesn't even make sense with double because the wunder-SVD code is only single precision anyway...
  95. //template void wunderSVD3x3_SSE<float>(Eigen::Matrix<float, 12, 3, 0, 12, 3> const&, Eigen::Matrix<float, 12, 3, 0, 12, 3>&, Eigen::Matrix<float, 12, 1, 0, 12, 1>&, Eigen::Matrix<float, 12, 3, 0, 12, 3>&);
  96. #endif