intrinPortable.h 6.0 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293
  1. /*
  2. Copyright (c) 2018 tevador
  3. This file is part of RandomX.
  4. RandomX is free software: you can redistribute it and/or modify
  5. it under the terms of the GNU General Public License as published by
  6. the Free Software Foundation, either version 3 of the License, or
  7. (at your option) any later version.
  8. RandomX is distributed in the hope that it will be useful,
  9. but WITHOUT ANY WARRANTY; without even the implied warranty of
  10. MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
  11. GNU General Public License for more details.
  12. You should have received a copy of the GNU General Public License
  13. along with RandomX. If not, see<http://www.gnu.org/licenses/>.
  14. */
  15. #pragma once
  16. #include <cstdint>
  17. constexpr int32_t unsigned32ToSigned2sCompl(uint32_t x) {
  18. return (-1 == ~0) ? (int32_t)x : (x > INT32_MAX ? (-(int32_t)(UINT32_MAX - x) - 1) : (int32_t)x);
  19. }
  20. constexpr int64_t unsigned64ToSigned2sCompl(uint64_t x) {
  21. return (-1 == ~0) ? (int64_t)x : (x > INT64_MAX ? (-(int64_t)(UINT64_MAX - x) - 1) : (int64_t)x);
  22. }
  23. constexpr uint64_t signExtend2sCompl(uint32_t x) {
  24. return (-1 == ~0) ? (int64_t)(int32_t)(x) : (x > INT32_MAX ? (x | 0xffffffff00000000ULL) : (uint64_t)x);
  25. }
  26. #if defined(_MSC_VER)
  27. #if defined(_M_X64) || (defined(_M_IX86_FP) && _M_IX86_FP == 2)
  28. #define __SSE2__ 1
  29. #endif
  30. #endif
  31. #ifdef __SSE2__
  32. #ifdef __GNUC__
  33. #include <x86intrin.h>
  34. #else
  35. #include <intrin.h>
  36. #endif
  37. inline __m128d _mm_abs(__m128d xd) {
  38. const __m128d absmask = _mm_castsi128_pd(_mm_set1_epi64x(~(1LL << 63)));
  39. return _mm_and_pd(xd, absmask);
  40. }
  41. #define PREFETCHNTA(x) _mm_prefetch((const char *)(x), _MM_HINT_NTA)
  42. #else
  43. #include <cstdint>
  44. #include <stdexcept>
  45. #include <cstdlib>
  46. #define _mm_malloc(a,b) malloc(a)
  47. #define _mm_free(a) free(a)
  48. #define PREFETCHNTA(x)
  49. typedef union {
  50. uint64_t u64[2];
  51. uint32_t u32[4];
  52. uint16_t u16[8];
  53. uint8_t u8[16];
  54. } __m128i;
  55. typedef union {
  56. struct {
  57. double lo;
  58. double hi;
  59. };
  60. __m128i i;
  61. } __m128d;
  62. inline __m128d _mm_load_pd(const double* pd) {
  63. __m128d x;
  64. x.i.u64[0] = load64(pd + 0);
  65. x.i.u64[1] = load64(pd + 1);
  66. return x;
  67. }
  68. inline void _mm_store_pd(double* mem_addr, __m128d a) {
  69. store64(mem_addr + 0, a.i.u64[0]);
  70. store64(mem_addr + 1, a.i.u64[1]);
  71. }
  72. inline __m128d _mm_shuffle_pd(__m128d a, __m128d b, int imm8) {
  73. __m128d x;
  74. x.lo = (imm8 & 1) ? a.hi : a.lo;
  75. x.hi = (imm8 & 2) ? b.hi : b.lo;
  76. return x;
  77. }
  78. inline __m128d _mm_add_pd(__m128d a, __m128d b) {
  79. __m128d x;
  80. x.lo = a.lo + b.lo;
  81. x.hi = a.hi + b.hi;
  82. return x;
  83. }
  84. inline __m128d _mm_sub_pd(__m128d a, __m128d b) {
  85. __m128d x;
  86. x.lo = a.lo - b.lo;
  87. x.hi = a.hi - b.hi;
  88. return x;
  89. }
  90. inline __m128d _mm_mul_pd(__m128d a, __m128d b) {
  91. __m128d x;
  92. x.lo = a.lo * b.lo;
  93. x.hi = a.hi * b.hi;
  94. return x;
  95. }
  96. inline __m128d _mm_div_pd(__m128d a, __m128d b) {
  97. __m128d x;
  98. x.lo = a.lo / b.lo;
  99. x.hi = a.hi / b.hi;
  100. return x;
  101. }
  102. inline __m128d _mm_sqrt_pd(__m128d a) {
  103. __m128d x;
  104. x.lo = sqrt(a.lo);
  105. x.hi = sqrt(a.hi);
  106. return x;
  107. }
  108. inline __m128i _mm_set1_epi64x(uint64_t a) {
  109. __m128i x;
  110. x.u64[0] = a;
  111. x.u64[1] = a;
  112. return x;
  113. }
  114. inline __m128d _mm_castsi128_pd(__m128i a) {
  115. __m128d x;
  116. x.i = a;
  117. return x;
  118. }
  119. inline __m128d _mm_abs(__m128d xd) {
  120. xd.lo = std::abs(xd.lo);
  121. xd.hi = std::abs(xd.hi);
  122. return xd;
  123. }
  124. inline __m128d _mm_xor_pd(__m128d a, __m128d b) {
  125. __m128d x;
  126. x.i.u64[0] = a.i.u64[0] ^ b.i.u64[0];
  127. x.i.u64[1] = a.i.u64[1] ^ b.i.u64[1];
  128. return x;
  129. }
  130. inline __m128d _mm_set_pd(double e1, double e0) {
  131. __m128d x;
  132. x.lo = e0;
  133. x.hi = e1;
  134. return x;
  135. }
  136. inline __m128d _mm_max_pd(__m128d a, __m128d b) {
  137. __m128d x;
  138. x.lo = a.lo > b.lo ? a.lo : b.lo;
  139. x.hi = a.hi > b.hi ? a.hi : b.hi;
  140. return x;
  141. }
  142. inline __m128d _mm_cvtepi32_pd(__m128i a) {
  143. __m128d x;
  144. x.lo = (double)unsigned32ToSigned2sCompl(a.u32[0]);
  145. x.hi = (double)unsigned32ToSigned2sCompl(a.u32[1]);
  146. return x;
  147. }
  148. static const char* platformError = "Platform doesn't support hardware AES";
  149. inline __m128i _mm_aeskeygenassist_si128(__m128i key, uint8_t rcon) {
  150. throw std::runtime_error(platformError);
  151. }
  152. inline __m128i _mm_aesenc_si128(__m128i v, __m128i rkey) {
  153. throw std::runtime_error(platformError);
  154. }
  155. inline __m128i _mm_aesdec_si128(__m128i v, __m128i rkey) {
  156. throw std::runtime_error(platformError);
  157. }
  158. inline int _mm_cvtsi128_si32(__m128i v) {
  159. return v.u32[0];
  160. }
  161. inline __m128i _mm_cvtsi32_si128(int si32) {
  162. __m128i v;
  163. v.u32[0] = si32;
  164. v.u32[1] = 0;
  165. v.u32[2] = 0;
  166. v.u32[3] = 0;
  167. return v;
  168. }
  169. inline __m128i _mm_set_epi64x(int64_t _I1, int64_t _I0) {
  170. __m128i v;
  171. v.u64[0] = _I0;
  172. v.u64[1] = _I1;
  173. return v;
  174. }
  175. inline __m128i _mm_set_epi32(int _I3, int _I2, int _I1, int _I0) {
  176. __m128i v;
  177. v.u32[0] = _I0;
  178. v.u32[1] = _I1;
  179. v.u32[2] = _I2;
  180. v.u32[3] = _I3;
  181. return v;
  182. };
  183. inline __m128i _mm_xor_si128(__m128i _A, __m128i _B) {
  184. __m128i c;
  185. c.u32[0] = _A.u32[0] ^ _B.u32[0];
  186. c.u32[1] = _A.u32[1] ^ _B.u32[1];
  187. c.u32[2] = _A.u32[2] ^ _B.u32[2];
  188. c.u32[3] = _A.u32[3] ^ _B.u32[3];
  189. return c;
  190. }
  191. inline __m128i _mm_shuffle_epi32(__m128i _A, int _Imm) {
  192. __m128i c;
  193. c.u32[0] = _A.u32[_Imm & 3];
  194. c.u32[1] = _A.u32[(_Imm >> 2) & 3];
  195. c.u32[2] = _A.u32[(_Imm >> 4) & 3];
  196. c.u32[3] = _A.u32[(_Imm >> 6) & 3];
  197. return c;
  198. }
  199. inline __m128i _mm_load_si128(__m128i const*_P) {
  200. return *_P;
  201. }
  202. inline void _mm_store_si128(__m128i *_P, __m128i _B) {
  203. *_P = _B;
  204. }
  205. inline __m128i _mm_slli_si128(__m128i _A, int _Imm) {
  206. _Imm &= 255;
  207. if (_Imm > 15) {
  208. _A.u64[0] = 0;
  209. _A.u64[1] = 0;
  210. }
  211. else {
  212. for (int i = 15; i >= _Imm; --i) {
  213. _A.u8[i] = _A.u8[i - _Imm];
  214. }
  215. for (int i = 0; i < _Imm; ++i) {
  216. _A.u8[i] = 0;
  217. }
  218. }
  219. return _A;
  220. }
  221. inline __m128i _mm_loadl_epi64(__m128i const* mem_addr) {
  222. __m128i x;
  223. x.u64[0] = load64(mem_addr);
  224. return x;
  225. }
  226. #endif
  227. constexpr int RoundToNearest = 0;
  228. constexpr int RoundDown = 1;
  229. constexpr int RoundUp = 2;
  230. constexpr int RoundToZero = 3;
  231. inline __m128d load_cvt_i32x2(const void* addr) {
  232. __m128i ix = _mm_loadl_epi64((const __m128i*)addr);
  233. return _mm_cvtepi32_pd(ix);
  234. }
  235. double loadDoublePortable(const void* addr);
  236. uint64_t mulh(uint64_t, uint64_t);
  237. int64_t smulh(int64_t, int64_t);
  238. uint64_t rotl(uint64_t, int);
  239. uint64_t rotr(uint64_t, int);
  240. void initFpu();
  241. void setRoundMode(uint32_t);
  242. bool condition(uint32_t, uint32_t, uint32_t);