jit_compiler_a64_static.S 14 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576577578579580581582583584585586587
  1. # Copyright (c) 2018-2019, tevador <tevador@gmail.com>
  2. # Copyright (c) 2019, SChernykh <https://github.com/SChernykh>
  3. #
  4. # All rights reserved.
  5. #
  6. # Redistribution and use in source and binary forms, with or without
  7. # modification, are permitted provided that the following conditions are met:
  8. # * Redistributions of source code must retain the above copyright
  9. # notice, this list of conditions and the following disclaimer.
  10. # * Redistributions in binary form must reproduce the above copyright
  11. # notice, this list of conditions and the following disclaimer in the
  12. # documentation and/or other materials provided with the distribution.
  13. # * Neither the name of the copyright holder nor the
  14. # names of its contributors may be used to endorse or promote products
  15. # derived from this software without specific prior written permission.
  16. #
  17. # THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND
  18. # ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED
  19. # WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
  20. # DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
  21. # FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
  22. # DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
  23. # SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
  24. # CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
  25. # OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
  26. # OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
  27. #if defined(__APPLE__)
  28. #define DECL(x) _##x
  29. #else
  30. #define DECL(x) x
  31. #endif
  32. .arch armv8-a
  33. .text
  34. .global DECL(randomx_program_aarch64)
  35. .global DECL(randomx_program_aarch64_main_loop)
  36. .global DECL(randomx_program_aarch64_vm_instructions)
  37. .global DECL(randomx_program_aarch64_imul_rcp_literals_end)
  38. .global DECL(randomx_program_aarch64_vm_instructions_end)
  39. .global DECL(randomx_program_aarch64_cacheline_align_mask1)
  40. .global DECL(randomx_program_aarch64_cacheline_align_mask2)
  41. .global DECL(randomx_program_aarch64_update_spMix1)
  42. .global DECL(randomx_program_aarch64_vm_instructions_end_light)
  43. .global DECL(randomx_program_aarch64_light_cacheline_align_mask)
  44. .global DECL(randomx_program_aarch64_light_dataset_offset)
  45. .global DECL(randomx_init_dataset_aarch64)
  46. .global DECL(randomx_init_dataset_aarch64_end)
  47. .global DECL(randomx_calc_dataset_item_aarch64)
  48. .global DECL(randomx_calc_dataset_item_aarch64_prefetch)
  49. .global DECL(randomx_calc_dataset_item_aarch64_mix)
  50. .global DECL(randomx_calc_dataset_item_aarch64_store_result)
  51. .global DECL(randomx_calc_dataset_item_aarch64_end)
  52. #include "configuration.h"
  53. # Register allocation
  54. # x0 -> pointer to reg buffer and then literal for IMUL_RCP
  55. # x1 -> pointer to mem buffer and then to dataset
  56. # x2 -> pointer to scratchpad
  57. # x3 -> loop counter
  58. # x4 -> "r0"
  59. # x5 -> "r1"
  60. # x6 -> "r2"
  61. # x7 -> "r3"
  62. # x8 -> fpcr (reversed bits)
  63. # x9 -> mx, ma
  64. # x10 -> spMix1
  65. # x11 -> literal for IMUL_RCP
  66. # x12 -> "r4"
  67. # x13 -> "r5"
  68. # x14 -> "r6"
  69. # x15 -> "r7"
  70. # x16 -> spAddr0
  71. # x17 -> spAddr1
  72. # x18 -> temporary
  73. # x19 -> temporary
  74. # x20 -> literal for IMUL_RCP
  75. # x21 -> literal for IMUL_RCP
  76. # x22 -> literal for IMUL_RCP
  77. # x23 -> literal for IMUL_RCP
  78. # x24 -> literal for IMUL_RCP
  79. # x25 -> literal for IMUL_RCP
  80. # x26 -> literal for IMUL_RCP
  81. # x27 -> literal for IMUL_RCP
  82. # x28 -> literal for IMUL_RCP
  83. # x29 -> literal for IMUL_RCP
  84. # x30 -> literal for IMUL_RCP
  85. # v0-v15 -> store 32-bit literals
  86. # v16 -> "f0"
  87. # v17 -> "f1"
  88. # v18 -> "f2"
  89. # v19 -> "f3"
  90. # v20 -> "e0"
  91. # v21 -> "e1"
  92. # v22 -> "e2"
  93. # v23 -> "e3"
  94. # v24 -> "a0"
  95. # v25 -> "a1"
  96. # v26 -> "a2"
  97. # v27 -> "a3"
  98. # v28 -> temporary
  99. # v29 -> E 'and' mask = 0x00ffffffffffffff00ffffffffffffff
  100. # v30 -> E 'or' mask = 0x3*00000000******3*00000000******
  101. # v31 -> scale mask = 0x81f000000000000081f0000000000000
  102. .balign 4
  103. DECL(randomx_program_aarch64):
  104. # Save callee-saved registers
  105. sub sp, sp, 192
  106. stp x16, x17, [sp]
  107. stp x18, x19, [sp, 16]
  108. stp x20, x21, [sp, 32]
  109. stp x22, x23, [sp, 48]
  110. stp x24, x25, [sp, 64]
  111. stp x26, x27, [sp, 80]
  112. stp x28, x29, [sp, 96]
  113. stp x8, x30, [sp, 112]
  114. stp d8, d9, [sp, 128]
  115. stp d10, d11, [sp, 144]
  116. stp d12, d13, [sp, 160]
  117. stp d14, d15, [sp, 176]
  118. # Zero integer registers
  119. mov x4, xzr
  120. mov x5, xzr
  121. mov x6, xzr
  122. mov x7, xzr
  123. mov x12, xzr
  124. mov x13, xzr
  125. mov x14, xzr
  126. mov x15, xzr
  127. # Load ma, mx and dataset pointer
  128. ldp x9, x1, [x1]
  129. # Load initial spMix value
  130. mov x10, x9
  131. # Load group A registers
  132. ldp q24, q25, [x0, 192]
  133. ldp q26, q27, [x0, 224]
  134. # Load E 'and' mask
  135. mov x16, 0x00FFFFFFFFFFFFFF
  136. ins v29.d[0], x16
  137. ins v29.d[1], x16
  138. # Load E 'or' mask (stored in reg.f[0])
  139. ldr q30, [x0, 64]
  140. # Load scale mask
  141. mov x16, 0x80f0000000000000
  142. ins v31.d[0], x16
  143. ins v31.d[1], x16
  144. # Read fpcr
  145. mrs x8, fpcr
  146. rbit x8, x8
  147. # Save x0
  148. str x0, [sp, -16]!
  149. # Read literals
  150. ldr x0, literal_x0
  151. ldr x11, literal_x11
  152. ldr x20, literal_x20
  153. ldr x21, literal_x21
  154. ldr x22, literal_x22
  155. ldr x23, literal_x23
  156. ldr x24, literal_x24
  157. ldr x25, literal_x25
  158. ldr x26, literal_x26
  159. ldr x27, literal_x27
  160. ldr x28, literal_x28
  161. ldr x29, literal_x29
  162. ldr x30, literal_x30
  163. ldr q0, literal_v0
  164. ldr q1, literal_v1
  165. ldr q2, literal_v2
  166. ldr q3, literal_v3
  167. ldr q4, literal_v4
  168. ldr q5, literal_v5
  169. ldr q6, literal_v6
  170. ldr q7, literal_v7
  171. ldr q8, literal_v8
  172. ldr q9, literal_v9
  173. ldr q10, literal_v10
  174. ldr q11, literal_v11
  175. ldr q12, literal_v12
  176. ldr q13, literal_v13
  177. ldr q14, literal_v14
  178. ldr q15, literal_v15
  179. DECL(randomx_program_aarch64_main_loop):
  180. # spAddr0 = spMix1 & ScratchpadL3Mask64;
  181. # spAddr1 = (spMix1 >> 32) & ScratchpadL3Mask64;
  182. lsr x18, x10, 32
  183. # Actual mask will be inserted by JIT compiler
  184. and w16, w10, 1
  185. and w17, w18, 1
  186. # x16 = scratchpad + spAddr0
  187. # x17 = scratchpad + spAddr1
  188. add x16, x16, x2
  189. add x17, x17, x2
  190. # xor integer registers with scratchpad data (spAddr0)
  191. ldp x18, x19, [x16]
  192. eor x4, x4, x18
  193. eor x5, x5, x19
  194. ldp x18, x19, [x16, 16]
  195. eor x6, x6, x18
  196. eor x7, x7, x19
  197. ldp x18, x19, [x16, 32]
  198. eor x12, x12, x18
  199. eor x13, x13, x19
  200. ldp x18, x19, [x16, 48]
  201. eor x14, x14, x18
  202. eor x15, x15, x19
  203. # Load group F registers (spAddr1)
  204. ldpsw x18, x19, [x17]
  205. ins v16.d[0], x18
  206. ins v16.d[1], x19
  207. ldpsw x18, x19, [x17, 8]
  208. ins v17.d[0], x18
  209. ins v17.d[1], x19
  210. ldpsw x18, x19, [x17, 16]
  211. ins v18.d[0], x18
  212. ins v18.d[1], x19
  213. ldpsw x18, x19, [x17, 24]
  214. ins v19.d[0], x18
  215. ins v19.d[1], x19
  216. scvtf v16.2d, v16.2d
  217. scvtf v17.2d, v17.2d
  218. scvtf v18.2d, v18.2d
  219. scvtf v19.2d, v19.2d
  220. # Load group E registers (spAddr1)
  221. ldpsw x18, x19, [x17, 32]
  222. ins v20.d[0], x18
  223. ins v20.d[1], x19
  224. ldpsw x18, x19, [x17, 40]
  225. ins v21.d[0], x18
  226. ins v21.d[1], x19
  227. ldpsw x18, x19, [x17, 48]
  228. ins v22.d[0], x18
  229. ins v22.d[1], x19
  230. ldpsw x18, x19, [x17, 56]
  231. ins v23.d[0], x18
  232. ins v23.d[1], x19
  233. scvtf v20.2d, v20.2d
  234. scvtf v21.2d, v21.2d
  235. scvtf v22.2d, v22.2d
  236. scvtf v23.2d, v23.2d
  237. and v20.16b, v20.16b, v29.16b
  238. and v21.16b, v21.16b, v29.16b
  239. and v22.16b, v22.16b, v29.16b
  240. and v23.16b, v23.16b, v29.16b
  241. orr v20.16b, v20.16b, v30.16b
  242. orr v21.16b, v21.16b, v30.16b
  243. orr v22.16b, v22.16b, v30.16b
  244. orr v23.16b, v23.16b, v30.16b
  245. # Execute VM instructions
  246. DECL(randomx_program_aarch64_vm_instructions):
  247. # buffer for generated instructions
  248. # FDIV_M is the largest instruction taking up to 12 ARMv8 instructions
  249. .fill RANDOMX_PROGRAM_SIZE*12,4,0
  250. literal_x0: .fill 1,8,0
  251. literal_x11: .fill 1,8,0
  252. literal_x20: .fill 1,8,0
  253. literal_x21: .fill 1,8,0
  254. literal_x22: .fill 1,8,0
  255. literal_x23: .fill 1,8,0
  256. literal_x24: .fill 1,8,0
  257. literal_x25: .fill 1,8,0
  258. literal_x26: .fill 1,8,0
  259. literal_x27: .fill 1,8,0
  260. literal_x28: .fill 1,8,0
  261. literal_x29: .fill 1,8,0
  262. literal_x30: .fill 1,8,0
  263. DECL(randomx_program_aarch64_imul_rcp_literals_end):
  264. literal_v0: .fill 2,8,0
  265. literal_v1: .fill 2,8,0
  266. literal_v2: .fill 2,8,0
  267. literal_v3: .fill 2,8,0
  268. literal_v4: .fill 2,8,0
  269. literal_v5: .fill 2,8,0
  270. literal_v6: .fill 2,8,0
  271. literal_v7: .fill 2,8,0
  272. literal_v8: .fill 2,8,0
  273. literal_v9: .fill 2,8,0
  274. literal_v10: .fill 2,8,0
  275. literal_v11: .fill 2,8,0
  276. literal_v12: .fill 2,8,0
  277. literal_v13: .fill 2,8,0
  278. literal_v14: .fill 2,8,0
  279. literal_v15: .fill 2,8,0
  280. DECL(randomx_program_aarch64_vm_instructions_end):
  281. # mx ^= r[readReg2] ^ r[readReg3];
  282. eor x9, x9, x18
  283. # Calculate dataset pointer for dataset prefetch
  284. mov w18, w9
  285. DECL(randomx_program_aarch64_cacheline_align_mask1):
  286. # Actual mask will be inserted by JIT compiler
  287. and x18, x18, 1
  288. add x18, x18, x1
  289. # Prefetch dataset data
  290. prfm pldl2strm, [x18]
  291. # mx <-> ma
  292. ror x9, x9, 32
  293. # Calculate dataset pointer for dataset read
  294. mov w10, w9
  295. DECL(randomx_program_aarch64_cacheline_align_mask2):
  296. # Actual mask will be inserted by JIT compiler
  297. and x10, x10, 1
  298. add x10, x10, x1
  299. DECL(randomx_program_aarch64_xor_with_dataset_line):
  300. # xor integer registers with dataset data
  301. ldp x18, x19, [x10]
  302. eor x4, x4, x18
  303. eor x5, x5, x19
  304. ldp x18, x19, [x10, 16]
  305. eor x6, x6, x18
  306. eor x7, x7, x19
  307. ldp x18, x19, [x10, 32]
  308. eor x12, x12, x18
  309. eor x13, x13, x19
  310. ldp x18, x19, [x10, 48]
  311. eor x14, x14, x18
  312. eor x15, x15, x19
  313. DECL(randomx_program_aarch64_update_spMix1):
  314. # JIT compiler will replace it with "eor x10, config.readReg0, config.readReg1"
  315. eor x10, x0, x0
  316. # Store integer registers to scratchpad (spAddr1)
  317. stp x4, x5, [x17, 0]
  318. stp x6, x7, [x17, 16]
  319. stp x12, x13, [x17, 32]
  320. stp x14, x15, [x17, 48]
  321. # xor group F and group E registers
  322. eor v16.16b, v16.16b, v20.16b
  323. eor v17.16b, v17.16b, v21.16b
  324. eor v18.16b, v18.16b, v22.16b
  325. eor v19.16b, v19.16b, v23.16b
  326. # Store FP registers to scratchpad (spAddr0)
  327. stp q16, q17, [x16, 0]
  328. stp q18, q19, [x16, 32]
  329. subs x3, x3, 1
  330. bne DECL(randomx_program_aarch64_main_loop)
  331. # Restore x0
  332. ldr x0, [sp], 16
  333. # Store integer registers
  334. stp x4, x5, [x0, 0]
  335. stp x6, x7, [x0, 16]
  336. stp x12, x13, [x0, 32]
  337. stp x14, x15, [x0, 48]
  338. # Store FP registers
  339. stp q16, q17, [x0, 64]
  340. stp q18, q19, [x0, 96]
  341. stp q20, q21, [x0, 128]
  342. stp q22, q23, [x0, 160]
  343. # Restore callee-saved registers
  344. ldp x16, x17, [sp]
  345. ldp x18, x19, [sp, 16]
  346. ldp x20, x21, [sp, 32]
  347. ldp x22, x23, [sp, 48]
  348. ldp x24, x25, [sp, 64]
  349. ldp x26, x27, [sp, 80]
  350. ldp x28, x29, [sp, 96]
  351. ldp x8, x30, [sp, 112]
  352. ldp d8, d9, [sp, 128]
  353. ldp d10, d11, [sp, 144]
  354. ldp d12, d13, [sp, 160]
  355. ldp d14, d15, [sp, 176]
  356. add sp, sp, 192
  357. ret
  358. DECL(randomx_program_aarch64_vm_instructions_end_light):
  359. sub sp, sp, 96
  360. stp x0, x1, [sp, 64]
  361. stp x2, x30, [sp, 80]
  362. # mx ^= r[readReg2] ^ r[readReg3];
  363. eor x9, x9, x18
  364. # mx <-> ma
  365. ror x9, x9, 32
  366. # x0 -> pointer to cache memory
  367. mov x0, x1
  368. # x1 -> pointer to output
  369. mov x1, sp
  370. DECL(randomx_program_aarch64_light_cacheline_align_mask):
  371. # Actual mask will be inserted by JIT compiler
  372. and w2, w9, 1
  373. # x2 -> item number
  374. lsr x2, x2, 6
  375. DECL(randomx_program_aarch64_light_dataset_offset):
  376. # Apply dataset offset (filled in by JIT compiler)
  377. add x2, x2, 0
  378. add x2, x2, 0
  379. bl DECL(randomx_calc_dataset_item_aarch64)
  380. mov x10, sp
  381. ldp x0, x1, [sp, 64]
  382. ldp x2, x30, [sp, 80]
  383. add sp, sp, 96
  384. b DECL(randomx_program_aarch64_xor_with_dataset_line)
  385. # Input parameters
  386. #
  387. # x0 -> pointer to cache
  388. # x1 -> pointer to dataset memory at startItem
  389. # x2 -> start item
  390. # x3 -> end item
  391. DECL(randomx_init_dataset_aarch64):
  392. # Save x30 (return address)
  393. str x30, [sp, -16]!
  394. # Load pointer to cache memory
  395. ldr x0, [x0]
  396. DECL(randomx_init_dataset_aarch64_main_loop):
  397. bl DECL(randomx_calc_dataset_item_aarch64)
  398. add x1, x1, 64
  399. add x2, x2, 1
  400. cmp x2, x3
  401. bne DECL(randomx_init_dataset_aarch64_main_loop)
  402. # Restore x30 (return address)
  403. ldr x30, [sp], 16
  404. ret
  405. DECL(randomx_init_dataset_aarch64_end):
  406. # Input parameters
  407. #
  408. # x0 -> pointer to cache memory
  409. # x1 -> pointer to output
  410. # x2 -> item number
  411. #
  412. # Register allocation
  413. #
  414. # x0-x7 -> output value (calculated dataset item)
  415. # x8 -> pointer to cache memory
  416. # x9 -> pointer to output
  417. # x10 -> registerValue
  418. # x11 -> mixBlock
  419. # x12 -> temporary
  420. # x13 -> temporary
  421. DECL(randomx_calc_dataset_item_aarch64):
  422. sub sp, sp, 112
  423. stp x0, x1, [sp]
  424. stp x2, x3, [sp, 16]
  425. stp x4, x5, [sp, 32]
  426. stp x6, x7, [sp, 48]
  427. stp x8, x9, [sp, 64]
  428. stp x10, x11, [sp, 80]
  429. stp x12, x13, [sp, 96]
  430. ldr x12, superscalarMul0
  431. mov x8, x0
  432. mov x9, x1
  433. mov x10, x2
  434. # rl[0] = (itemNumber + 1) * superscalarMul0;
  435. madd x0, x2, x12, x12
  436. # rl[1] = rl[0] ^ superscalarAdd1;
  437. ldr x12, superscalarAdd1
  438. eor x1, x0, x12
  439. # rl[2] = rl[0] ^ superscalarAdd2;
  440. ldr x12, superscalarAdd2
  441. eor x2, x0, x12
  442. # rl[3] = rl[0] ^ superscalarAdd3;
  443. ldr x12, superscalarAdd3
  444. eor x3, x0, x12
  445. # rl[4] = rl[0] ^ superscalarAdd4;
  446. ldr x12, superscalarAdd4
  447. eor x4, x0, x12
  448. # rl[5] = rl[0] ^ superscalarAdd5;
  449. ldr x12, superscalarAdd5
  450. eor x5, x0, x12
  451. # rl[6] = rl[0] ^ superscalarAdd6;
  452. ldr x12, superscalarAdd6
  453. eor x6, x0, x12
  454. # rl[7] = rl[0] ^ superscalarAdd7;
  455. ldr x12, superscalarAdd7
  456. eor x7, x0, x12
  457. b DECL(randomx_calc_dataset_item_aarch64_prefetch)
  458. superscalarMul0: .quad 6364136223846793005
  459. superscalarAdd1: .quad 9298411001130361340
  460. superscalarAdd2: .quad 12065312585734608966
  461. superscalarAdd3: .quad 9306329213124626780
  462. superscalarAdd4: .quad 5281919268842080866
  463. superscalarAdd5: .quad 10536153434571861004
  464. superscalarAdd6: .quad 3398623926847679864
  465. superscalarAdd7: .quad 9549104520008361294
  466. # Prefetch -> SuperScalar hash -> Mix will be repeated N times
  467. DECL(randomx_calc_dataset_item_aarch64_prefetch):
  468. # Actual mask will be inserted by JIT compiler
  469. and x11, x10, 1
  470. add x11, x8, x11, lsl 6
  471. prfm pldl2strm, [x11]
  472. # Generated SuperScalar hash program goes here
  473. DECL(randomx_calc_dataset_item_aarch64_mix):
  474. ldp x12, x13, [x11]
  475. eor x0, x0, x12
  476. eor x1, x1, x13
  477. ldp x12, x13, [x11, 16]
  478. eor x2, x2, x12
  479. eor x3, x3, x13
  480. ldp x12, x13, [x11, 32]
  481. eor x4, x4, x12
  482. eor x5, x5, x13
  483. ldp x12, x13, [x11, 48]
  484. eor x6, x6, x12
  485. eor x7, x7, x13
  486. DECL(randomx_calc_dataset_item_aarch64_store_result):
  487. stp x0, x1, [x9]
  488. stp x2, x3, [x9, 16]
  489. stp x4, x5, [x9, 32]
  490. stp x6, x7, [x9, 48]
  491. ldp x0, x1, [sp]
  492. ldp x2, x3, [sp, 16]
  493. ldp x4, x5, [sp, 32]
  494. ldp x6, x7, [sp, 48]
  495. ldp x8, x9, [sp, 64]
  496. ldp x10, x11, [sp, 80]
  497. ldp x12, x13, [sp, 96]
  498. add sp, sp, 112
  499. ret
  500. DECL(randomx_calc_dataset_item_aarch64_end):