nb_kernel313_x86_64_sse.intel_syntax.s
来自「最著名最快的分子模拟软件」· S 代码 · 共 2,333 行 · 第 1/5 页
S
2,333 行
movaps xmm3, xmm2 mulps xmm2, xmm2 movaps xmm0, [rsp + nb313nf_three] mulps xmm2, xmm5 ;# rsq*lu*lu subps xmm0, xmm2 ;# 30-rsq*lu*lu mulps xmm0, xmm3 ;# lu*(3-rsq*lu*lu) mulps xmm0, [rsp + nb313nf_half] movaps [rsp + nb313nf_rinvH2], xmm0 ;# rinvH2 in xmm4 mulps xmm5, xmm0 movaps [rsp + nb313nf_rH2], xmm5 ;# rsqM - seed to xmm2 rsqrtps xmm2, xmm4 movaps xmm3, xmm2 mulps xmm2, xmm2 movaps xmm0, [rsp + nb313nf_three] mulps xmm2, xmm4 ;# rsq*lu*lu subps xmm0, xmm2 ;# 30-rsq*lu*lu mulps xmm0, xmm3 ;# lu*(3-rsq*lu*lu) mulps xmm0, [rsp + nb313nf_half] movaps [rsp + nb313nf_rinvM], xmm0 ;# rinvM in xmm5 mulps xmm4, xmm0 movaps [rsp + nb313nf_rM], xmm4 ;# Do the O LJ-only interaction directly. rcpps xmm2, xmm7 movaps xmm1, [rsp + nb313nf_two] mulps xmm7, xmm2 subps xmm1, xmm7 mulps xmm2, xmm1 ;# rinvsq movaps xmm0, xmm2 mulps xmm0, xmm2 ;# r4 mulps xmm0, xmm2 ;# r6 movaps xmm1, xmm0 mulps xmm1, xmm1 ;# r12 mulps xmm0, [rsp + nb313nf_c6] mulps xmm1, [rsp + nb313nf_c12] movaps xmm3, xmm1 subps xmm3, xmm0 ;# Vvdw12-Vvdw6 addps xmm3, [rsp + nb313nf_Vvdwtot] movaps [rsp + nb313nf_Vvdwtot], xmm3 ;# Do H1 interaction mov rsi, [rbp + nb313nf_VFtab] movaps xmm7, [rsp + nb313nf_rH1] mulps xmm7, [rsp + nb313nf_tsc] movhlps xmm4, xmm7 cvttps2pi mm6, xmm7 cvttps2pi mm7, xmm4 ;# mm6/mm7 contain lu indices cvtpi2ps xmm3, mm6 cvtpi2ps xmm4, mm7 movlhps xmm3, xmm4 subps xmm7, xmm3 movaps xmm1, xmm7 ;# xmm1=eps movaps xmm2, xmm1 mulps xmm2, xmm2 ;# xmm2=eps2 pslld mm6, 2 pslld mm7, 2 movd mm0, eax movd mm1, ebx movd mm2, ecx movd mm3, edx movd eax, mm6 psrlq mm6, 32 movd ecx, mm7 psrlq mm7, 32 movd ebx, mm6 movd edx, mm7 movlps xmm5, [rsi + rax*4] movlps xmm7, [rsi + rcx*4] movhps xmm5, [rsi + rbx*4] movhps xmm7, [rsi + rdx*4] ;# got half coulomb table movaps xmm4, xmm5 shufps xmm4, xmm7, 136 ;# 10001000 shufps xmm5, xmm7, 221 ;# 11011101 movlps xmm7, [rsi + rax*4 + 8] movlps xmm3, [rsi + rcx*4 + 8] movhps xmm7, [rsi + rbx*4 + 8] movhps xmm3, [rsi + rdx*4 + 8] ;# other half of coulomb table movaps xmm6, xmm7 shufps xmm6, xmm3, 136 ;# 10001000 shufps xmm7, xmm3, 221 ;# 11011101 ;# coulomb table ready, in xmm4-xmm7 mulps xmm6, xmm1 ;# xmm6=Geps mulps xmm7, xmm2 ;# xmm7=Heps2 addps xmm5, xmm6 addps xmm5, xmm7 ;# xmm5=Fp movaps xmm0, [rsp + nb313nf_qqH] mulps xmm5, xmm1 ;# xmm5=eps*Fp addps xmm5, xmm4 ;# xmm5=VV mulps xmm5, xmm0 ;# vcoul=qq*VV ;# at this point mm5 contains vcoul ;# increment vcoul addps xmm5, [rsp + nb313nf_vctot] movaps [rsp + nb313nf_vctot], xmm5 ;# Done with H1, do H2 interactions movaps xmm7, [rsp + nb313nf_rH2] mulps xmm7, [rsp + nb313nf_tsc] movhlps xmm4, xmm7 cvttps2pi mm6, xmm7 cvttps2pi mm7, xmm4 ;# mm6/mm7 contain lu indices cvtpi2ps xmm3, mm6 cvtpi2ps xmm4, mm7 movlhps xmm3, xmm4 subps xmm7, xmm3 movaps xmm1, xmm7 ;# xmm1=eps movaps xmm2, xmm1 mulps xmm2, xmm2 ;# xmm2=eps2 pslld mm6, 2 pslld mm7, 2 movd eax, mm6 psrlq mm6, 32 movd ecx, mm7 psrlq mm7, 32 movd ebx, mm6 movd edx, mm7 movlps xmm5, [rsi + rax*4] movlps xmm7, [rsi + rcx*4] movhps xmm5, [rsi + rbx*4] movhps xmm7, [rsi + rdx*4] ;# got half coulomb table movaps xmm4, xmm5 shufps xmm4, xmm7, 136 ;# shuffle 10001000 shufps xmm5, xmm7, 221 ;# shuffle 11011101 movlps xmm7, [rsi + rax*4 + 8] movlps xmm3, [rsi + rcx*4 + 8] movhps xmm7, [rsi + rbx*4 + 8] movhps xmm3, [rsi + rdx*4 + 8] ;# other half of coulomb table movaps xmm6, xmm7 shufps xmm6, xmm3, 136 ;# shuf 10001000 shufps xmm7, xmm3, 221 ;# shuf 11011101 ;# coulomb table ready, in xmm4-xmm7 mulps xmm6, xmm1 ;# xmm6=Geps mulps xmm7, xmm2 ;# xmm7=Heps2 addps xmm5, xmm6 addps xmm5, xmm7 ;# xmm5=Fp movaps xmm0, [rsp + nb313nf_qqH] mulps xmm5, xmm1 ;# xmm5=eps*Fp addps xmm5, xmm4 ;# xmm5=VV mulps xmm5, xmm0 ;# vcoul=qq*VV ;# at this point mm5 contains vcoul ;# increment vcoul addps xmm5, [rsp + nb313nf_vctot] movaps [rsp + nb313nf_vctot], xmm5 ;# Done with H2, do M interactions movaps xmm7, [rsp + nb313nf_rM] mulps xmm7, [rsp + nb313nf_tsc] movhlps xmm4, xmm7 cvttps2pi mm6, xmm7 cvttps2pi mm7, xmm4 ;# mm6/mm7 contain lu indices cvtpi2ps xmm3, mm6 cvtpi2ps xmm4, mm7 movlhps xmm3, xmm4 subps xmm7, xmm3 movaps xmm1, xmm7 ;# xmm1=eps movaps xmm2, xmm1 mulps xmm2, xmm2 ;# xmm2=eps2 pslld mm6, 2 pslld mm7, 2 movd eax, mm6 psrlq mm6, 32 movd ecx, mm7 psrlq mm7, 32 movd ebx, mm6 movd edx, mm7 movlps xmm5, [rsi + rax*4] movlps xmm7, [rsi + rcx*4] movhps xmm5, [rsi + rbx*4] movhps xmm7, [rsi + rdx*4] ;# got half coulomb table movaps xmm4, xmm5 shufps xmm4, xmm7, 136 ;# 10001000 shufps xmm5, xmm7, 221 ;# 11011101 movlps xmm7, [rsi + rax*4 + 8] movlps xmm3, [rsi + rcx*4 + 8] movhps xmm7, [rsi + rbx*4 + 8] movhps xmm3, [rsi + rdx*4 + 8] ;# other half of coulomb table movaps xmm6, xmm7 shufps xmm6, xmm3, 136 ;# 10001000 shufps xmm7, xmm3, 221 ;# 11011101 ;# coulomb table ready, in xmm4-xmm7 mulps xmm6, xmm1 ;# xmm6=Geps mulps xmm7, xmm2 ;# xmm7=Heps2 addps xmm5, xmm6 addps xmm5, xmm7 ;# xmm5=Fp movaps xmm0, [rsp + nb313nf_qqM] mulps xmm5, xmm1 ;# xmm5=eps*Fp addps xmm5, xmm4 ;# xmm5=VV mulps xmm5, xmm0 ;# vcoul=qq*VV ;# at this point mm5 contains vcoul ;# increment vcoul addps xmm5, [rsp + nb313nf_vctot] movaps [rsp + nb313nf_vctot], xmm5 ;# should we do one more iteration? sub dword ptr [rsp + nb313nf_innerk], 4 jl .nb313nf_odd_inner jmp .nb313nf_unroll_loop.nb313nf_odd_inner: add dword ptr [rsp + nb313nf_innerk], 4 jnz .nb313nf_odd_loop jmp .nb313nf_updateouterdata.nb313nf_odd_loop: mov rdx, [rsp + nb313nf_innerjjnr] ;# pointer to jjnr[k] mov eax, [rdx] add qword ptr [rsp + nb313nf_innerjjnr], 4 xorps xmm4, xmm4 ;# clear reg. movss xmm4, [rsp + nb313nf_iqM] mov rsi, [rbp + nb313nf_charge] movhps xmm4, [rsp + nb313nf_iqH] ;# [qM 0 qH qH] shufps xmm4, xmm4, 41 ;# [0 qH qH qM] movss xmm3, [rsi + rax*4] ;# charge in xmm3 shufps xmm3, xmm3, 0 mulps xmm3, xmm4 movaps [rsp + nb313nf_qqM], xmm3 ;# use dummy qq for storage xorps xmm6, xmm6 mov rsi, [rbp + nb313nf_type] mov ebx, [rsi + rax*4] mov rsi, [rbp + nb313nf_vdwparam] shl ebx, 1 add ebx, [rsp + nb313nf_ntia] movlps xmm6, [rsi + rbx*4] movaps xmm7, xmm6 shufps xmm6, xmm6, 252 ;# 11111100 shufps xmm7, xmm7, 253 ;# 11111101 movaps [rsp + nb313nf_c6], xmm6 movaps [rsp + nb313nf_c12], xmm7 mov rsi, [rbp + nb313nf_pos] lea rax, [rax + rax*2] movss xmm3, [rsp + nb313nf_ixO] movss xmm4, [rsp + nb313nf_iyO] movss xmm5, [rsp + nb313nf_izO] movss xmm0, [rsp + nb313nf_ixH1] movss xmm1, [rsp + nb313nf_iyH1] movss xmm2, [rsp + nb313nf_izH1] unpcklps xmm3, [rsp + nb313nf_ixH2] ;# ixO ixH2 - - unpcklps xmm4, [rsp + nb313nf_iyH2] ;# iyO iyH2 - - unpcklps xmm5, [rsp + nb313nf_izH2] ;# izO izH2 - - unpcklps xmm0, [rsp + nb313nf_ixM] ;# ixH1 ixM - - unpcklps xmm1, [rsp + nb313nf_iyM] ;# iyH1 iyM - - unpcklps xmm2, [rsp + nb313nf_izM] ;# izH1 izM - - unpcklps xmm3, xmm0 ;# ixO ixH1 ixH2 ixM unpcklps xmm4, xmm1 ;# same for y unpcklps xmm5, xmm2 ;# same for z ;# move j coords to xmm0-xmm2 movss xmm0, [rsi + rax*4] movss xmm1, [rsi + rax*4 + 4] movss xmm2, [rsi + rax*4 + 8] shufps xmm0, xmm0, 0 shufps xmm1, xmm1, 0 shufps xmm2, xmm2, 0 subps xmm3, xmm0 subps xmm4, xmm1 subps xmm5, xmm2 mulps xmm3, xmm3 mulps xmm4, xmm4 mulps xmm5, xmm5 addps xmm4, xmm3 addps xmm4, xmm5 ;# rsq in xmm4 rsqrtps xmm5, xmm4 ;# lookup seed in xmm5 movaps xmm2, xmm5 mulps xmm5, xmm5 movaps xmm1, [rsp + nb313nf_three] mulps xmm5, xmm4 ;# rsq*lu*lu movaps xmm0, [rsp + nb313nf_half] subps xmm1, xmm5 ;# 30-rsq*lu*lu mulps xmm1, xmm2 mulps xmm0, xmm1 ;# xmm0=rinv movaps [rsp + nb313nf_rinvM], xmm0 mulps xmm4, xmm0 ;# r mulps xmm4, [rsp + nb313nf_tsc] movhlps xmm7, xmm4 cvttps2pi mm6, xmm4 cvttps2pi mm7, xmm7 ;# mm6/mm7 contain lu indices cvtpi2ps xmm3, mm6 cvtpi2ps xmm7, mm7 movlhps xmm3, xmm7 subps xmm4, xmm3 movaps xmm1, xmm4 ;# xmm1=eps movaps xmm2, xmm1 mulps xmm2, xmm2 ;# xmm2=eps2 pslld mm6, 2 pslld mm7, 2 mov rsi, [rbp + nb313nf_VFtab] psrlq mm6, 32 movd ebx, mm6 movd ecx, mm7 psrlq mm7, 32 movd edx, mm7 xorps xmm5, xmm5 movlps xmm3, [rsi + rcx*4] ;# data: Y3 F3 - - movhps xmm5, [rsi + rbx*4] ;# data: 0 0 Y2 F2 movhps xmm3, [rsi + rdx*4] ;# data: Y3 F3 Y4 F4 movaps xmm4, xmm5 ;# data: 0 0 Y2 F2 shufps xmm4, xmm3, 0x88 ;# data: 0 Y2 Y3 Y3 shufps xmm5, xmm3, 0xDD ;# data: 0 F2 F3 F4 xorps xmm7, xmm7 movlps xmm3, [rsi + rcx*4 + 8] ;# data: G3 H3 - - movhps xmm7, [rsi + rbx*4 + 8] ;# data: 0 0 G2 H2 movhps xmm3, [rsi + rdx*4 + 8] ;# data: G3 H3 G4 H4 movaps xmm6, xmm7 ;# data: 0 0 G2 H2 shufps xmm6, xmm3, 0x88 ;# data: 0 G2 G3 G3 shufps xmm7, xmm3, 0xDD ;# data: 0 H2 H3 H4 ;# xmm4 = 0 Y2 Y3 Y4 ;# xmm5 = 0 F2 F3 F4 ;# xmm6 = 0 G2 G3 G4 ;# xmm7 = 0 H2 H3 H4 ;# coulomb table ready, in xmm4-xmm7 mulps xmm6, xmm1 ;# xmm6=Geps mulps xmm7, xmm2 ;# xmm7=Heps2 addps xmm5, xmm6 addps xmm5, xmm7 ;# xmm5=Fp movaps xmm0, [rsp + nb313nf_qqM] mulps xmm5, xmm1 ;# xmm5=eps*Fp addps xmm5, xmm4 ;# xmm5=VV mulps xmm5, xmm0 ;# vcoul=qq*VV ;# at this point mm5 contains vcoul ;# increment vcoul addps xmm5, [rsp + nb313nf_vctot] movaps [rsp + nb313nf_vctot], xmm5 ;# do nontable L-J in first element only. movaps xmm2, [rsp + nb313nf_rinvM] mulss xmm2, xmm2 movaps xmm1, xmm2 mulss xmm1, xmm1 mulss xmm1, xmm2 ;# xmm1=rinvsix xorps xmm4, xmm4 movss xmm4, xmm1 mulss xmm4, xmm4 ;# xmm4=rinvtwelve mulss xmm1, [rsp + nb313nf_c6] mulss xmm4, [rsp + nb313nf_c12] movaps xmm3, xmm4 subss xmm3, xmm1 ;# xmm3=Vvdw12-Vvdw6 addss xmm3, [rsp + nb313nf_Vvdwtot] movss [rsp + nb313nf_Vvdwtot], xmm3 dec dword ptr [rsp + nb313nf_innerk] jz .nb313nf_updateouterdata jmp .nb313nf_odd_loop.nb313nf_updateouterdata: ;# get n from stack mov esi, [rsp + nb313nf_n] ;# get group index for i particle mov rdx, [rbp + nb313nf_gid] ;# base of gid[] mov edx, [rdx + rsi*4] ;# ggid=gid[n] ;# accumulate total potential energy and update it movaps xmm7, [rsp + nb313nf_vctot] ;# accumulate movhlps xmm6, xmm7 addps xmm7, xmm6 ;# pos 0-1 in xmm7 have the sum now movaps xmm6, xmm7 shufps xmm6, xmm6, 1 addss xmm7, xmm6 ;# add earlier value from mem mov rax, [rbp + nb313nf_Vc] addss xmm7, [rax + rdx*4] ;# move back to mem movss [rax + rdx*4], xmm7 ;# accumulate total lj energy and update it movaps xmm7, [rsp + nb313nf_Vvdwtot] ;# accumulate movhlps xmm6, xmm7 addps xmm7, xmm6 ;# pos 0-1 in xmm7 have the sum now movaps xmm6, xmm7 shufps xmm6, xmm6, 1 addss xmm7, xmm6 ;# add earlier value from mem mov rax, [rbp + nb313nf_Vvdw] addss xmm7, [rax + rdx*4] ;# move back to mem movss [rax + rdx*4], xmm7 ;# finish if last mov ecx, [rsp + nb313nf_nn1] ;# esi already loaded with n inc esi sub ecx, esi jz .nb313nf_outerend ;# not last, iterate outer loop once more! mov [rsp + nb313nf_n], esi jmp .nb313nf_outer.nb313nf_outerend: ;# check if more outer neighborlists remain mov ecx, [rsp + nb313nf_nri] ;# esi already loaded with n above sub ecx, esi jz .nb313nf_end ;# non-zero, do one more workunit jmp .nb313nf_threadloop.nb313nf_end: mov eax, [rsp + nb313nf_nouter] mov ebx, [rsp + nb313nf_ninner] mov rcx, [rbp + nb313nf_outeriter] mov rdx, [rbp + nb313nf_inneriter] mov [rcx], eax mov [rdx], ebx add rsp, 584 emms pop r15 pop r14 pop r13 pop r12 pop rbx pop rbp ret
⌨️ 快捷键说明
复制代码Ctrl + C
搜索代码Ctrl + F
全屏模式F11
增大字号Ctrl + =
减小字号Ctrl + -
显示快捷键?