nb_kernel214_x86_64_sse2.s

来自「最著名最快的分子模拟软件」· S 代码 · 共 2,300 行 · 第 1/5 页

S
2,300
字号
        ## 1/x lookup seed in xmm6         movapd nb214_two(%rsp),%xmm0        movapd %xmm4,%xmm5        mulpd %xmm6,%xmm4       ## lu*rsq         subpd %xmm4,%xmm0       ## 2-lu*rsq         mulpd %xmm0,%xmm6       ## (new lu)         movapd nb214_two(%rsp),%xmm0        mulpd %xmm6,%xmm5       ## lu*rsq         subpd %xmm5,%xmm0       ## 2-lu*rsq         mulpd %xmm6,%xmm0       ## xmm0=rinvsq         movapd %xmm0,%xmm1        mulpd  %xmm0,%xmm1        mulpd  %xmm0,%xmm1      ## xmm1=rinvsix         movapd %xmm1,%xmm2        mulpd  %xmm2,%xmm2      ## xmm2=rinvtwelve         mulpd  nb214_c6(%rsp),%xmm1     ## mult by c6        mulpd  nb214_c12(%rsp),%xmm2     ## mult by c12        movapd %xmm2,%xmm5        subpd  %xmm1,%xmm5      ## Vvdw=Vvdw12-Vvdw6         mulpd  nb214_six(%rsp),%xmm1        mulpd  nb214_twelve(%rsp),%xmm2        subpd  %xmm1,%xmm2        mulpd  %xmm2,%xmm0      ## xmm0=total fscal     ## increment potential    addpd  nb214_Vvdwtot(%rsp),%xmm5    movapd %xmm5,nb214_Vvdwtot(%rsp)        movq   nb214_faction(%rbp),%rdi        mulpd  %xmm0,%xmm9        mulpd  %xmm0,%xmm10        mulpd  %xmm0,%xmm11    movapd nb214_fixO(%rsp),%xmm0    movapd nb214_fiyO(%rsp),%xmm1    movapd nb214_fizO(%rsp),%xmm2    ## accumulate i forces    addpd %xmm9,%xmm0    addpd %xmm10,%xmm1    addpd %xmm11,%xmm2    movapd %xmm0,nb214_fixO(%rsp)    movapd %xmm1,nb214_fiyO(%rsp)    movapd %xmm2,nb214_fizO(%rsp)        ## the fj's - start by accumulating forces from memory         movlpd (%rdi,%rax,8),%xmm3        movlpd 8(%rdi,%rax,8),%xmm4        movlpd 16(%rdi,%rax,8),%xmm5        movhpd (%rdi,%rbx,8),%xmm3        movhpd 8(%rdi,%rbx,8),%xmm4        movhpd 16(%rdi,%rbx,8),%xmm5        addpd %xmm9,%xmm3        addpd %xmm10,%xmm4        addpd %xmm11,%xmm5        movlpd %xmm3,(%rdi,%rax,8)        movlpd %xmm4,8(%rdi,%rax,8)        movlpd %xmm5,16(%rdi,%rax,8)        movhpd %xmm3,(%rdi,%rbx,8)        movhpd %xmm4,8(%rdi,%rbx,8)        movhpd %xmm5,16(%rdi,%rbx,8)    ## done with OO interaction    ## move j H1 coordinates to local temp variables     movlpd 24(%rsi,%rax,8),%xmm0    movlpd 32(%rsi,%rax,8),%xmm1    movlpd 40(%rsi,%rax,8),%xmm2    movhpd 24(%rsi,%rbx,8),%xmm0    movhpd 32(%rsi,%rbx,8),%xmm1    movhpd 40(%rsi,%rbx,8),%xmm2    ## xmm0 = H1x    ## xmm1 = H1y    ## xmm2 = H1z    movapd %xmm0,%xmm3    movapd %xmm1,%xmm4    movapd %xmm2,%xmm5    movapd %xmm0,%xmm6    movapd %xmm1,%xmm7    movapd %xmm2,%xmm8    subpd nb214_ixH1(%rsp),%xmm0    subpd nb214_iyH1(%rsp),%xmm1    subpd nb214_izH1(%rsp),%xmm2    subpd nb214_ixH2(%rsp),%xmm3    subpd nb214_iyH2(%rsp),%xmm4    subpd nb214_izH2(%rsp),%xmm5    subpd nb214_ixM(%rsp),%xmm6    subpd nb214_iyM(%rsp),%xmm7    subpd nb214_izM(%rsp),%xmm8        movapd %xmm0,nb214_dxH1H1(%rsp)        movapd %xmm1,nb214_dyH1H1(%rsp)        movapd %xmm2,nb214_dzH1H1(%rsp)        mulpd  %xmm0,%xmm0        mulpd  %xmm1,%xmm1        mulpd  %xmm2,%xmm2        movapd %xmm3,nb214_dxH2H1(%rsp)        movapd %xmm4,nb214_dyH2H1(%rsp)        movapd %xmm5,nb214_dzH2H1(%rsp)        mulpd  %xmm3,%xmm3        mulpd  %xmm4,%xmm4        mulpd  %xmm5,%xmm5        movapd %xmm6,nb214_dxMH1(%rsp)        movapd %xmm7,nb214_dyMH1(%rsp)        movapd %xmm8,nb214_dzMH1(%rsp)        mulpd  %xmm6,%xmm6        mulpd  %xmm7,%xmm7        mulpd  %xmm8,%xmm8        addpd  %xmm1,%xmm0        addpd  %xmm2,%xmm0        addpd  %xmm4,%xmm3        addpd  %xmm5,%xmm3    addpd  %xmm7,%xmm6    addpd  %xmm8,%xmm6        ## start doing invsqrt for jH1 atoms    cvtpd2ps %xmm0,%xmm1    cvtpd2ps %xmm3,%xmm4    cvtpd2ps %xmm6,%xmm7        rsqrtps %xmm1,%xmm1        rsqrtps %xmm4,%xmm4    rsqrtps %xmm7,%xmm7    cvtps2pd %xmm1,%xmm1    cvtps2pd %xmm4,%xmm4    cvtps2pd %xmm7,%xmm7        movapd  %xmm1,%xmm2        movapd  %xmm4,%xmm5    movapd  %xmm7,%xmm8        mulpd   %xmm1,%xmm1 ## lu*lu        mulpd   %xmm4,%xmm4 ## lu*lu    mulpd   %xmm7,%xmm7 ## lu*lu        movapd  nb214_three(%rsp),%xmm9        movapd  %xmm9,%xmm10    movapd  %xmm9,%xmm11        mulpd   %xmm0,%xmm1 ## rsq*lu*lu        mulpd   %xmm3,%xmm4 ## rsq*lu*lu     mulpd   %xmm6,%xmm7 ## rsq*lu*lu        subpd   %xmm1,%xmm9        subpd   %xmm4,%xmm10    subpd   %xmm7,%xmm11 ## 3-rsq*lu*lu        mulpd   %xmm2,%xmm9        mulpd   %xmm5,%xmm10    mulpd   %xmm8,%xmm11 ## lu*(3-rsq*lu*lu)        movapd  nb214_half(%rsp),%xmm15        mulpd   %xmm15,%xmm9 ## first iteration for rinvH1H1         mulpd   %xmm15,%xmm10 ## first iteration for rinvH2H1    mulpd   %xmm15,%xmm11 ## first iteration for rinvMH1     ## second iteration step            movapd  %xmm9,%xmm2        movapd  %xmm10,%xmm5    movapd  %xmm11,%xmm8        mulpd   %xmm2,%xmm2 ## lu*lu        mulpd   %xmm5,%xmm5 ## lu*lu    mulpd   %xmm8,%xmm8 ## lu*lu        movapd  nb214_three(%rsp),%xmm1        movapd  %xmm1,%xmm4    movapd  %xmm1,%xmm7        mulpd   %xmm0,%xmm2 ## rsq*lu*lu        mulpd   %xmm3,%xmm5 ## rsq*lu*lu     mulpd   %xmm6,%xmm8 ## rsq*lu*lu        subpd   %xmm2,%xmm1        subpd   %xmm5,%xmm4    subpd   %xmm8,%xmm7 ## 3-rsq*lu*lu        mulpd   %xmm1,%xmm9        mulpd   %xmm4,%xmm10    mulpd   %xmm7,%xmm11 ## lu*(3-rsq*lu*lu)        movapd  nb214_half(%rsp),%xmm15        mulpd   %xmm15,%xmm9 ##  rinvH1H1         mulpd   %xmm15,%xmm10 ##   rinvH2H1    mulpd   %xmm15,%xmm11 ##   rinvMH1        ## H1 interactions     ## rsq in xmm0,xmm3,xmm6      ## rinv in xmm9, xmm10, xmm11    movapd %xmm9,%xmm1 ## copy of rinv    movapd %xmm10,%xmm4    movapd %xmm11,%xmm7    movapd nb214_krf(%rsp),%xmm2    mulpd  %xmm9,%xmm9  ## rinvsq    mulpd  %xmm10,%xmm10    mulpd  %xmm11,%xmm11    mulpd  %xmm2,%xmm0 ## k*rsq    mulpd  %xmm2,%xmm3    mulpd  %xmm2,%xmm6    movapd %xmm0,%xmm2 ## copy of k*rsq    movapd %xmm3,%xmm5    movapd %xmm6,%xmm8    addpd  %xmm1,%xmm2 ## rinv+krsq    addpd  %xmm4,%xmm5    addpd  %xmm7,%xmm8    movapd nb214_crf(%rsp),%xmm14    subpd  %xmm14,%xmm2  ## rinv+krsq-crf    subpd  %xmm14,%xmm5    subpd  %xmm14,%xmm8    movapd nb214_qqHH(%rsp),%xmm12    movapd nb214_qqMH(%rsp),%xmm13    mulpd  %xmm12,%xmm2 ## voul=qq*(rinv+ krsq-crf)    mulpd  %xmm12,%xmm5 ## voul=qq*(rinv+ krsq-crf)    mulpd  %xmm13,%xmm8 ## voul=qq*(rinv+ krsq-crf)    addpd  %xmm0,%xmm0 ## 2*krsq    addpd  %xmm3,%xmm3    addpd  %xmm6,%xmm6    subpd  %xmm0,%xmm1 ## rinv-2*krsq    subpd  %xmm3,%xmm4    subpd  %xmm6,%xmm7    mulpd  %xmm12,%xmm1  ## (rinv-2*krsq)*qq    mulpd  %xmm12,%xmm4    mulpd  %xmm13,%xmm7    addpd  nb214_vctot(%rsp),%xmm2    addpd  %xmm8,%xmm5    addpd  %xmm5,%xmm2    movapd %xmm2,nb214_vctot(%rsp)    mulpd  %xmm1,%xmm9  ## fscal    mulpd  %xmm4,%xmm10    mulpd  %xmm7,%xmm11    ## move j H1 forces to xmm0-xmm2        movlpd 24(%rdi,%rax,8),%xmm0        movlpd 32(%rdi,%rax,8),%xmm1        movlpd 40(%rdi,%rax,8),%xmm2        movhpd 24(%rdi,%rbx,8),%xmm0        movhpd 32(%rdi,%rbx,8),%xmm1        movhpd 40(%rdi,%rbx,8),%xmm2    movapd %xmm9,%xmm7    movapd %xmm9,%xmm8    movapd %xmm11,%xmm13    movapd %xmm11,%xmm14    movapd %xmm11,%xmm15    movapd %xmm10,%xmm11    movapd %xmm10,%xmm12        mulpd nb214_dxH1H1(%rsp),%xmm7        mulpd nb214_dyH1H1(%rsp),%xmm8        mulpd nb214_dzH1H1(%rsp),%xmm9        mulpd nb214_dxH2H1(%rsp),%xmm10        mulpd nb214_dyH2H1(%rsp),%xmm11        mulpd nb214_dzH2H1(%rsp),%xmm12        mulpd nb214_dxMH1(%rsp),%xmm13        mulpd nb214_dyMH1(%rsp),%xmm14        mulpd nb214_dzMH1(%rsp),%xmm15    addpd %xmm7,%xmm0    addpd %xmm8,%xmm1    addpd %xmm9,%xmm2    addpd nb214_fixH1(%rsp),%xmm7    addpd nb214_fiyH1(%rsp),%xmm8    addpd nb214_fizH1(%rsp),%xmm9    addpd %xmm10,%xmm0    addpd %xmm11,%xmm1    addpd %xmm12,%xmm2    addpd nb214_fixH2(%rsp),%xmm10    addpd nb214_fiyH2(%rsp),%xmm11    addpd nb214_fizH2(%rsp),%xmm12    addpd %xmm13,%xmm0    addpd %xmm14,%xmm1    addpd %xmm15,%xmm2    addpd nb214_fixM(%rsp),%xmm13    addpd nb214_fiyM(%rsp),%xmm14    addpd nb214_fizM(%rsp),%xmm15    movapd %xmm7,nb214_fixH1(%rsp)    movapd %xmm8,nb214_fiyH1(%rsp)    movapd %xmm9,nb214_fizH1(%rsp)    movapd %xmm10,nb214_fixH2(%rsp)    movapd %xmm11,nb214_fiyH2(%rsp)    movapd %xmm12,nb214_fizH2(%rsp)    movapd %xmm13,nb214_fixM(%rsp)    movapd %xmm14,nb214_fiyM(%rsp)    movapd %xmm15,nb214_fizM(%rsp)    ## store back j H1 forces from xmm0-xmm2        movlpd %xmm0,24(%rdi,%rax,8)        movlpd %xmm1,32(%rdi,%rax,8)        movlpd %xmm2,40(%rdi,%rax,8)        movhpd %xmm0,24(%rdi,%rbx,8)        movhpd %xmm1,32(%rdi,%rbx,8)        movhpd %xmm2,40(%rdi,%rbx,8)        ## move j H2 coordinates to local temp variables     movlpd 48(%rsi,%rax,8),%xmm0    movlpd 56(%rsi,%rax,8),%xmm1    movlpd 64(%rsi,%rax,8),%xmm2    movhpd 48(%rsi,%rbx,8),%xmm0    movhpd 56(%rsi,%rbx,8),%xmm1    movhpd 64(%rsi,%rbx,8),%xmm2    ## xmm0 = H2x    ## xmm1 = H2y    ## xmm2 = H2z    movapd %xmm0,%xmm3    movapd %xmm1,%xmm4    movapd %xmm2,%xmm5    movapd %xmm0,%xmm6    movapd %xmm1,%xmm7    movapd %xmm2,%xmm8    subpd nb214_ixH1(%rsp),%xmm0    subpd nb214_iyH1(%rsp),%xmm1    subpd nb214_izH1(%rsp),%xmm2    subpd nb214_ixH2(%rsp),%xmm3    subpd nb214_iyH2(%rsp),%xmm4    subpd nb214_izH2(%rsp),%xmm5    subpd nb214_ixM(%rsp),%xmm6    subpd nb214_iyM(%rsp),%xmm7    subpd nb214_izM(%rsp),%xmm8        movapd %xmm0,nb214_dxH1H2(%rsp)        movapd %xmm1,nb214_dyH1H2(%rsp)        movapd %xmm2,nb214_dzH1H2(%rsp)        mulpd  %xmm0,%xmm0        mulpd  %xmm1,%xmm1        mulpd  %xmm2,%xmm2        movapd %xmm3,nb214_dxH2H2(%rsp)        movapd %xmm4,nb214_dyH2H2(%rsp)        movapd %xmm5,nb214_dzH2H2(%rsp)        mulpd  %xmm3,%xmm3        mulpd  %xmm4,%xmm4        mulpd  %xmm5,%xmm5        movapd %xmm6,nb214_dxMH2(%rsp)        movapd %xmm7,nb214_dyMH2(%rsp)        movapd %xmm8,nb214_dzMH2(%rsp)        mulpd  %xmm6,%xmm6        mulpd  %xmm7,%xmm7        mulpd  %xmm8,%xmm8        addpd  %xmm1,%xmm0        addpd  %xmm2,%xmm0        addpd  %xmm4,%xmm3        addpd  %xmm5,%xmm3    addpd  %xmm7,%xmm6    addpd  %xmm8,%xmm6        ## start doing invsqrt for jH2 atoms    cvtpd2ps %xmm0,%xmm1    cvtpd2ps %xmm3,%xmm4    cvtpd2ps %xmm6,%xmm7        rsqrtps %xmm1,%xmm1        rsqrtps %xmm4,%xmm4    rsqrtps %xmm7,%xmm7    cvtps2pd %xmm1,%xmm1    cvtps2pd %xmm4,%xmm4    cvtps2pd %xmm7,%xmm7        movapd  %xmm1,%xmm2        movapd  %xmm4,%xmm5    movapd  %xmm7,%xmm8        mulpd   %xmm1,%xmm1 ## lu*lu        mulpd   %xmm4,%xmm4 ## lu*lu    mulpd   %xmm7,%xmm7 ## lu*lu        movapd  nb214_three(%rsp),%xmm9        movapd  %xmm9,%xmm10    movapd  %xmm9,%xmm11        mulpd   %xmm0,%xmm1 ## rsq*lu*lu        mulpd   %xmm3,%xmm4 ## rsq*lu*lu     mulpd   %xmm6,%xmm7 ## rsq*lu*lu        subpd   %xmm1,%xmm9        subpd   %xmm4,%xmm10    subpd   %xmm7,%xmm11 ## 3-rsq*lu*lu        mulpd   %xmm2,%xmm9        mulpd   %xmm5,%xmm10    mulpd   %xmm8,%xmm11 ## lu*(3-rsq*lu*lu)        movapd  nb214_half(%rsp),%xmm15        mulpd   %xmm15,%xmm9 ## first iteration for rinvH1H2         mulpd   %xmm15,%xmm10 ## first iteration for rinvH2H2    mulpd   %xmm15,%xmm11 ## first iteration for rinvMH2    ## second iteration step            movapd  %xmm9,%xmm2        movapd  %xmm10,%xmm5    movapd  %xmm11,%xmm8        mulpd   %xmm2,%xmm2 ## lu*lu        mulpd   %xmm5,%xmm5 ## lu*lu    mulpd   %xmm8,%xmm8 ## lu*lu        movapd  nb214_three(%rsp),%xmm1        movapd  %xmm1,%xmm4    movapd  %xmm1,%xmm7        mulpd   %xmm0,%xmm2 ## rsq*lu*lu        mulpd   %xmm3,%xmm5 ## rsq*lu*lu     mulpd   %xmm6,%xmm8 ## rsq*lu*lu        subpd   %xmm2,%xmm1        subpd   %xmm5,%xmm4    subpd   %xmm8,%xmm7 ## 3-rsq*lu*lu        mulpd   %xmm1,%xmm9        mulpd   %xmm4,%xmm10    mulpd   %xmm7,%xmm11 ## lu*(3-rsq*lu*lu)        movapd  nb214_half(%rsp),%xmm15        mulpd   %xmm15,%xmm9 ##  rinvH1H2        mulpd   %xmm15,%xmm10 ##   rinvH2H2    mulpd   %xmm15,%xmm11 ##   rinvMH2        ## H2 interactions     ## rsq in xmm0,xmm3,xmm6      ## rinv in xmm9, xmm10, xmm11    movapd %xmm9,%xmm1 ## copy of rinv    movapd %xmm10,%xmm4    movapd %xmm11,%xmm7    movapd nb214_krf(%rsp),%xmm2    mulpd  %xmm9,%xmm9  ## rinvsq    mulpd  %xmm10,%xmm10    mulpd  %xmm11,%xmm11    mulpd  %xmm2,%xmm0 ## k*rsq    mulpd  %xmm2,%xmm3    mulpd  %xmm2,%xmm6    movapd %xmm0,%xmm2 ## copy of k*rsq    movapd %xmm3,%xmm5    movapd %xmm6,%xmm8    addpd  %xmm1,%xmm2 ## rinv+krsq    addpd  %xmm4,%xmm5    addpd  %xmm7,%xmm8    movapd nb214_crf(%rsp),%xmm14    subpd  %xmm14,%xmm2  ## rinv+krsq-crf    subpd  %xmm14,%xmm5    subpd  %xmm14,%xmm8    movapd nb214_qqHH(%rsp),%xmm12    movapd nb214_qqMH(%rsp),%xmm13    mulpd  %xmm12,%xmm2 ## xmm6=voul=qq*(rinv+ krsq-crf)    mulpd  %xmm12,%xmm5 ## xmm6=voul=qq*(rinv+ krsq-crf)    mulpd  %xmm13,%xmm8 ## xmm6=voul=qq*(rinv+ krsq-crf)    addpd  %xmm0,%xmm0 ## 2*krsq    addpd  %xmm3,%xmm3    addpd  %xmm6,%xmm6

⌨️ 快捷键说明

复制代码Ctrl + C
搜索代码Ctrl + F
全屏模式F11
增大字号Ctrl + =
减小字号Ctrl + -
显示快捷键?