nb_kernel313_x86_64_sse.intel_syntax.s

来自「最著名最快的分子模拟软件」· S 代码 · 共 2,333 行 · 第 1/5 页

S
2,333
字号
	movaps  xmm3, xmm2	mulps   xmm2, xmm2	movaps  xmm0, [rsp + nb313nf_three]	mulps   xmm2, xmm5	;# rsq*lu*lu 	subps   xmm0, xmm2	;# 30-rsq*lu*lu 	mulps   xmm0, xmm3	;# lu*(3-rsq*lu*lu) 	mulps   xmm0, [rsp + nb313nf_half]	movaps  [rsp + nb313nf_rinvH2], xmm0	;# rinvH2 in xmm4 	mulps   xmm5, xmm0	movaps  [rsp + nb313nf_rH2], xmm5	;# rsqM - seed to xmm2 	rsqrtps xmm2, xmm4	movaps  xmm3, xmm2	mulps   xmm2, xmm2	movaps  xmm0, [rsp + nb313nf_three]	mulps   xmm2, xmm4	;# rsq*lu*lu 	subps   xmm0, xmm2	;# 30-rsq*lu*lu 	mulps   xmm0, xmm3	;# lu*(3-rsq*lu*lu) 	mulps   xmm0, [rsp + nb313nf_half]	movaps  [rsp + nb313nf_rinvM], xmm0	;# rinvM in xmm5 	mulps   xmm4, xmm0	movaps  [rsp + nb313nf_rM], xmm4			;# Do the O LJ-only interaction directly.		rcpps   xmm2, xmm7	movaps  xmm1, [rsp + nb313nf_two]	mulps   xmm7, xmm2	subps   xmm1, xmm7	mulps   xmm2, xmm1 ;# rinvsq 	movaps  xmm0, xmm2	mulps   xmm0, xmm2  	;# r4	mulps   xmm0, xmm2 	;# r6	movaps  xmm1, xmm0	mulps   xmm1, xmm1  	;# r12	mulps   xmm0, [rsp + nb313nf_c6]	mulps   xmm1, [rsp + nb313nf_c12]	movaps  xmm3, xmm1	subps   xmm3, xmm0  	;# Vvdw12-Vvdw6	addps   xmm3, [rsp + nb313nf_Vvdwtot]	movaps  [rsp + nb313nf_Vvdwtot], xmm3	;# Do H1 interaction	mov  rsi, [rbp + nb313nf_VFtab]		movaps xmm7, [rsp + nb313nf_rH1]	mulps   xmm7, [rsp + nb313nf_tsc]	movhlps xmm4, xmm7	cvttps2pi mm6, xmm7	cvttps2pi mm7, xmm4	;# mm6/mm7 contain lu indices 		cvtpi2ps xmm3, mm6	cvtpi2ps xmm4, mm7	movlhps xmm3, xmm4		subps xmm7, xmm3	movaps xmm1, xmm7	;# xmm1=eps 	movaps xmm2, xmm1	mulps  xmm2, xmm2	;# xmm2=eps2 	pslld mm6, 2	pslld mm7, 2	movd mm0, eax	movd mm1, ebx	movd mm2, ecx	movd mm3, edx		movd eax, mm6	psrlq mm6, 32	movd ecx, mm7	psrlq mm7, 32	movd ebx, mm6	movd edx, mm7	movlps xmm5, [rsi + rax*4]	movlps xmm7, [rsi + rcx*4]	movhps xmm5, [rsi + rbx*4]	movhps xmm7, [rsi + rdx*4] ;# got half coulomb table 	movaps xmm4, xmm5	shufps xmm4, xmm7, 136  ;# 10001000	shufps xmm5, xmm7, 221  ;# 11011101	movlps xmm7, [rsi + rax*4 + 8]	movlps xmm3, [rsi + rcx*4 + 8]	movhps xmm7, [rsi + rbx*4 + 8]	movhps xmm3, [rsi + rdx*4 + 8] ;# other half of coulomb table  	movaps xmm6, xmm7	shufps xmm6, xmm3, 136  ;# 10001000	shufps xmm7, xmm3, 221  ;# 11011101	;# coulomb table ready, in xmm4-xmm7              	mulps  xmm6, xmm1   	;# xmm6=Geps 	mulps  xmm7, xmm2   	;# xmm7=Heps2 	addps  xmm5, xmm6	addps  xmm5, xmm7   	;# xmm5=Fp        	movaps xmm0, [rsp + nb313nf_qqH]	mulps  xmm5, xmm1 ;# xmm5=eps*Fp 	addps  xmm5, xmm4 ;# xmm5=VV 	mulps  xmm5, xmm0 ;# vcoul=qq*VV  	;# at this point mm5 contains vcoul 	;# increment vcoul 	addps  xmm5, [rsp + nb313nf_vctot]	movaps [rsp + nb313nf_vctot], xmm5 	;# Done with H1, do H2 interactions 	movaps xmm7, [rsp + nb313nf_rH2]	mulps   xmm7, [rsp + nb313nf_tsc]	movhlps xmm4, xmm7	cvttps2pi mm6, xmm7	cvttps2pi mm7, xmm4	;# mm6/mm7 contain lu indices 		cvtpi2ps xmm3, mm6	cvtpi2ps xmm4, mm7	movlhps xmm3, xmm4		subps xmm7, xmm3	movaps xmm1, xmm7	;# xmm1=eps 	movaps xmm2, xmm1	mulps  xmm2, xmm2	;# xmm2=eps2 	pslld mm6, 2	pslld mm7, 2		movd eax, mm6	psrlq mm6, 32	movd ecx, mm7	psrlq mm7, 32	movd ebx, mm6	movd edx, mm7	movlps xmm5, [rsi + rax*4]	movlps xmm7, [rsi + rcx*4]	movhps xmm5, [rsi + rbx*4]	movhps xmm7, [rsi + rdx*4] ;# got half coulomb table 	movaps xmm4, xmm5	shufps xmm4, xmm7, 136  ;# shuffle 10001000	shufps xmm5, xmm7, 221  ;# shuffle 11011101	movlps xmm7, [rsi + rax*4 + 8]	movlps xmm3, [rsi + rcx*4 + 8]	movhps xmm7, [rsi + rbx*4 + 8]	movhps xmm3, [rsi + rdx*4 + 8] ;# other half of coulomb table  	movaps xmm6, xmm7	shufps xmm6, xmm3, 136  ;# shuf 10001000	shufps xmm7, xmm3, 221  ;# shuf 11011101	;# coulomb table ready, in xmm4-xmm7              	mulps  xmm6, xmm1   	;# xmm6=Geps 	mulps  xmm7, xmm2   	;# xmm7=Heps2 	addps  xmm5, xmm6	addps  xmm5, xmm7   	;# xmm5=Fp        	movaps xmm0, [rsp + nb313nf_qqH]	mulps  xmm5, xmm1 ;# xmm5=eps*Fp 	addps  xmm5, xmm4 ;# xmm5=VV 	mulps  xmm5, xmm0 ;# vcoul=qq*VV  	;# at this point mm5 contains vcoul 	;# increment vcoul 	addps  xmm5, [rsp + nb313nf_vctot]	movaps [rsp + nb313nf_vctot], xmm5 	;# Done with H2, do M interactions 	movaps xmm7, [rsp + nb313nf_rM]	mulps   xmm7, [rsp + nb313nf_tsc]	movhlps xmm4, xmm7	cvttps2pi mm6, xmm7	cvttps2pi mm7, xmm4	;# mm6/mm7 contain lu indices 		cvtpi2ps xmm3, mm6	cvtpi2ps xmm4, mm7	movlhps xmm3, xmm4		subps xmm7, xmm3	movaps xmm1, xmm7	;# xmm1=eps 	movaps xmm2, xmm1	mulps  xmm2, xmm2	;# xmm2=eps2 	pslld mm6, 2	pslld mm7, 2		movd eax, mm6	psrlq mm6, 32	movd ecx, mm7	psrlq mm7, 32	movd ebx, mm6	movd edx, mm7	movlps xmm5, [rsi + rax*4]	movlps xmm7, [rsi + rcx*4]	movhps xmm5, [rsi + rbx*4]	movhps xmm7, [rsi + rdx*4] ;# got half coulomb table 	movaps xmm4, xmm5	shufps xmm4, xmm7, 136  ;# 10001000	shufps xmm5, xmm7, 221  ;# 11011101	movlps xmm7, [rsi + rax*4 + 8]	movlps xmm3, [rsi + rcx*4 + 8]	movhps xmm7, [rsi + rbx*4 + 8]	movhps xmm3, [rsi + rdx*4 + 8] ;# other half of coulomb table  	movaps xmm6, xmm7	shufps xmm6, xmm3, 136  ;# 10001000	shufps xmm7, xmm3, 221  ;# 11011101	;# coulomb table ready, in xmm4-xmm7              	mulps  xmm6, xmm1   	;# xmm6=Geps 	mulps  xmm7, xmm2   	;# xmm7=Heps2 	addps  xmm5, xmm6	addps  xmm5, xmm7   	;# xmm5=Fp        	movaps xmm0, [rsp + nb313nf_qqM]	mulps  xmm5, xmm1 ;# xmm5=eps*Fp 	addps  xmm5, xmm4 ;# xmm5=VV 	mulps  xmm5, xmm0 ;# vcoul=qq*VV  	;# at this point mm5 contains vcoul 	;# increment vcoul 	addps  xmm5, [rsp + nb313nf_vctot]	movaps [rsp + nb313nf_vctot], xmm5 	;# should we do one more iteration? 	sub dword ptr [rsp + nb313nf_innerk],  4	jl    .nb313nf_odd_inner	jmp   .nb313nf_unroll_loop.nb313nf_odd_inner:		add dword ptr [rsp + nb313nf_innerk],  4	jnz   .nb313nf_odd_loop	jmp   .nb313nf_updateouterdata.nb313nf_odd_loop:	mov   rdx, [rsp + nb313nf_innerjjnr] 	;# pointer to jjnr[k] 	mov   eax, [rdx]		add qword ptr [rsp + nb313nf_innerjjnr],  4	 	xorps xmm4, xmm4  	;# clear reg.	movss xmm4, [rsp + nb313nf_iqM]	mov rsi, [rbp + nb313nf_charge] 	movhps xmm4, [rsp + nb313nf_iqH]  ;# [qM  0  qH  qH] 	shufps xmm4, xmm4, 41	;# [0 qH qH qM]	movss xmm3, [rsi + rax*4]	;# charge in xmm3 	shufps xmm3, xmm3, 0	mulps xmm3, xmm4	movaps [rsp + nb313nf_qqM], xmm3	;# use dummy qq for storage 		xorps xmm6, xmm6	mov rsi, [rbp + nb313nf_type]	mov ebx, [rsi + rax*4]	mov rsi, [rbp + nb313nf_vdwparam]	shl ebx, 1		add ebx, [rsp + nb313nf_ntia]	movlps xmm6, [rsi + rbx*4]	movaps xmm7, xmm6	shufps xmm6, xmm6, 252  ;# 11111100	shufps xmm7, xmm7, 253  ;# 11111101	movaps [rsp + nb313nf_c6], xmm6	movaps [rsp + nb313nf_c12], xmm7		mov rsi, [rbp + nb313nf_pos]	lea rax, [rax + rax*2]  	movss xmm3, [rsp + nb313nf_ixO]	movss xmm4, [rsp + nb313nf_iyO]	movss xmm5, [rsp + nb313nf_izO]	movss xmm0, [rsp + nb313nf_ixH1]	movss xmm1, [rsp + nb313nf_iyH1]	movss xmm2, [rsp + nb313nf_izH1]	unpcklps xmm3, [rsp + nb313nf_ixH2] 	;# ixO ixH2 - -	unpcklps xmm4, [rsp + nb313nf_iyH2]  	;# iyO iyH2 - -	unpcklps xmm5, [rsp + nb313nf_izH2]	;# izO izH2 - -	unpcklps xmm0, [rsp + nb313nf_ixM] 	;# ixH1 ixM - -	unpcklps xmm1, [rsp + nb313nf_iyM]  	;# iyH1 iyM - -	unpcklps xmm2, [rsp + nb313nf_izM]	;# izH1 izM - -	unpcklps xmm3, xmm0  	;# ixO ixH1 ixH2 ixM	unpcklps xmm4, xmm1 	;# same for y	unpcklps xmm5, xmm2 	;# same for z		;# move j coords to xmm0-xmm2 	movss xmm0, [rsi + rax*4]	movss xmm1, [rsi + rax*4 + 4]	movss xmm2, [rsi + rax*4 + 8]	shufps xmm0, xmm0, 0	shufps xmm1, xmm1, 0	shufps xmm2, xmm2, 0		subps xmm3, xmm0	subps xmm4, xmm1	subps xmm5, xmm2	mulps  xmm3, xmm3	mulps  xmm4, xmm4	mulps  xmm5, xmm5	addps  xmm4, xmm3	addps  xmm4, xmm5	;# rsq in xmm4 	rsqrtps xmm5, xmm4	;# lookup seed in xmm5 	movaps xmm2, xmm5	mulps xmm5, xmm5	movaps xmm1, [rsp + nb313nf_three]	mulps xmm5, xmm4	;# rsq*lu*lu 				movaps xmm0, [rsp + nb313nf_half]	subps xmm1, xmm5	;# 30-rsq*lu*lu 	mulps xmm1, xmm2		mulps xmm0, xmm1	;# xmm0=rinv		movaps [rsp + nb313nf_rinvM], xmm0	mulps  xmm4, xmm0  	;# r	 	mulps xmm4, [rsp + nb313nf_tsc]	movhlps xmm7, xmm4	cvttps2pi mm6, xmm4	cvttps2pi mm7, xmm7	;# mm6/mm7 contain lu indices 	cvtpi2ps xmm3, mm6	cvtpi2ps xmm7, mm7	movlhps xmm3, xmm7	subps   xmm4, xmm3		movaps xmm1, xmm4	;# xmm1=eps 	movaps xmm2, xmm1	mulps  xmm2, xmm2	;# xmm2=eps2 	pslld mm6, 2	pslld mm7, 2	mov  rsi, [rbp + nb313nf_VFtab]    	psrlq mm6, 32	movd ebx, mm6	movd ecx, mm7	psrlq mm7, 32	movd edx, mm7	xorps  xmm5, xmm5	movlps xmm3, [rsi + rcx*4]	;# data: Y3 F3  -  - 	movhps xmm5, [rsi + rbx*4]	;# data:  0  0 Y2 F2	movhps xmm3, [rsi + rdx*4]      ;# data: Y3 F3 Y4 F4 	movaps xmm4, xmm5		;# data:  0  0 Y2 F2 	shufps xmm4, xmm3, 0x88		;# data:  0 Y2 Y3 Y3	shufps xmm5, xmm3, 0xDD	        ;# data:  0 F2 F3 F4 	xorps  xmm7, xmm7	movlps xmm3, [rsi + rcx*4 + 8]	;# data: G3 H3  -  - 	movhps xmm7, [rsi + rbx*4 + 8]	;# data:  0  0 G2 H2	movhps xmm3, [rsi + rdx*4 + 8]  ;# data: G3 H3 G4 H4 	movaps xmm6, xmm7		;# data:  0  0 G2 H2 	shufps xmm6, xmm3, 0x88		;# data:  0 G2 G3 G3	shufps xmm7, xmm3, 0xDD	        ;# data:  0 H2 H3 H4 	;# xmm4 =  0  Y2 Y3 Y4	;# xmm5 =  0  F2 F3 F4	;# xmm6 =  0  G2 G3 G4	;# xmm7 =  0  H2 H3 H4		;# coulomb table ready, in xmm4-xmm7      	mulps  xmm6, xmm1   	;# xmm6=Geps 	mulps  xmm7, xmm2   	;# xmm7=Heps2 	addps  xmm5, xmm6	addps  xmm5, xmm7   	;# xmm5=Fp        	movaps xmm0, [rsp + nb313nf_qqM]	mulps  xmm5, xmm1 ;# xmm5=eps*Fp 	addps  xmm5, xmm4 ;# xmm5=VV 	mulps  xmm5, xmm0 ;# vcoul=qq*VV  	;# at this point mm5 contains vcoul 	;# increment vcoul 	addps  xmm5, [rsp + nb313nf_vctot]	movaps [rsp + nb313nf_vctot], xmm5		;# do nontable L-J  in first element only.	movaps xmm2, [rsp + nb313nf_rinvM]	mulss  xmm2, xmm2	movaps xmm1, xmm2	mulss  xmm1, xmm1	mulss  xmm1, xmm2	;# xmm1=rinvsix	xorps  xmm4, xmm4	movss  xmm4, xmm1	mulss  xmm4, xmm4	;# xmm4=rinvtwelve 	mulss  xmm1, [rsp + nb313nf_c6]	mulss  xmm4, [rsp + nb313nf_c12]	movaps xmm3, xmm4	subss  xmm3, xmm1	;# xmm3=Vvdw12-Vvdw6	addss  xmm3, [rsp + nb313nf_Vvdwtot]	movss [rsp + nb313nf_Vvdwtot], xmm3	dec dword ptr [rsp + nb313nf_innerk]	jz    .nb313nf_updateouterdata	jmp   .nb313nf_odd_loop.nb313nf_updateouterdata:	;# get n from stack	mov esi, [rsp + nb313nf_n]        ;# get group index for i particle         mov   rdx, [rbp + nb313nf_gid]      	;# base of gid[]        mov   edx, [rdx + rsi*4]		;# ggid=gid[n]	;# accumulate total potential energy and update it 	movaps xmm7, [rsp + nb313nf_vctot]	;# accumulate 	movhlps xmm6, xmm7	addps  xmm7, xmm6	;# pos 0-1 in xmm7 have the sum now 	movaps xmm6, xmm7	shufps xmm6, xmm6, 1	addss  xmm7, xmm6		        	;# add earlier value from mem 	mov   rax, [rbp + nb313nf_Vc]	addss xmm7, [rax + rdx*4] 	;# move back to mem 	movss [rax + rdx*4], xmm7 		;# accumulate total lj energy and update it 	movaps xmm7, [rsp + nb313nf_Vvdwtot]	;# accumulate 	movhlps xmm6, xmm7	addps  xmm7, xmm6	;# pos 0-1 in xmm7 have the sum now 	movaps xmm6, xmm7	shufps xmm6, xmm6, 1	addss  xmm7, xmm6			;# add earlier value from mem 	mov   rax, [rbp + nb313nf_Vvdw]	addss xmm7, [rax + rdx*4] 	;# move back to mem 	movss [rax + rdx*4], xmm7 	        ;# finish if last         mov ecx, [rsp + nb313nf_nn1]	;# esi already loaded with n	inc esi        sub ecx, esi        jz .nb313nf_outerend        ;# not last, iterate outer loop once more!          mov [rsp + nb313nf_n], esi        jmp .nb313nf_outer.nb313nf_outerend:        ;# check if more outer neighborlists remain        mov   ecx, [rsp + nb313nf_nri]	;# esi already loaded with n above        sub   ecx, esi        jz .nb313nf_end        ;# non-zero, do one more workunit        jmp   .nb313nf_threadloop.nb313nf_end:	mov eax, [rsp + nb313nf_nouter]	mov ebx, [rsp + nb313nf_ninner]	mov rcx, [rbp + nb313nf_outeriter]	mov rdx, [rbp + nb313nf_inneriter]	mov [rcx], eax	mov [rdx], ebx	add rsp, 584	emms        pop r15        pop r14        pop r13        pop r12	pop rbx	pop	rbp	ret		

⌨️ 快捷键说明

复制代码Ctrl + C
搜索代码Ctrl + F
全屏模式F11
增大字号Ctrl + =
减小字号Ctrl + -
显示快捷键?