nb_kernel430_ia32_sse.intel_syntax.s

来自「最著名最快的分子模拟软件」· S 代码 · 共 2,411 行 · 第 1/5 页

S
2,411
字号
	shufps xmm3, xmm7, 0xEE 	shufps xmm6, xmm7, 0x44	movaps xmm7, xmm4	shufps xmm7, xmm5, 0xEE	shufps xmm4, xmm5, 0x44	movaps xmm5, xmm4	shufps xmm5, xmm6, 0xDD	shufps xmm4, xmm6, 0x88	movaps xmm6, xmm7	shufps xmm6, xmm3, 0x88	shufps xmm7, xmm3, 0xDD	;# table ready, in xmm4-xmm7 		mulps  xmm6, xmm1	;# xmm6=Geps 	mulps  xmm7, xmm2	;# xmm7=Heps2 	addps  xmm5, xmm6	addps  xmm5, xmm7	;# xmm5=Fp 		mulps  xmm5, xmm1 ;# xmm5=eps*Fp 	addps  xmm5, xmm4 ;# xmm5=VV  		mulps  xmm5, [esp + nb430nf_c12] ;# Vvdw12	addps  xmm5, [esp + nb430nf_Vvdwtot]	movaps [esp + nb430nf_Vvdwtot], xmm5			;# should we do one more iteration? 	sub dword ptr [esp + nb430nf_innerk],  4	jl    .nb430nf_finish_inner	jmp   .nb430nf_unroll_loop.nb430nf_finish_inner:	;# check if at least two particles remain 	add dword ptr [esp + nb430nf_innerk],  4	mov   edx, [esp + nb430nf_innerk]	and   edx, 2	jnz   .nb430nf_dopair	jmp   .nb430nf_checksingle.nb430nf_dopair:		mov   ecx, [esp + nb430nf_innerjjnr]		mov   eax, [ecx]		mov   ebx, [ecx + 4]              	add dword ptr [esp + nb430nf_innerjjnr],  8		xorps xmm2, xmm2	movaps xmm6, xmm2		;# load isa2	mov esi, [ebp + nb430nf_invsqrta]	movss xmm2, [esi + eax*4]	movss xmm3, [esi + ebx*4]	unpcklps xmm2, xmm3	;# isa2 in xmm3(0,1)	mulps  xmm2, [esp + nb430nf_isai]	movaps [esp + nb430nf_isaprod], xmm2		movaps xmm1, xmm2	mulps xmm1, [esp + nb430nf_gbtsc]	movaps [esp + nb430nf_gbscale], xmm1			mov esi, [ebp + nb430nf_charge]    ;# base of charge[] 		movss xmm3, [esi + eax*4]			movss xmm6, [esi + ebx*4]	unpcklps xmm3, xmm6 ;# constant 00001000 ;# xmm3(0,1) has the charges 	mulps  xmm2, [esp + nb430nf_iq]	mulps  xmm3, xmm2	movaps [esp + nb430nf_qq], xmm3	mov esi, [ebp + nb430nf_type]	mov   ecx, eax	mov   edx, ebx	mov ecx, [esi + ecx*4]	mov edx, [esi + edx*4]		mov esi, [ebp + nb430nf_vdwparam]	shl ecx, 1		shl edx, 1		mov edi, [esp + nb430nf_ntia]	add ecx, edi	add edx, edi	movlps xmm6, [esi + ecx*4]	movhps xmm6, [esi + edx*4]	mov edi, [ebp + nb430nf_pos]			movaps xmm4, xmm6	shufps xmm4, xmm4, 8 ;# constant 00001000 		shufps xmm6, xmm6, 13 ;# constant 00001101	movlhps xmm4, xmm7	movlhps xmm6, xmm7		movaps [esp + nb430nf_c6], xmm4	movaps [esp + nb430nf_c12], xmm6					lea   eax, [eax + eax*2]	lea   ebx, [ebx + ebx*2]	;# move coordinates to xmm0-xmm2 	movlps xmm1, [edi + eax*4]	movss xmm2, [edi + eax*4 + 8]		movhps xmm1, [edi + ebx*4]	movss xmm0, [edi + ebx*4 + 8]		movlhps xmm3, xmm7		shufps xmm2, xmm0, 0		movaps xmm0, xmm1	shufps xmm2, xmm2, 136  ;# constant 10001000		shufps xmm0, xmm0, 136  ;# constant 10001000	shufps xmm1, xmm1, 221  ;# constant 11011101				mov    edi, [ebp + nb430nf_faction]	;# move ix-iz to xmm4-xmm6 	xorps   xmm7, xmm7		movaps xmm4, [esp + nb430nf_ix]	movaps xmm5, [esp + nb430nf_iy]	movaps xmm6, [esp + nb430nf_iz]	;# calc dr 	subps xmm4, xmm0	subps xmm5, xmm1	subps xmm6, xmm2	;# square it 	mulps xmm4,xmm4	mulps xmm5,xmm5	mulps xmm6,xmm6	addps xmm4, xmm5	addps xmm4, xmm6	;# rsq in xmm4 	rsqrtps xmm5, xmm4	;# lookup seed in xmm5 	movaps xmm2, xmm5	mulps xmm5, xmm5	movaps xmm1, [esp + nb430nf_three]	mulps xmm5, xmm4	;# rsq*lu*lu 				movaps xmm0, [esp + nb430nf_half]	subps xmm1, xmm5	;# constant 30-rsq*lu*lu 	mulps xmm1, xmm2		mulps xmm0, xmm1	;# xmm0=rinv 	mulps xmm4, xmm0	;# xmm4=r 	movaps [esp + nb430nf_r], xmm4	mulps xmm4, [esp + nb430nf_gbscale]	cvttps2pi mm6, xmm4     ;# mm6 contain lu indices 	cvtpi2ps xmm6, mm6	subps xmm4, xmm6		movaps xmm1, xmm4	;# xmm1=eps 	movaps xmm2, xmm1		mulps  xmm2, xmm2	;# xmm2=eps2 	pslld mm6, 2	mov  esi, [ebp + nb430nf_GBtab]	movd ecx, mm6	psrlq mm6, 32	movd edx, mm6	;# load coulomb table	movaps xmm4, [esi + ecx*4]	movaps xmm7, [esi + edx*4]	;# transpose, using xmm3 for scratch	movaps xmm6, xmm4	unpcklps xmm4, xmm7  	;# Y1 Y2 F1 F2 	unpckhps xmm6, xmm7     ;# G1 G2 H1 H2	movhlps  xmm5, xmm4    	;# F1 F2 	movhlps  xmm7, xmm6     ;# H1 H2	;# coulomb table ready, in xmm4-xmm7  		mulps  xmm6, xmm1	;# xmm6=Geps 	mulps  xmm7, xmm2	;# xmm7=Heps2 	addps  xmm5, xmm6	addps  xmm5, xmm7	;# xmm5=Fp 		movaps xmm3, [esp + nb430nf_qq]	mulps  xmm5, xmm1 ;# xmm5=eps*Fp 	addps  xmm5, xmm4 ;# xmm5=VV 	mulps  xmm5, xmm3 ;# vcoul=qq*VV  	addps  xmm5, [esp + nb430nf_vctot]	movaps [esp + nb430nf_vctot], xmm5 	movaps xmm4, [esp + nb430nf_r]	mulps xmm4, [esp + nb430nf_tsc]		cvttps2pi mm6, xmm4	cvtpi2ps xmm6, mm6	subps xmm4, xmm6		movaps xmm1, xmm4	;# xmm1=eps 	movaps xmm2, xmm1		mulps  xmm2, xmm2	;# xmm2=eps2 	pslld mm6, 3		mov  esi, [ebp + nb430nf_VFtab]	movd ecx, mm6	psrlq mm6, 32	movd edx, mm6				;# dispersion 	movaps xmm4, [esi + ecx*4]	movaps xmm7, [esi + edx*4]	;# transpose, using xmm3 for scratch	movaps xmm6, xmm4	unpcklps xmm4, xmm7  	;# Y1 Y2 F1 F2 	unpckhps xmm6, xmm7     ;# G1 G2 H1 H2	movhlps  xmm5, xmm4    	;# F1 F2 	movhlps  xmm7, xmm6     ;# H1 H2	;# dispersion table ready, in xmm4-xmm7 		mulps  xmm6, xmm1	;# xmm6=Geps 	mulps  xmm7, xmm2	;# xmm7=Heps2 	addps  xmm5, xmm6	addps  xmm5, xmm7	;# xmm5=Fp 		mulps  xmm5, xmm1 ;# xmm5=eps*Fp 	addps  xmm5, xmm4 ;# xmm5=VV 	mulps  xmm5, [esp + nb430nf_c6]	 ;# Vvdw6 	addps  xmm5, [esp + nb430nf_Vvdwtot]	movaps [esp + nb430nf_Vvdwtot], xmm5	;# repulsion 	movaps xmm4, [esi + ecx*4 + 16]	movaps xmm7, [esi + edx*4 + 16]	;# transpose, using xmm3 for scratch	movaps xmm6, xmm4	unpcklps xmm4, xmm7  	;# Y1 Y2 F1 F2 	unpckhps xmm6, xmm7     ;# G1 G2 H1 H2	movhlps  xmm5, xmm4    	;# F1 F2 	movhlps  xmm7, xmm6     ;# H1 H2	;# table ready, in xmm4-xmm7 		mulps  xmm6, xmm1	;# xmm6=Geps 	mulps  xmm7, xmm2	;# xmm7=Heps2 	addps  xmm5, xmm6	addps  xmm5, xmm7	;# xmm5=Fp 		mulps  xmm5, xmm1 ;# xmm5=eps*Fp 	addps  xmm5, xmm4 ;# xmm5=VV  		mulps  xmm5, [esp + nb430nf_c12] ;# Vvdw12 		addps  xmm5, [esp + nb430nf_Vvdwtot]	movaps [esp + nb430nf_Vvdwtot], xmm5.nb430nf_checksingle:					mov   edx, [esp + nb430nf_innerk]	and   edx, 1	jnz    .nb430nf_dosingle	jmp    .nb430nf_updateouterdata.nb430nf_dosingle:	mov esi, [ebp + nb430nf_charge]	mov edx, [ebp + nb430nf_invsqrta]	mov edi, [ebp + nb430nf_pos]	mov   ecx, [esp + nb430nf_innerjjnr]	mov   eax, [ecx]		xorps  xmm2, xmm2	movaps xmm6, xmm2	movss xmm2, [edx + eax*4]	;# isa2	mulss xmm2, [esp + nb430nf_isai]	movss [esp + nb430nf_isaprod], xmm2		movss xmm1, xmm2	mulss xmm1, [esp + nb430nf_gbtsc]	movss [esp + nb430nf_gbscale], xmm1			mulss  xmm2, [esp + nb430nf_iq]	movss xmm6, [esi + eax*4]	;# xmm6(0) has the charge 		mulss  xmm6, xmm2	movss [esp + nb430nf_qq], xmm6			mov esi, [ebp + nb430nf_type]	mov ecx, eax	mov ecx, [esi + ecx*4]		mov esi, [ebp + nb430nf_vdwparam]	shl ecx, 1	add ecx, [esp + nb430nf_ntia]	movlps xmm6, [esi + ecx*4]	movaps xmm4, xmm6	shufps xmm4, xmm4, 252  ;# constant 11111100		shufps xmm6, xmm6, 253  ;# constant 11111101					movss [esp + nb430nf_c6], xmm4	movss [esp + nb430nf_c12], xmm6				lea   eax, [eax + eax*2]		;# move coordinates to xmm0-xmm2 	movss xmm0, [edi + eax*4]		movss xmm1, [edi + eax*4 + 4]		movss xmm2, [edi + eax*4 + 8]	 		movss xmm4, [esp + nb430nf_ix]	movss xmm5, [esp + nb430nf_iy]	movss xmm6, [esp + nb430nf_iz]	;# calc dr 	subss xmm4, xmm0	subss xmm5, xmm1	subss xmm6, xmm2	;# square it 	mulss xmm4,xmm4	mulss xmm5,xmm5	mulss xmm6,xmm6	addss xmm4, xmm5	addss xmm4, xmm6	;# rsq in xmm4 	rsqrtss xmm5, xmm4	;# lookup seed in xmm5 	movaps xmm2, xmm5	mulss xmm5, xmm5	movss xmm1, [esp + nb430nf_three]	mulss xmm5, xmm4	;# rsq*lu*lu 				movss xmm0, [esp + nb430nf_half]	subss xmm1, xmm5	;# constant 30-rsq*lu*lu 	mulss xmm1, xmm2		mulss xmm0, xmm1	;# xmm0=rinv 	mulss xmm4, xmm0	;# xmm4=r 	movaps [esp + nb430nf_r], xmm4	mulss xmm4, [esp + nb430nf_gbscale]	cvttss2si ebx, xmm4     ;# mm6 contain lu indices 	cvtsi2ss xmm6, ebx	subss xmm4, xmm6		movaps xmm1, xmm4	;# xmm1=eps 	movaps xmm2, xmm1		mulss  xmm2, xmm2	;# xmm2=eps2 	shl ebx, 2	mov  esi, [ebp + nb430nf_GBtab]							movaps xmm4, [esi + ebx*4]		movhlps xmm6, xmm4	movaps xmm5, xmm4	movaps xmm7, xmm6	shufps xmm5, xmm5, 1	shufps xmm7, xmm7, 1	;# table ready in xmm4-xmm7 	mulss  xmm6, xmm1	;# xmm6=Geps 	mulss  xmm7, xmm2	;# xmm7=Heps2 	addss  xmm5, xmm6	addss  xmm5, xmm7	;# xmm5=Fp 		movss xmm3, [esp + nb430nf_qq]	mulss  xmm5, xmm1 ;# xmm5=eps*Fp 	addss  xmm5, xmm4 ;# xmm5=VV 	mulss  xmm5, xmm3 ;# vcoul=qq*VV  	addss  xmm5, [esp + nb430nf_vctot]	movss [esp + nb430nf_vctot], xmm5		movss xmm4, [esp + nb430nf_r]	mulps xmm4, [esp + nb430nf_tsc]		cvttss2si ebx, xmm4	cvtsi2ss xmm6, ebx	subss xmm4, xmm6		movss xmm1, xmm4	;# xmm1=eps 	movss xmm2, xmm1		mulss  xmm2, xmm2	;# xmm2=eps2 	shl ebx, 3	mov  esi, [ebp + nb430nf_VFtab]				;# dispersion 	movaps xmm4, [esi + ebx*4]		movhlps xmm6, xmm4	movaps xmm5, xmm4	movaps xmm7, xmm6	shufps xmm5, xmm5, 1	shufps xmm7, xmm7, 1	;# table ready in xmm4-xmm7 		mulss  xmm6, xmm1	;# xmm6=Geps 	mulss  xmm7, xmm2	;# xmm7=Heps2 	addss  xmm5, xmm6	addss  xmm5, xmm7	;# xmm5=Fp 		mulss  xmm5, xmm1 ;# xmm5=eps*Fp 	addss  xmm5, xmm4 ;# xmm5=VV 	mulss  xmm5, [esp + nb430nf_c6]	 ;# Vvdw6	addss  xmm5, [esp + nb430nf_Vvdwtot]	movss [esp + nb430nf_Vvdwtot], xmm5	;# repulsion 	movaps xmm4, [esi + ebx*4 + 16]		movhlps xmm6, xmm4	movaps xmm5, xmm4	movaps xmm7, xmm6	shufps xmm5, xmm5, 1	shufps xmm7, xmm7, 1	;# table ready in xmm4-xmm7 		mulss  xmm6, xmm1	;# xmm6=Geps 	mulss  xmm7, xmm2	;# xmm7=Heps2 	addss  xmm5, xmm6	addss  xmm5, xmm7	;# xmm5=Fp 		mulss  xmm5, xmm1 ;# xmm5=eps*Fp 	addss  xmm5, xmm4 ;# xmm5=VV  		mulss  xmm5, [esp + nb430nf_c12] ;# Vvdw12 		addss  xmm5, [esp + nb430nf_Vvdwtot]	movss [esp + nb430nf_Vvdwtot], xmm5.nb430nf_updateouterdata:	;# get n from stack	mov esi, [esp + nb430nf_n]        ;# get group index for i particle         mov   edx, [ebp + nb430nf_gid]      	;# base of gid[]        mov   edx, [edx + esi*4]		;# ggid=gid[n]	;# accumulate total potential energy and update it 	movaps xmm7, [esp + nb430nf_vctot]	;# accumulate 	movhlps xmm6, xmm7	addps  xmm7, xmm6	;# pos 0-1 in xmm7 have the sum now 	movaps xmm6, xmm7	shufps xmm6, xmm6, 1	addss  xmm7, xmm6			;# add earlier value from mem 	mov   eax, [ebp + nb430nf_Vc]	addss xmm7, [eax + edx*4] 	;# move back to mem 	movss [eax + edx*4], xmm7 		;# accumulate total lj energy and update it 	movaps xmm7, [esp + nb430nf_Vvdwtot]	;# accumulate 	movhlps xmm6, xmm7	addps  xmm7, xmm6	;# pos 0-1 in xmm7 have the sum now 	movaps xmm6, xmm7	shufps xmm6, xmm6, 1	addss  xmm7, xmm6			;# add earlier value from mem 	mov   eax, [ebp + nb430nf_Vvdw]	addss xmm7, [eax + edx*4] 	;# move back to mem 	movss [eax + edx*4], xmm7 	        ;# finish if last         mov ecx, [esp + nb430nf_nn1]	;# esi already loaded with n	inc esi        sub ecx, esi        jecxz .nb430nf_outerend        ;# not last, iterate outer loop once more!          mov [esp + nb430nf_n], esi        jmp .nb430nf_outer.nb430nf_outerend:        ;# check if more outer neighborlists remain        mov   ecx, [esp + nb430nf_nri]	;# esi already loaded with n above        sub   ecx, esi        jecxz .nb430nf_end        ;# non-zero, do one more workunit        jmp   .nb430nf_threadloop.nb430nf_end:	emms	mov eax, [esp + nb430nf_nouter]	mov ebx, [esp + nb430nf_ninner]	mov ecx, [ebp + nb430nf_outeriter]	mov edx, [ebp + nb430nf_inneriter]	mov [ecx], eax	mov [edx], ebx	mov eax, [esp + nb430nf_salign]	add esp, eax	add esp, 324	pop edi	pop esi    	pop edx    	pop ecx    	pop ebx    	pop eax	leave	ret	

⌨️ 快捷键说明

复制代码Ctrl + C
搜索代码Ctrl + F
全屏模式F11
增大字号Ctrl + =
减小字号Ctrl + -
显示快捷键?