nb_kernel133_ia32_sse.intel_syntax.s

来自「最著名最快的分子模拟软件」· S 代码 · 共 2,185 行 · 第 1/4 页

S
2,185
字号
		movhlps xmm3, xmm0	movhlps xmm4, xmm1	movhlps xmm5, xmm2	addps   xmm3, xmm0	addps   xmm4, xmm1	addps   xmm5, xmm2	movaps  xmm0, xmm3	movaps  xmm1, xmm4	movaps  xmm2, xmm5			shufps xmm3, xmm3, 0x39	;# shift right 	shufps xmm4, xmm4, 0x39	shufps xmm5, xmm5, 0x39	addss  xmm0, xmm3	addss  xmm1, xmm4	addss  xmm2, xmm5	unpcklps xmm0, xmm1 	;# x,y sum in xmm0, z sum in xmm2		subps    xmm6, xmm0	subss    xmm7, xmm2		movlps [edi + eax*4],     xmm6	movss  [edi + eax*4 + 8], xmm7	dec dword ptr [esp + nb133_innerk]	jz    .nb133_updateouterdata	jmp   .nb133_odd_loop.nb133_updateouterdata:	mov   ecx, [esp + nb133_ii3]	mov   edi, [ebp + nb133_faction]	mov   esi, [ebp + nb133_fshift]	mov   edx, [esp + nb133_is3]	;# accumulate  Oi forces in xmm0, xmm1, xmm2 	movaps xmm0, [esp + nb133_fixO]	movaps xmm1, [esp + nb133_fiyO]	movaps xmm2, [esp + nb133_fizO]	movhlps xmm3, xmm0	movhlps xmm4, xmm1	movhlps xmm5, xmm2	addps  xmm0, xmm3	addps  xmm1, xmm4	addps  xmm2, xmm5 ;# sum is in 1/2 in xmm0-xmm2 	movaps xmm3, xmm0		movaps xmm4, xmm1		movaps xmm5, xmm2		shufps xmm3, xmm3, 1	shufps xmm4, xmm4, 1	shufps xmm5, xmm5, 1	addss  xmm0, xmm3	addss  xmm1, xmm4	addss  xmm2, xmm5	;# xmm0-xmm2 has single force in pos0 	;# increment i force 	movss  xmm3, [edi + ecx*4]	movss  xmm4, [edi + ecx*4 + 4]	movss  xmm5, [edi + ecx*4 + 8]	addss  xmm3, xmm0	addss  xmm4, xmm1	addss  xmm5, xmm2	movss  [edi + ecx*4],     xmm3	movss  [edi + ecx*4 + 4], xmm4	movss  [edi + ecx*4 + 8], xmm5	;# accumulate force in xmm6/xmm7 for fshift 	movaps xmm6, xmm0	movss xmm7, xmm2	movlhps xmm6, xmm1	shufps  xmm6, xmm6, 8 ;# constant 00001000		;# accumulate H1i forces in xmm0, xmm1, xmm2 	movaps xmm0, [esp + nb133_fixH1]	movaps xmm1, [esp + nb133_fiyH1]	movaps xmm2, [esp + nb133_fizH1]	movhlps xmm3, xmm0	movhlps xmm4, xmm1	movhlps xmm5, xmm2	addps  xmm0, xmm3	addps  xmm1, xmm4	addps  xmm2, xmm5 ;# sum is in 1/2 in xmm0-xmm2 	movaps xmm3, xmm0		movaps xmm4, xmm1		movaps xmm5, xmm2		shufps xmm3, xmm3, 1	shufps xmm4, xmm4, 1	shufps xmm5, xmm5, 1	addss  xmm0, xmm3	addss  xmm1, xmm4	addss  xmm2, xmm5	;# xmm0-xmm2 has single force in pos0 	;# increment i force 	movss  xmm3, [edi + ecx*4 + 12]	movss  xmm4, [edi + ecx*4 + 16]	movss  xmm5, [edi + ecx*4 + 20]	addss  xmm3, xmm0	addss  xmm4, xmm1	addss  xmm5, xmm2	movss  [edi + ecx*4 + 12], xmm3	movss  [edi + ecx*4 + 16], xmm4	movss  [edi + ecx*4 + 20], xmm5	;# accumulate force in xmm6/xmm7 for fshift 	addss xmm7, xmm2	movlhps xmm0, xmm1	shufps  xmm0, xmm0, 8 ;# constant 00001000		addps   xmm6, xmm0	;# accumulate H2i forces in xmm0, xmm1, xmm2 	movaps xmm0, [esp + nb133_fixH2]	movaps xmm1, [esp + nb133_fiyH2]	movaps xmm2, [esp + nb133_fizH2]	movhlps xmm3, xmm0	movhlps xmm4, xmm1	movhlps xmm5, xmm2	addps  xmm0, xmm3	addps  xmm1, xmm4	addps  xmm2, xmm5 ;# sum is in 1/2 in xmm0-xmm2 	movaps xmm3, xmm0		movaps xmm4, xmm1		movaps xmm5, xmm2		shufps xmm3, xmm3, 1	shufps xmm4, xmm4, 1	shufps xmm5, xmm5, 1	addss  xmm0, xmm3	addss  xmm1, xmm4	addss  xmm2, xmm5	;# xmm0-xmm2 has single force in pos0 	;# increment i force 	movss  xmm3, [edi + ecx*4 + 24]	movss  xmm4, [edi + ecx*4 + 28]	movss  xmm5, [edi + ecx*4 + 32]	addss  xmm3, xmm0	addss  xmm4, xmm1	addss  xmm5, xmm2	movss  [edi + ecx*4 + 24], xmm3	movss  [edi + ecx*4 + 28], xmm4	movss  [edi + ecx*4 + 32], xmm5	;# accumulate force in xmm6/xmm7 for fshift 	addss xmm7, xmm2	movlhps xmm0, xmm1	shufps  xmm0, xmm0, 8 ;# constant 00001000		addps   xmm6, xmm0	;# accumulate Mi forces in xmm0, xmm1, xmm2 	movaps xmm0, [esp + nb133_fixM]	movaps xmm1, [esp + nb133_fiyM]	movaps xmm2, [esp + nb133_fizM]	movhlps xmm3, xmm0	movhlps xmm4, xmm1	movhlps xmm5, xmm2	addps  xmm0, xmm3	addps  xmm1, xmm4	addps  xmm2, xmm5 ;# sum is in 1/2 in xmm0-xmm2 	movaps xmm3, xmm0		movaps xmm4, xmm1		movaps xmm5, xmm2		shufps xmm3, xmm3, 1	shufps xmm4, xmm4, 1	shufps xmm5, xmm5, 1	addss  xmm0, xmm3	addss  xmm1, xmm4	addss  xmm2, xmm5	;# xmm0-xmm2 has single force in pos0 	;# increment i force 	movss  xmm3, [edi + ecx*4 + 36]	movss  xmm4, [edi + ecx*4 + 40]	movss  xmm5, [edi + ecx*4 + 44]	addss  xmm3, xmm0	addss  xmm4, xmm1	addss  xmm5, xmm2	movss  [edi + ecx*4 + 36], xmm3	movss  [edi + ecx*4 + 40], xmm4	movss  [edi + ecx*4 + 44], xmm5	;# accumulate force in xmm6/xmm7 for fshift 	addss xmm7, xmm2	movlhps xmm0, xmm1	shufps  xmm0, xmm0, 8 ;# constant 00001000		addps   xmm6, xmm0	;# increment fshift force  	movlps  xmm3, [esi + edx*4]	movss  xmm4, [esi + edx*4 + 8]	addps  xmm3, xmm6	addss  xmm4, xmm7	movlps  [esi + edx*4],    xmm3	movss  [esi + edx*4 + 8], xmm4	;# get n from stack	mov esi, [esp + nb133_n]        ;# get group index for i particle         mov   edx, [ebp + nb133_gid]      	;# base of gid[]        mov   edx, [edx + esi*4]		;# ggid=gid[n]	;# accumulate total potential energy and update it 	movaps xmm7, [esp + nb133_vctot]	;# accumulate 	movhlps xmm6, xmm7	addps  xmm7, xmm6	;# pos 0-1 in xmm7 have the sum now 	movaps xmm6, xmm7	shufps xmm6, xmm6, 1	addss  xmm7, xmm6		        	;# add earlier value from mem 	mov   eax, [ebp + nb133_Vc]	addss xmm7, [eax + edx*4] 	;# move back to mem 	movss [eax + edx*4], xmm7 		;# accumulate total lj energy and update it 	movaps xmm7, [esp + nb133_Vvdwtot]	;# accumulate 	movhlps xmm6, xmm7	addps  xmm7, xmm6	;# pos 0-1 in xmm7 have the sum now 	movaps xmm6, xmm7	shufps xmm6, xmm6, 1	addss  xmm7, xmm6			;# add earlier value from mem 	mov   eax, [ebp + nb133_Vvdw]	addss xmm7, [eax + edx*4] 	;# move back to mem 	movss [eax + edx*4], xmm7 	        ;# finish if last         mov ecx, [esp + nb133_nn1]	;# esi already loaded with n	inc esi        sub ecx, esi        jecxz .nb133_outerend        ;# not last, iterate outer loop once more!          mov [esp + nb133_n], esi        jmp .nb133_outer.nb133_outerend:        ;# check if more outer neighborlists remain        mov   ecx, [esp + nb133_nri]	;# esi already loaded with n above        sub   ecx, esi        jecxz .nb133_end        ;# non-zero, do one more workunit        jmp   .nb133_threadloop.nb133_end:	emms	mov eax, [esp + nb133_nouter]	mov ebx, [esp + nb133_ninner]	mov ecx, [ebp + nb133_outeriter]	mov edx, [ebp + nb133_inneriter]	mov [ecx], eax	mov [edx], ebx	mov eax, [esp + nb133_salign]	add esp, eax	add esp, 1004	pop edi	pop esi	pop edx	pop ecx	pop ebx	pop eax	leave	ret	.globl nb_kernel133nf_ia32_sse.globl _nb_kernel133nf_ia32_ssenb_kernel133nf_ia32_sse:	_nb_kernel133nf_ia32_sse:	.equiv          nb133nf_p_nri,            8.equiv          nb133nf_iinr,             12.equiv          nb133nf_jindex,           16.equiv          nb133nf_jjnr,             20.equiv          nb133nf_shift,            24.equiv          nb133nf_shiftvec,         28.equiv          nb133nf_fshift,           32.equiv          nb133nf_gid,              36.equiv          nb133nf_pos,              40.equiv          nb133nf_faction,          44.equiv          nb133nf_charge,           48.equiv          nb133nf_p_facel,          52.equiv          nb133nf_argkrf,           56.equiv          nb133nf_argcrf,           60.equiv          nb133nf_Vc,               64.equiv          nb133nf_type,             68.equiv          nb133nf_p_ntype,          72.equiv          nb133nf_vdwparam,         76.equiv          nb133nf_Vvdw,             80.equiv          nb133nf_p_tabscale,       84.equiv          nb133nf_VFtab,            88.equiv          nb133nf_invsqrta,         92.equiv          nb133nf_dvda,             96.equiv          nb133nf_p_gbtabscale,     100.equiv          nb133nf_GBtab,            104.equiv          nb133nf_p_nthreads,       108.equiv          nb133nf_count,            112.equiv          nb133nf_mtx,              116.equiv          nb133nf_outeriter,        120.equiv          nb133nf_inneriter,        124.equiv          nb133nf_work,             128	;# stack offsets for local variables  	;# bottom of stack is cache-aligned for sse use .equiv          nb133nf_ixO,              0.equiv          nb133nf_iyO,              16.equiv          nb133nf_izO,              32.equiv          nb133nf_ixH1,             48.equiv          nb133nf_iyH1,             64.equiv          nb133nf_izH1,             80.equiv          nb133nf_ixH2,             96.equiv          nb133nf_iyH2,             112.equiv          nb133nf_izH2,             128.equiv          nb133nf_ixM,              144.equiv          nb133nf_iyM,              160.equiv          nb133nf_izM,              176.equiv          nb133nf_iqM,              192.equiv          nb133nf_iqH,              208.equiv          nb133nf_qqH,              224.equiv          nb133nf_rinvH1,           240.equiv          nb133nf_rinvH2,           256.equiv          nb133nf_rinvM,            272.equiv          nb133nf_c6,               288.equiv          nb133nf_c12,              304.equiv          nb133nf_tsc,              320.equiv          nb133nf_vctot,            416.equiv          nb133nf_Vvdwtot,          432.equiv          nb133nf_half,             448.equiv          nb133nf_three,            464.equiv          nb133nf_qqM,              480.equiv          nb133nf_is3,              496.equiv          nb133nf_ii3,              500.equiv          nb133nf_ntia,             504.equiv          nb133nf_innerjjnr,        508.equiv          nb133nf_innerk,           512.equiv          nb133nf_n,                516.equiv          nb133nf_nn1,              520.equiv          nb133nf_nri,              524.equiv          nb133nf_nouter,           528.equiv          nb133nf_ninner,           532.equiv          nb133nf_salign,           536	push ebp	mov ebp,esp		push eax	push ebx	push ecx	push edx	push esi	push edi	sub esp, 540		;# local stack space 	mov  eax, esp	and  eax, 0xf	sub esp, eax	mov [esp + nb133nf_salign], eax	emms	;# Move args passed by reference to stack	mov ecx, [ebp + nb133nf_p_nri]	mov ecx, [ecx]	mov [esp + nb133nf_nri], ecx	;# zero iteration counters	mov eax, 0	mov [esp + nb133nf_nouter], eax	mov [esp + nb133nf_ninner], eax	mov eax, [ebp + nb133nf_p_tabscale]	movss xmm3, [eax]	shufps xmm3, xmm3, 0	movaps [esp + nb133nf_tsc], xmm3	;# create constant floating-point factors on stack	mov eax, 0x3f000000     ;# constant 0.5 in IEEE (hex)	mov [esp + nb133nf_half], eax	movss xmm1, [esp + nb133nf_half]	shufps xmm1, xmm1, 0    ;# splat to all elements	movaps xmm2, xmm1       	addps  xmm2, xmm2	;# constant 1.0	movaps xmm3, xmm2	addps  xmm2, xmm2	;# constant 2.0	addps  xmm3, xmm2	;# constant 3.0	movaps [esp + nb133nf_half],  xmm1	movaps [esp + nb133nf_three],  xmm3		;# assume we have at least one i particle - start directly 	mov   ecx, [ebp + nb133nf_iinr]       ;# ecx = pointer into iinr[] 		mov   ebx, [ecx]	    ;# ebx =ii 	mov   edx, [ebp + nb133nf_charge]	movss xmm4, [edx + ebx*4 + 4]		movss xmm3, [edx + ebx*4 + 12]		mov esi, [ebp + nb133nf_p_facel]	movss xmm5, [esi]	mulss  xmm3, xmm5	mulss  xmm4, xmm5	shufps xmm3, xmm3, 0	shufps xmm4, xmm4, 0	movaps [esp + nb133nf_iqM], xmm3	movaps [esp + nb133nf_iqH], xmm4		mov   edx, [ebp + nb133nf_type]	mov   ecx, [edx + ebx*4]	shl   ecx, 1	mov edi, [ebp + nb133nf_p_ntype]	imul  ecx, [edi]      ;# ecx = ntia = 2*ntype*type[ii0] 	mov   [esp + nb133nf_ntia], ecx		.nb133nf_threadloop:        mov   esi, [ebp + nb133nf_count]          ;# pointer to sync counter        mov   eax, [esi].nb133nf_spinlock:        mov   ebx, eax                          ;# ebx=*count=nn0        add   ebx, 1                           ;# ebx=nn1=nn0+10        lock        cmpxchg [esi], ebx                      ;# write nn1 to *counter,                                                ;# if it hasnt changed.                                                ;# or reread *counter to eax.        pause                                   ;# -> better p4 performance        jnz .nb133nf_spinlock        ;# if(nn1>nri) nn1=nri        mov ecx, [esp + nb133nf_nri]        mov edx, ecx        sub ecx, ebx        cmovle ebx, edx                         ;# if(nn1>nri) nn1=nri        ;# Cleared the spinlock if we got here.        ;# eax contains nn0, ebx contains nn1.        mov [esp + nb133nf_n], eax        mov [esp + nb133nf_nn1], ebx        sub ebx, eax                            ;# calc number of outer lists	mov esi, eax				;# copy n to esi        jg  .nb133nf_outerstart        jmp .nb133nf_end	.nb133nf_outerstart:	;# ebx contains number of outer iterations	add ebx, [esp + nb133nf_nouter]	mov [esp + nb133nf_nouter], ebx.nb133nf_outer:	mov   eax, [ebp + nb133nf_shift]      ;# eax = pointer into shift[] 	mov   ebx, [eax + esi*4]		;# ebx=shift[n] 		lea   ebx, [ebx + ebx*2]	;# ebx=3*is 	mov   [esp + nb133nf_is3],ebx    	;# store is3 	mov   eax, [ebp + nb133nf_shiftvec]   ;# eax = base of shiftvec[] 	movss xmm0, [eax + ebx*4]	movss xmm1, [eax + ebx*4 + 4]	movss xmm2, [eax + ebx*4 + 8] 	mov   ecx, [ebp + nb133nf_iinr]   	;# ecx = pointer into iinr[] 		mov   ebx, [ecx + esi*4]		;# ebx =ii 	movaps xmm3, xmm0	movaps xmm4, xmm1	movaps xmm5, xmm2		movaps xmm6, xmm0	movaps xmm7, xmm1		lea   ebx, [ebx + ebx*2]	;# ebx = 3*ii=ii3 	mov   eax, [ebp + nb133nf_pos]	;# eax = base of pos[]  	mov   [esp + nb133nf_ii3], ebx	addss xmm3, [eax + ebx*4]  	;# ox	addss xmm4, [eax + ebx*4 + 4]  ;# oy	addss xmm5, [eax + ebx*4 + 8]  ;# oz	addss xmm6, [eax + ebx*4 + 12] ;# h1x	addss xmm7, [eax + ebx*4 + 16] ;# h1y	shufps xmm3, xmm3, 0	shufps xmm4, xmm4, 0	shufps xmm5, xmm5, 0	shufps xmm6, xmm6, 0	shufps xmm7, xmm7, 0	movaps [esp + nb133nf_ixO], xmm3	movaps [esp + nb133nf_iyO], xmm4	movaps [esp + nb133nf_izO], xmm5	movaps [esp + nb133nf_ixH1], xmm6	movaps [esp + nb133nf_iyH1], xmm7	movss xmm6, xmm2	movss xmm3, xmm0	movss xmm4, xmm1	movss xmm5, xmm2	addss xmm6, [eax + ebx*4 + 20] ;# h1z	addss xmm0, [eax + ebx*4 + 24] ;# h2x	addss xmm1, [eax + ebx*4 + 28] ;# h2y	addss xmm2, [eax + ebx*4 + 32] ;# h2z	addss xmm3, [eax + ebx*4 + 36] ;# mx	addss xmm4, [eax + ebx*4 + 40] ;# my	addss xmm5, [eax + ebx*4 + 44] ;# mz	shufps xmm6, xmm6, 0	shufps xmm0, xmm0, 0	shufps xmm1, xmm1, 0	shufps xmm2, xmm2, 0	shufps xmm3, xmm3, 0	shufps xmm4, xmm4, 0	shufps xmm5, xmm5, 0	movaps [esp + nb133nf_izH1], xmm6	movaps [esp + nb133nf_ixH2], xmm0	movaps [esp + nb133nf_iyH2], xmm1	movaps [esp + nb133nf_izH2], xmm2	movaps [esp + nb133nf_ixM], xmm3	movaps [esp + nb133nf_iyM], xmm4	movaps [esp + nb133nf_izM], xmm5		;# clear vctot 	xorps xmm4, xmm4	movaps [esp + nb133nf_vctot], xmm4	movaps [esp + nb133nf_Vvdwtot], xmm4	mov   eax, [ebp + nb133nf_jindex]	mov   ecx, [eax + esi*4]	 	;# jindex[n] 	mov   edx, [eax + esi*4 + 4]	 	;# jindex[n+1] 	sub   edx, ecx           	;# number of innerloop atoms 	mov   esi, [ebp + nb133nf_pos]	mov   eax, [ebp + nb133nf_jjnr]	shl   ecx, 2	add   eax, ecx	mov   [esp + nb133nf_innerjjnr], eax 	;# pointer to jjnr[nj0] 	mov   ecx, edx	sub   edx,  4	add   ecx, [esp + nb133nf_ninner]	mov   [esp + nb133nf_ninner], ecx	add   edx, 0	mov   [esp + nb133nf_innerk], edx	;# number of innerloop atoms 	jge   .nb133nf_unroll_loop	jmp   .nb133nf_odd_inner.nb133nf_unroll_loop:	;# quad-unroll innerloop here 

⌨️ 快捷键说明

复制代码Ctrl + C
搜索代码Ctrl + F
全屏模式F11
增大字号Ctrl + =
减小字号Ctrl + -
显示快捷键?