nb_kernel132_ia32_sse.intel_syntax.s

来自「最著名最快的分子模拟软件」· S 代码 · 共 2,176 行 · 第 1/5 页

S
2,176
字号
	leave	ret.globl nb_kernel132nf_ia32_sse.globl _nb_kernel132nf_ia32_ssenb_kernel132nf_ia32_sse:	_nb_kernel132nf_ia32_sse:	.equiv          nb132nf_p_nri,            8.equiv          nb132nf_iinr,             12.equiv          nb132nf_jindex,           16.equiv          nb132nf_jjnr,             20.equiv          nb132nf_shift,            24.equiv          nb132nf_shiftvec,         28.equiv          nb132nf_fshift,           32.equiv          nb132nf_gid,              36.equiv          nb132nf_pos,              40.equiv          nb132nf_faction,          44.equiv          nb132nf_charge,           48.equiv          nb132nf_p_facel,          52.equiv          nb132nf_argkrf,           56.equiv          nb132nf_argcrf,           60.equiv          nb132nf_Vc,               64.equiv          nb132nf_type,             68.equiv          nb132nf_p_ntype,          72.equiv          nb132nf_vdwparam,         76.equiv          nb132nf_Vvdw,             80.equiv          nb132nf_p_tabscale,       84.equiv          nb132nf_VFtab,            88.equiv          nb132nf_invsqrta,         92.equiv          nb132nf_dvda,             96.equiv          nb132nf_p_gbtabscale,     100.equiv          nb132nf_GBtab,            104.equiv          nb132nf_p_nthreads,       108.equiv          nb132nf_count,            112.equiv          nb132nf_mtx,              116.equiv          nb132nf_outeriter,        120.equiv          nb132nf_inneriter,        124.equiv          nb132nf_work,             128	;# stack offsets for local variables  	;# bottom of stack is cache-aligned for sse use .equiv          nb132nf_ixO,              0.equiv          nb132nf_iyO,              16.equiv          nb132nf_izO,              32.equiv          nb132nf_ixH1,             48.equiv          nb132nf_iyH1,             64.equiv          nb132nf_izH1,             80.equiv          nb132nf_ixH2,             96.equiv          nb132nf_iyH2,             112.equiv          nb132nf_izH2,             128.equiv          nb132nf_jxO,              144.equiv          nb132nf_jyO,              160.equiv          nb132nf_jzO,              176.equiv          nb132nf_jxH1,             192.equiv          nb132nf_jyH1,             208.equiv          nb132nf_jzH1,             224.equiv          nb132nf_jxH2,             240.equiv          nb132nf_jyH2,             256.equiv          nb132nf_jzH2,             272.equiv          nb132nf_qqOO,             288.equiv          nb132nf_qqOH,             304.equiv          nb132nf_qqHH,             320.equiv          nb132nf_c6,               336.equiv          nb132nf_c12,              352.equiv          nb132nf_tsc,              368.equiv          nb132nf_vctot,            384.equiv          nb132nf_Vvdwtot,          400.equiv          nb132nf_half,             416.equiv          nb132nf_three,            432.equiv          nb132nf_rsqOO,            448.equiv          nb132nf_rsqOH1,           464.equiv          nb132nf_rsqOH2,           480.equiv          nb132nf_rsqH1O,           496.equiv          nb132nf_rsqH1H1,          512.equiv          nb132nf_rsqH1H2,          528.equiv          nb132nf_rsqH2O,           544.equiv          nb132nf_rsqH2H1,          560.equiv          nb132nf_rsqH2H2,          576.equiv          nb132nf_rinvOO,           592.equiv          nb132nf_rinvOH1,          608.equiv          nb132nf_rinvOH2,          624.equiv          nb132nf_rinvH1O,          640.equiv          nb132nf_rinvH1H1,         656.equiv          nb132nf_rinvH1H2,         672.equiv          nb132nf_rinvH2O,          688.equiv          nb132nf_rinvH2H1,         704.equiv          nb132nf_rinvH2H2,         720.equiv          nb132nf_is3,              768.equiv          nb132nf_ii3,              772.equiv          nb132nf_innerjjnr,        776.equiv          nb132nf_innerk,           780.equiv          nb132nf_n,                784.equiv          nb132nf_nn1,              788.equiv          nb132nf_nri,              792.equiv          nb132nf_nouter,           796.equiv          nb132nf_ninner,           800.equiv          nb132nf_salign,           804	push ebp	mov ebp,esp	    	push eax    	push ebx    	push ecx    	push edx	push esi	push edi	sub esp, 808		;# local stack space 	mov  eax, esp	and  eax, 0xf	sub esp, eax	mov [esp + nb132nf_salign], eax	emms	;# Move args passed by reference to stack	mov ecx, [ebp + nb132nf_p_nri]	mov ecx, [ecx]	mov [esp + nb132nf_nri], ecx	;# zero iteration counters	mov eax, 0	mov [esp + nb132nf_nouter], eax	mov [esp + nb132nf_ninner], eax	mov eax, [ebp + nb132nf_p_tabscale]	movss xmm3, [eax]	shufps xmm3, xmm3, 0	movaps [esp + nb132nf_tsc], xmm3	;# assume we have at least one i particle - start directly 	mov   ecx, [ebp + nb132nf_iinr]       ;# ecx = pointer into iinr[] 		mov   ebx, [ecx]	    ;# ebx =ii 	mov   edx, [ebp + nb132nf_charge]	movss xmm3, [edx + ebx*4]		movss xmm4, xmm3		movss xmm5, [edx + ebx*4 + 4]		mov esi, [ebp + nb132nf_p_facel]	movss xmm6, [esi]	mulss  xmm3, xmm3	mulss  xmm4, xmm5	mulss  xmm5, xmm5	mulss  xmm3, xmm6	mulss  xmm4, xmm6	mulss  xmm5, xmm6	shufps xmm3, xmm3, 0	shufps xmm4, xmm4, 0	shufps xmm5, xmm5, 0	movaps [esp + nb132nf_qqOO], xmm3	movaps [esp + nb132nf_qqOH], xmm4	movaps [esp + nb132nf_qqHH], xmm5			xorps xmm0, xmm0	mov   edx, [ebp + nb132nf_type]	mov   ecx, [edx + ebx*4]	shl   ecx, 1	mov   edx, ecx	mov edi, [ebp + nb132nf_p_ntype]	imul  ecx, [edi]      ;# ecx = ntia = 2*ntype*type[ii0] 	add   edx, ecx	mov   eax, [ebp + nb132nf_vdwparam]	movlps xmm0, [eax + edx*4] 	movaps xmm1, xmm0	shufps xmm0, xmm0, 0	shufps xmm1, xmm1, 85  ;# constant 01010101	movaps [esp + nb132nf_c6], xmm0	movaps [esp + nb132nf_c12], xmm1	;# create constant floating-point factors on stack	mov eax, 0x3f000000     ;# constant 0.5 in IEEE (hex)	mov [esp + nb132nf_half], eax	movss xmm1, [esp + nb132nf_half]	shufps xmm1, xmm1, 0    ;# splat to all elements	movaps xmm2, xmm1       	addps  xmm2, xmm2	;# constant 1.0	movaps xmm3, xmm2	addps  xmm2, xmm2	;# constant 2.0	addps  xmm3, xmm2	;# constant 3.0	movaps [esp + nb132nf_half],  xmm1	movaps [esp + nb132nf_three],  xmm3.nb132nf_threadloop:        mov   esi, [ebp + nb132nf_count]          ;# pointer to sync counter        mov   eax, [esi].nb132nf_spinlock:        mov   ebx, eax                          ;# ebx=*count=nn0        add   ebx, 1                           ;# ebx=nn1=nn0+10        lock        cmpxchg [esi], ebx                      ;# write nn1 to *counter,                                                ;# if it hasnt changed.                                                ;# or reread *counter to eax.        pause                                   ;# -> better p4 performance        jnz .nb132nf_spinlock        ;# if(nn1>nri) nn1=nri        mov ecx, [esp + nb132nf_nri]        mov edx, ecx        sub ecx, ebx        cmovle ebx, edx                         ;# if(nn1>nri) nn1=nri        ;# Cleared the spinlock if we got here.        ;# eax contains nn0, ebx contains nn1.        mov [esp + nb132nf_n], eax        mov [esp + nb132nf_nn1], ebx        sub ebx, eax                            ;# calc number of outer lists	mov esi, eax				;# copy n to esi        jg  .nb132nf_outerstart        jmp .nb132nf_end.nb132nf_outerstart:	;# ebx contains number of outer iterations	add ebx, [esp + nb132nf_nouter]	mov [esp + nb132nf_nouter], ebx.nb132nf_outer:	mov   eax, [ebp + nb132nf_shift]      ;# eax = pointer into shift[] 	mov   ebx, [eax + esi*4]		;# ebx=shift[n] 		lea   ebx, [ebx + ebx*2]    ;# ebx=3*is 	mov   [esp + nb132nf_is3],ebx    	;# store is3 	mov   eax, [ebp + nb132nf_shiftvec]   ;# eax = base of shiftvec[] 	movss xmm0, [eax + ebx*4]	movss xmm1, [eax + ebx*4 + 4]	movss xmm2, [eax + ebx*4 + 8] 	mov   ecx, [ebp + nb132nf_iinr]       ;# ecx = pointer into iinr[] 		mov   ebx, [ecx + esi*4]	    ;# ebx =ii 	lea   ebx, [ebx + ebx*2]	;# ebx = 3*ii=ii3 	mov   eax, [ebp + nb132nf_pos]    ;# eax = base of pos[]  	mov   [esp + nb132nf_ii3], ebx			movaps xmm3, xmm0	movaps xmm4, xmm1	movaps xmm5, xmm2	addss xmm3, [eax + ebx*4]	addss xmm4, [eax + ebx*4 + 4]	addss xmm5, [eax + ebx*4 + 8]			shufps xmm3, xmm3, 0	shufps xmm4, xmm4, 0	shufps xmm5, xmm5, 0	movaps [esp + nb132nf_ixO], xmm3	movaps [esp + nb132nf_iyO], xmm4	movaps [esp + nb132nf_izO], xmm5	movss xmm3, xmm0	movss xmm4, xmm1	movss xmm5, xmm2	addss xmm0, [eax + ebx*4 + 12]	addss xmm1, [eax + ebx*4 + 16]	addss xmm2, [eax + ebx*4 + 20]			addss xmm3, [eax + ebx*4 + 24]	addss xmm4, [eax + ebx*4 + 28]	addss xmm5, [eax + ebx*4 + 32]			shufps xmm0, xmm0, 0	shufps xmm1, xmm1, 0	shufps xmm2, xmm2, 0	shufps xmm3, xmm3, 0	shufps xmm4, xmm4, 0	shufps xmm5, xmm5, 0	movaps [esp + nb132nf_ixH1], xmm0	movaps [esp + nb132nf_iyH1], xmm1	movaps [esp + nb132nf_izH1], xmm2	movaps [esp + nb132nf_ixH2], xmm3	movaps [esp + nb132nf_iyH2], xmm4	movaps [esp + nb132nf_izH2], xmm5	;# clear vctot	xorps xmm4, xmm4	movaps [esp + nb132nf_vctot], xmm4	movaps [esp + nb132nf_Vvdwtot], xmm4		mov   eax, [ebp + nb132nf_jindex]	mov   ecx, [eax + esi*4]	     ;# jindex[n] 	mov   edx, [eax + esi*4 + 4]	     ;# jindex[n+1] 	sub   edx, ecx               ;# number of innerloop atoms 	mov   esi, [ebp + nb132nf_pos]	mov   eax, [ebp + nb132nf_jjnr]	shl   ecx, 2	add   eax, ecx	mov   [esp + nb132nf_innerjjnr], eax     ;# pointer to jjnr[nj0] 	mov   ecx, edx	sub   edx,  4	add   ecx, [esp + nb132nf_ninner]	mov   [esp + nb132nf_ninner], ecx	add   edx, 0	mov   [esp + nb132nf_innerk], edx    ;# number of innerloop atoms 	jge   .nb132nf_unroll_loop	jmp   .nb132nf_single_check.nb132nf_unroll_loop:		;# quad-unroll innerloop here 	mov   edx, [esp + nb132nf_innerjjnr]     ;# pointer to jjnr[k] 	mov   eax, [edx]		mov   ebx, [edx + 4] 	mov   ecx, [edx + 8]	mov   edx, [edx + 12]         ;# eax-edx=jnr1-4 		add dword ptr [esp + nb132nf_innerjjnr],  16 ;# advance pointer (unrolled 4) 	mov esi, [ebp + nb132nf_pos]       ;# base of pos[] 	lea   eax, [eax + eax*2]     ;# replace jnr with j3 	lea   ebx, [ebx + ebx*2]		lea   ecx, [ecx + ecx*2]     ;# replace jnr with j3 	lea   edx, [edx + edx*2]			;# move j coordinates to local temp variables 	movlps xmm2, [esi + eax*4]	movlps xmm3, [esi + eax*4 + 12]	movlps xmm4, [esi + eax*4 + 24]	movlps xmm5, [esi + ebx*4]	movlps xmm6, [esi + ebx*4 + 12]	movlps xmm7, [esi + ebx*4 + 24]	movhps xmm2, [esi + ecx*4]	movhps xmm3, [esi + ecx*4 + 12]	movhps xmm4, [esi + ecx*4 + 24]	movhps xmm5, [esi + edx*4]	movhps xmm6, [esi + edx*4 + 12]	movhps xmm7, [esi + edx*4 + 24]	;# current state: 		;# xmm2= jxOa  jyOa  jxOc  jyOc 	;# xmm3= jxH1a jyH1a jxH1c jyH1c 	;# xmm4= jxH2a jyH2a jxH2c jyH2c 	;# xmm5= jxOb  jyOb  jxOd  jyOd 	;# xmm6= jxH1b jyH1b jxH1d jyH1d 	;# xmm7= jxH2b jyH2b jxH2d jyH2d 		movaps xmm0, xmm2	movaps xmm1, xmm3	unpcklps xmm0, xmm5	;# xmm0= jxOa  jxOb  jyOa  jyOb 	unpcklps xmm1, xmm6	;# xmm1= jxH1a jxH1b jyH1a jyH1b 	unpckhps xmm2, xmm5	;# xmm2= jxOc  jxOd  jyOc  jyOd 	unpckhps xmm3, xmm6	;# xmm3= jxH1c jxH1d jyH1c jyH1d 	movaps xmm5, xmm4	movaps   xmm6, xmm0	unpcklps xmm4, xmm7	;# xmm4= jxH2a jxH2b jyH2a jyH2b 			unpckhps xmm5, xmm7	;# xmm5= jxH2c jxH2d jyH2c jyH2d 	movaps   xmm7, xmm1	movlhps  xmm0, xmm2	;# xmm0= jxOa  jxOb  jxOc  jxOd 	movaps [esp + nb132nf_jxO], xmm0	movhlps  xmm2, xmm6	;# xmm2= jyOa  jyOb  jyOc  jyOd 	movaps [esp + nb132nf_jyO], xmm2	movlhps  xmm1, xmm3	movaps [esp + nb132nf_jxH1], xmm1	movhlps  xmm3, xmm7	movaps   xmm6, xmm4	movaps [esp + nb132nf_jyH1], xmm3	movlhps  xmm4, xmm5	movaps [esp + nb132nf_jxH2], xmm4	movhlps  xmm5, xmm6	movaps [esp + nb132nf_jyH2], xmm5	movss  xmm0, [esi + eax*4 + 8]	movss  xmm1, [esi + eax*4 + 20]	movss  xmm2, [esi + eax*4 + 32]	movss  xmm3, [esi + ecx*4 + 8]	movss  xmm4, [esi + ecx*4 + 20]	movss  xmm5, [esi + ecx*4 + 32]	movhps xmm0, [esi + ebx*4 + 4]	movhps xmm1, [esi + ebx*4 + 16]	movhps xmm2, [esi + ebx*4 + 28]		movhps xmm3, [esi + edx*4 + 4]	movhps xmm4, [esi + edx*4 + 16]	movhps xmm5, [esi + edx*4 + 28]		shufps xmm0, xmm3, 204  ;# constant 11001100	shufps xmm1, xmm4, 204  ;# constant 11001100	shufps xmm2, xmm5, 204  ;# constant 11001100	movaps [esp + nb132nf_jzO],  xmm0	movaps [esp + nb132nf_jzH1],  xmm1	movaps [esp + nb132nf_jzH2],  xmm2	movaps xmm0, [esp + nb132nf_ixO]	movaps xmm1, [esp + nb132nf_iyO]	movaps xmm2, [esp + nb132nf_izO]	movaps xmm3, [esp + nb132nf_ixO]	movaps xmm4, [esp + nb132nf_iyO]	movaps xmm5, [esp + nb132nf_izO]	subps  xmm0, [esp + nb132nf_jxO]	subps  xmm1, [esp + nb132nf_jyO]	subps  xmm2, [esp + nb132nf_jzO]	subps  xmm3, [esp + nb132nf_jxH1]	subps  xmm4, [esp + nb132nf_jyH1]	subps  xmm5, [esp + nb132nf_jzH1]	mulps  xmm0, xmm0	mulps  xmm1, xmm1	mulps  xmm2, xmm2	mulps  xmm3, xmm3	mulps  xmm4, xmm4	mulps  xmm5, xmm5	addps  xmm0, xmm1	addps  xmm0, xmm2	addps  xmm3, xmm4	addps  xmm3, xmm5	movaps [esp + nb132nf_rsqOO], xmm0	movaps [esp + nb132nf_rsqOH1], xmm3	movaps xmm0, [esp + nb132nf_ixO]	movaps xmm1, [esp + nb132nf_iyO]	movaps xmm2, [esp + nb132nf_izO]	movaps xmm3, [esp + nb132nf_ixH1]	movaps xmm4, [esp + nb132nf_iyH1]	movaps xmm5, [esp + nb132nf_izH1]	subps  xmm0, [esp + nb132nf_jxH2]	subps  xmm1, [esp + nb132nf_jyH2]	subps  xmm2, [esp + nb132nf_jzH2]	subps  xmm3, [esp + nb132nf_jxO]	subps  xmm4, [esp + nb132nf_jyO]	subps  xmm5, [esp + nb132nf_jzO]	mulps  xmm0, xmm0	mulps  xmm1, xmm1	mulps  xmm2, xmm2	mulps  xmm3, xmm3	mulps  xmm4, xmm4	mulps  xmm5, xmm5	addps  xmm0, xmm1	addps  xmm0, xmm2	addps  xmm3, xmm4	addps  xmm3, xmm5	movaps [esp + nb132nf_rsqOH2], xmm0	movaps [esp + nb132nf_rsqH1O], xmm3

⌨️ 快捷键说明

复制代码Ctrl + C
搜索代码Ctrl + F
全屏模式F11
增大字号Ctrl + =
减小字号Ctrl + -
显示快捷键?