nb_kernel430_ia32_sse.intel_syntax.s
来自「最著名最快的分子模拟软件」· S 代码 · 共 2,411 行 · 第 1/5 页
S
2,411 行
.equiv nb430nf_jindex, 16.equiv nb430nf_jjnr, 20.equiv nb430nf_shift, 24.equiv nb430nf_shiftvec, 28.equiv nb430nf_fshift, 32.equiv nb430nf_gid, 36.equiv nb430nf_pos, 40.equiv nb430nf_faction, 44.equiv nb430nf_charge, 48.equiv nb430nf_p_facel, 52.equiv nb430nf_argkrf, 56.equiv nb430nf_argcrf, 60.equiv nb430nf_Vc, 64.equiv nb430nf_type, 68.equiv nb430nf_p_ntype, 72.equiv nb430nf_vdwparam, 76.equiv nb430nf_Vvdw, 80.equiv nb430nf_p_tabscale, 84.equiv nb430nf_VFtab, 88.equiv nb430nf_invsqrta, 92.equiv nb430nf_dvda, 96.equiv nb430nf_p_gbtabscale, 100.equiv nb430nf_GBtab, 104.equiv nb430nf_p_nthreads, 108.equiv nb430nf_count, 112.equiv nb430nf_mtx, 116.equiv nb430nf_outeriter, 120.equiv nb430nf_inneriter, 124.equiv nb430nf_work, 128 ;# stack offsets for local variables ;# bottom of stack is cache-aligned for sse use .equiv nb430nf_ix, 0.equiv nb430nf_iy, 16.equiv nb430nf_iz, 32.equiv nb430nf_iq, 48.equiv nb430nf_gbtsc, 64.equiv nb430nf_tsc, 80.equiv nb430nf_qq, 96.equiv nb430nf_c6, 112.equiv nb430nf_c12, 128.equiv nb430nf_vctot, 144.equiv nb430nf_Vvdwtot, 160.equiv nb430nf_half, 176.equiv nb430nf_three, 192.equiv nb430nf_isai, 208.equiv nb430nf_isaprod, 224.equiv nb430nf_gbscale, 240.equiv nb430nf_r, 256.equiv nb430nf_is3, 272.equiv nb430nf_ii3, 276.equiv nb430nf_ntia, 280.equiv nb430nf_innerjjnr, 284.equiv nb430nf_innerk, 288.equiv nb430nf_n, 292.equiv nb430nf_nn1, 296.equiv nb430nf_nri, 300.equiv nb430nf_facel, 304.equiv nb430nf_ntype, 308.equiv nb430nf_nouter, 312.equiv nb430nf_ninner, 316.equiv nb430nf_salign, 320 push ebp mov ebp,esp push eax push ebx push ecx push edx push esi push edi sub esp, 324 ;# local stack space mov eax, esp and eax, 0xf sub esp, eax mov [esp + nb430nf_salign], eax emms ;# Move args passed by reference to stack mov ecx, [ebp + nb430nf_p_nri] mov esi, [ebp + nb430nf_p_facel] mov edi, [ebp + nb430nf_p_ntype] mov ecx, [ecx] mov esi, [esi] mov edi, [edi] mov [esp + nb430nf_nri], ecx mov [esp + nb430nf_facel], esi mov [esp + nb430nf_ntype], edi ;# zero iteration counters mov eax, 0 mov [esp + nb430nf_nouter], eax mov [esp + nb430nf_ninner], eax mov eax, [ebp + nb430nf_p_gbtabscale] movss xmm3, [eax] mov eax, [ebp + nb430nf_p_tabscale] movss xmm4, [eax] shufps xmm3, xmm3, 0 shufps xmm4, xmm4, 0 movaps [esp + nb430nf_gbtsc], xmm3 movaps [esp + nb430nf_tsc], xmm4 ;# create constant floating-point factors on stack mov eax, 0x3f000000 ;# constant 0.5 in IEEE (hex) mov [esp + nb430nf_half], eax movss xmm1, [esp + nb430nf_half] shufps xmm1, xmm1, 0 ;# splat to all elements movaps xmm2, xmm1 addps xmm2, xmm2 ;# constant 1.0 movaps xmm3, xmm2 addps xmm2, xmm2 ;# constant 2.0 addps xmm3, xmm2 ;# constant 3.0 movaps [esp + nb430nf_half], xmm1 movaps [esp + nb430nf_three], xmm3.nb430nf_threadloop: mov esi, [ebp + nb430nf_count] ;# pointer to sync counter mov eax, [esi].nb430nf_spinlock: mov ebx, eax ;# ebx=*count=nn0 add ebx, 1 ;# ebx=nn1=nn0+10 lock cmpxchg [esi], ebx ;# write nn1 to *counter, ;# if it hasnt changed. ;# or reread *counter to eax. pause ;# -> better p4 performance jnz .nb430nf_spinlock ;# if(nn1>nri) nn1=nri mov ecx, [esp + nb430nf_nri] mov edx, ecx sub ecx, ebx cmovle ebx, edx ;# if(nn1>nri) nn1=nri ;# Cleared the spinlock if we got here. ;# eax contains nn0, ebx contains nn1. mov [esp + nb430nf_n], eax mov [esp + nb430nf_nn1], ebx sub ebx, eax ;# calc number of outer lists mov esi, eax ;# copy n to esi jg .nb430nf_outerstart jmp .nb430nf_end.nb430nf_outerstart: ;# ebx contains number of outer iterations add ebx, [esp + nb430nf_nouter] mov [esp + nb430nf_nouter], ebx.nb430nf_outer: mov eax, [ebp + nb430nf_shift] ;# eax = pointer into shift[] mov ebx, [eax + esi*4] ;# ebx=shift[n] lea ebx, [ebx + ebx*2] ;# ebx=3*is mov [esp + nb430nf_is3],ebx ;# store is3 mov eax, [ebp + nb430nf_shiftvec] ;# eax = base of shiftvec[] movss xmm0, [eax + ebx*4] movss xmm1, [eax + ebx*4 + 4] movss xmm2, [eax + ebx*4 + 8] mov ecx, [ebp + nb430nf_iinr] ;# ecx = pointer into iinr[] mov ebx, [ecx + esi*4] ;# ebx =ii mov edx, [ebp + nb430nf_charge] movss xmm3, [edx + ebx*4] mulss xmm3, [esp + nb430nf_facel] shufps xmm3, xmm3, 0 mov edx, [ebp + nb430nf_invsqrta] ;# load invsqrta[ii] movss xmm4, [edx + ebx*4] shufps xmm4, xmm4, 0 mov edx, [ebp + nb430nf_type] mov edx, [edx + ebx*4] imul edx, [esp + nb430nf_ntype] shl edx, 1 mov [esp + nb430nf_ntia], edx lea ebx, [ebx + ebx*2] ;# ebx = 3*ii=ii3 mov eax, [ebp + nb430nf_pos] ;# eax = base of pos[] addss xmm0, [eax + ebx*4] addss xmm1, [eax + ebx*4 + 4] addss xmm2, [eax + ebx*4 + 8] movaps [esp + nb430nf_iq], xmm3 movaps [esp + nb430nf_isai], xmm4 shufps xmm0, xmm0, 0 shufps xmm1, xmm1, 0 shufps xmm2, xmm2, 0 movaps [esp + nb430nf_ix], xmm0 movaps [esp + nb430nf_iy], xmm1 movaps [esp + nb430nf_iz], xmm2 mov [esp + nb430nf_ii3], ebx ;# clear vctot xorps xmm4, xmm4 movaps [esp + nb430nf_vctot], xmm4 movaps [esp + nb430nf_Vvdwtot], xmm4 mov eax, [ebp + nb430nf_jindex] mov ecx, [eax + esi*4] ;# jindex[n] mov edx, [eax + esi*4 + 4] ;# jindex[n+1] sub edx, ecx ;# number of innerloop atoms mov esi, [ebp + nb430nf_pos] mov edi, [ebp + nb430nf_faction] mov eax, [ebp + nb430nf_jjnr] shl ecx, 2 add eax, ecx mov [esp + nb430nf_innerjjnr], eax ;# pointer to jjnr[nj0] mov ecx, edx sub edx, 4 add ecx, [esp + nb430nf_ninner] mov [esp + nb430nf_ninner], ecx add edx, 0 mov [esp + nb430nf_innerk], edx ;# number of innerloop atoms jge .nb430nf_unroll_loop jmp .nb430nf_finish_inner.nb430nf_unroll_loop: ;# quad-unroll innerloop here mov edx, [esp + nb430nf_innerjjnr] ;# pointer to jjnr[k] mov eax, [edx] mov ebx, [edx + 4] mov ecx, [edx + 8] mov edx, [edx + 12] ;# eax-edx=jnr1-4 add dword ptr [esp + nb430nf_innerjjnr], 16 ;# advance pointer (unrolled 4) ;# load isa2 mov esi, [ebp + nb430nf_invsqrta] movss xmm3, [esi + eax*4] movss xmm4, [esi + ecx*4] movss xmm6, [esi + ebx*4] movss xmm7, [esi + edx*4] movaps xmm2, [esp + nb430nf_isai] shufps xmm3, xmm6, 0 shufps xmm4, xmm7, 0 shufps xmm3, xmm4, 136 ;# constant 10001000 ;# all charges in xmm3 mulps xmm2, xmm3 movaps [esp + nb430nf_isaprod], xmm2 movaps xmm1, xmm2 mulps xmm1, [esp + nb430nf_gbtsc] movaps [esp + nb430nf_gbscale], xmm1 mov esi, [ebp + nb430nf_charge] ;# base of charge[] movss xmm3, [esi + eax*4] movss xmm4, [esi + ecx*4] movss xmm6, [esi + ebx*4] movss xmm7, [esi + edx*4] mulps xmm2, [esp + nb430nf_iq] shufps xmm3, xmm6, 0 shufps xmm4, xmm7, 0 shufps xmm3, xmm4, 136 ;# constant 10001000 ;# all charges in xmm3 mulps xmm3, xmm2 movaps [esp + nb430nf_qq], xmm3 movd mm0, eax ;# use mmx registers as temp storage movd mm1, ebx movd mm2, ecx movd mm3, edx mov esi, [ebp + nb430nf_type] mov eax, [esi + eax*4] mov ebx, [esi + ebx*4] mov ecx, [esi + ecx*4] mov edx, [esi + edx*4] mov esi, [ebp + nb430nf_vdwparam] shl eax, 1 shl ebx, 1 shl ecx, 1 shl edx, 1 mov edi, [esp + nb430nf_ntia] add eax, edi add ebx, edi add ecx, edi add edx, edi movlps xmm6, [esi + eax*4] movlps xmm7, [esi + ecx*4] movhps xmm6, [esi + ebx*4] movhps xmm7, [esi + edx*4] movaps xmm4, xmm6 shufps xmm4, xmm7, 136 ;# constant 10001000 shufps xmm6, xmm7, 221 ;# constant 11011101 movd eax, mm0 movd ebx, mm1 movd ecx, mm2 movd edx, mm3 movaps [esp + nb430nf_c6], xmm4 movaps [esp + nb430nf_c12], xmm6 mov esi, [ebp + nb430nf_pos] ;# base of pos[] lea eax, [eax + eax*2] ;# replace jnr with j3 lea ebx, [ebx + ebx*2] lea ecx, [ecx + ecx*2] ;# replace jnr with j3 lea edx, [edx + edx*2] ;# move four coordinates to xmm0-xmm2 movlps xmm4, [esi + eax*4] movlps xmm5, [esi + ecx*4] movss xmm2, [esi + eax*4 + 8] movss xmm6, [esi + ecx*4 + 8] movhps xmm4, [esi + ebx*4] movhps xmm5, [esi + edx*4] movss xmm0, [esi + ebx*4 + 8] movss xmm1, [esi + edx*4 + 8] shufps xmm2, xmm0, 0 shufps xmm6, xmm1, 0 movaps xmm0, xmm4 movaps xmm1, xmm4 shufps xmm2, xmm6, 136 ;# constant 10001000 shufps xmm0, xmm5, 136 ;# constant 10001000 shufps xmm1, xmm5, 221 ;# constant 11011101 ;# move ix-iz to xmm4-xmm6 movaps xmm4, [esp + nb430nf_ix] movaps xmm5, [esp + nb430nf_iy] movaps xmm6, [esp + nb430nf_iz] ;# calc dr subps xmm4, xmm0 subps xmm5, xmm1 subps xmm6, xmm2 ;# square it mulps xmm4,xmm4 mulps xmm5,xmm5 mulps xmm6,xmm6 addps xmm4, xmm5 addps xmm4, xmm6 ;# rsq in xmm4 rsqrtps xmm5, xmm4 ;# lookup seed in xmm5 movaps xmm2, xmm5 mulps xmm5, xmm5 movaps xmm1, [esp + nb430nf_three] mulps xmm5, xmm4 ;# rsq*lu*lu movaps xmm0, [esp + nb430nf_half] subps xmm1, xmm5 ;# constant 30-rsq*lu*lu mulps xmm1, xmm2 mulps xmm0, xmm1 ;# xmm0=rinv mulps xmm4, xmm0 ;# xmm4=r movaps [esp + nb430nf_r], xmm4 mulps xmm4, [esp + nb430nf_gbscale] movhlps xmm5, xmm4 cvttps2pi mm6, xmm4 cvttps2pi mm7, xmm5 ;# mm6/mm7 contain lu indices cvtpi2ps xmm6, mm6 cvtpi2ps xmm5, mm7 movlhps xmm6, xmm5 subps xmm4, xmm6 movaps xmm1, xmm4 ;# xmm1=eps movaps xmm2, xmm1 mulps xmm2, xmm2 ;# xmm2=eps2 pslld mm6, 2 pslld mm7, 2 movd mm0, eax movd mm1, ebx movd mm2, ecx movd mm3, edx mov esi, [ebp + nb430nf_GBtab] movd eax, mm6 psrlq mm6, 32 movd ecx, mm7 psrlq mm7, 32 movd ebx, mm6 movd edx, mm7 ;# load coulomb table movaps xmm4, [esi + eax*4] movaps xmm5, [esi + ebx*4] movaps xmm6, [esi + ecx*4] movaps xmm7, [esi + edx*4] ;# transpose, using xmm3 for scratch movaps xmm3, xmm6 shufps xmm3, xmm7, 0xEE shufps xmm6, xmm7, 0x44 movaps xmm7, xmm4 shufps xmm7, xmm5, 0xEE shufps xmm4, xmm5, 0x44 movaps xmm5, xmm4 shufps xmm5, xmm6, 0xDD shufps xmm4, xmm6, 0x88 movaps xmm6, xmm7 shufps xmm6, xmm3, 0x88 shufps xmm7, xmm3, 0xDD ;# coulomb table ready, in xmm4-xmm7 mulps xmm6, xmm1 ;# xmm6=Geps mulps xmm7, xmm2 ;# xmm7=Heps2 addps xmm5, xmm6 addps xmm5, xmm7 ;# xmm5=Fp movaps xmm3, [esp + nb430nf_qq] mulps xmm5, xmm1 ;# xmm5=eps*Fp addps xmm5, xmm4 ;# xmm5=VV mulps xmm5, xmm3 ;# vcoul=qq*VV addps xmm5, [esp + nb430nf_vctot] movaps [esp + nb430nf_vctot], xmm5 movaps xmm4, [esp + nb430nf_r] mulps xmm4, [esp + nb430nf_tsc] movhlps xmm5, xmm4 cvttps2pi mm6, xmm4 cvttps2pi mm7, xmm5 ;# mm6/mm7 contain lu indices cvtpi2ps xmm6, mm6 cvtpi2ps xmm5, mm7 movlhps xmm6, xmm5 subps xmm4, xmm6 movaps xmm1, xmm4 ;# xmm1=eps movaps xmm2, xmm1 mulps xmm2, xmm2 ;# xmm2=eps2 pslld mm6, 3 pslld mm7, 3 mov esi, [ebp + nb430nf_VFtab] movd eax, mm6 psrlq mm6, 32 movd ecx, mm7 psrlq mm7, 32 movd ebx, mm6 movd edx, mm7 ;# dispersion movaps xmm4, [esi + eax*4] movaps xmm5, [esi + ebx*4] movaps xmm6, [esi + ecx*4] movaps xmm7, [esi + edx*4] ;# transpose, using xmm3 for scratch movaps xmm3, xmm6 shufps xmm3, xmm7, 0xEE shufps xmm6, xmm7, 0x44 movaps xmm7, xmm4 shufps xmm7, xmm5, 0xEE shufps xmm4, xmm5, 0x44 movaps xmm5, xmm4 shufps xmm5, xmm6, 0xDD shufps xmm4, xmm6, 0x88 movaps xmm6, xmm7 shufps xmm6, xmm3, 0x88 shufps xmm7, xmm3, 0xDD ;# dispersion table ready, in xmm4-xmm7 mulps xmm6, xmm1 ;# xmm6=Geps mulps xmm7, xmm2 ;# xmm7=Heps2 addps xmm5, xmm6 addps xmm5, xmm7 ;# xmm5=Fp mulps xmm5, xmm1 ;# xmm5=eps*Fp addps xmm5, xmm4 ;# xmm5=VV mulps xmm5, [esp + nb430nf_c6] ;# Vvdw6 addps xmm5, [esp + nb430nf_Vvdwtot] movaps [esp + nb430nf_Vvdwtot], xmm5 ;# repulsion movaps xmm4, [esi + eax*4 + 16] movaps xmm5, [esi + ebx*4 + 16] movaps xmm6, [esi + ecx*4 + 16] movaps xmm7, [esi + edx*4 + 16] ;# transpose, using xmm3 for scratch movaps xmm3, xmm6
⌨️ 快捷键说明
复制代码Ctrl + C
搜索代码Ctrl + F
全屏模式F11
增大字号Ctrl + =
减小字号Ctrl + -
显示快捷键?