nb_kernel132_ia32_sse.intel_syntax.s
来自「最著名最快的分子模拟软件」· S 代码 · 共 2,176 行 · 第 1/5 页
S
2,176 行
leave ret.globl nb_kernel132nf_ia32_sse.globl _nb_kernel132nf_ia32_ssenb_kernel132nf_ia32_sse: _nb_kernel132nf_ia32_sse: .equiv nb132nf_p_nri, 8.equiv nb132nf_iinr, 12.equiv nb132nf_jindex, 16.equiv nb132nf_jjnr, 20.equiv nb132nf_shift, 24.equiv nb132nf_shiftvec, 28.equiv nb132nf_fshift, 32.equiv nb132nf_gid, 36.equiv nb132nf_pos, 40.equiv nb132nf_faction, 44.equiv nb132nf_charge, 48.equiv nb132nf_p_facel, 52.equiv nb132nf_argkrf, 56.equiv nb132nf_argcrf, 60.equiv nb132nf_Vc, 64.equiv nb132nf_type, 68.equiv nb132nf_p_ntype, 72.equiv nb132nf_vdwparam, 76.equiv nb132nf_Vvdw, 80.equiv nb132nf_p_tabscale, 84.equiv nb132nf_VFtab, 88.equiv nb132nf_invsqrta, 92.equiv nb132nf_dvda, 96.equiv nb132nf_p_gbtabscale, 100.equiv nb132nf_GBtab, 104.equiv nb132nf_p_nthreads, 108.equiv nb132nf_count, 112.equiv nb132nf_mtx, 116.equiv nb132nf_outeriter, 120.equiv nb132nf_inneriter, 124.equiv nb132nf_work, 128 ;# stack offsets for local variables ;# bottom of stack is cache-aligned for sse use .equiv nb132nf_ixO, 0.equiv nb132nf_iyO, 16.equiv nb132nf_izO, 32.equiv nb132nf_ixH1, 48.equiv nb132nf_iyH1, 64.equiv nb132nf_izH1, 80.equiv nb132nf_ixH2, 96.equiv nb132nf_iyH2, 112.equiv nb132nf_izH2, 128.equiv nb132nf_jxO, 144.equiv nb132nf_jyO, 160.equiv nb132nf_jzO, 176.equiv nb132nf_jxH1, 192.equiv nb132nf_jyH1, 208.equiv nb132nf_jzH1, 224.equiv nb132nf_jxH2, 240.equiv nb132nf_jyH2, 256.equiv nb132nf_jzH2, 272.equiv nb132nf_qqOO, 288.equiv nb132nf_qqOH, 304.equiv nb132nf_qqHH, 320.equiv nb132nf_c6, 336.equiv nb132nf_c12, 352.equiv nb132nf_tsc, 368.equiv nb132nf_vctot, 384.equiv nb132nf_Vvdwtot, 400.equiv nb132nf_half, 416.equiv nb132nf_three, 432.equiv nb132nf_rsqOO, 448.equiv nb132nf_rsqOH1, 464.equiv nb132nf_rsqOH2, 480.equiv nb132nf_rsqH1O, 496.equiv nb132nf_rsqH1H1, 512.equiv nb132nf_rsqH1H2, 528.equiv nb132nf_rsqH2O, 544.equiv nb132nf_rsqH2H1, 560.equiv nb132nf_rsqH2H2, 576.equiv nb132nf_rinvOO, 592.equiv nb132nf_rinvOH1, 608.equiv nb132nf_rinvOH2, 624.equiv nb132nf_rinvH1O, 640.equiv nb132nf_rinvH1H1, 656.equiv nb132nf_rinvH1H2, 672.equiv nb132nf_rinvH2O, 688.equiv nb132nf_rinvH2H1, 704.equiv nb132nf_rinvH2H2, 720.equiv nb132nf_is3, 768.equiv nb132nf_ii3, 772.equiv nb132nf_innerjjnr, 776.equiv nb132nf_innerk, 780.equiv nb132nf_n, 784.equiv nb132nf_nn1, 788.equiv nb132nf_nri, 792.equiv nb132nf_nouter, 796.equiv nb132nf_ninner, 800.equiv nb132nf_salign, 804 push ebp mov ebp,esp push eax push ebx push ecx push edx push esi push edi sub esp, 808 ;# local stack space mov eax, esp and eax, 0xf sub esp, eax mov [esp + nb132nf_salign], eax emms ;# Move args passed by reference to stack mov ecx, [ebp + nb132nf_p_nri] mov ecx, [ecx] mov [esp + nb132nf_nri], ecx ;# zero iteration counters mov eax, 0 mov [esp + nb132nf_nouter], eax mov [esp + nb132nf_ninner], eax mov eax, [ebp + nb132nf_p_tabscale] movss xmm3, [eax] shufps xmm3, xmm3, 0 movaps [esp + nb132nf_tsc], xmm3 ;# assume we have at least one i particle - start directly mov ecx, [ebp + nb132nf_iinr] ;# ecx = pointer into iinr[] mov ebx, [ecx] ;# ebx =ii mov edx, [ebp + nb132nf_charge] movss xmm3, [edx + ebx*4] movss xmm4, xmm3 movss xmm5, [edx + ebx*4 + 4] mov esi, [ebp + nb132nf_p_facel] movss xmm6, [esi] mulss xmm3, xmm3 mulss xmm4, xmm5 mulss xmm5, xmm5 mulss xmm3, xmm6 mulss xmm4, xmm6 mulss xmm5, xmm6 shufps xmm3, xmm3, 0 shufps xmm4, xmm4, 0 shufps xmm5, xmm5, 0 movaps [esp + nb132nf_qqOO], xmm3 movaps [esp + nb132nf_qqOH], xmm4 movaps [esp + nb132nf_qqHH], xmm5 xorps xmm0, xmm0 mov edx, [ebp + nb132nf_type] mov ecx, [edx + ebx*4] shl ecx, 1 mov edx, ecx mov edi, [ebp + nb132nf_p_ntype] imul ecx, [edi] ;# ecx = ntia = 2*ntype*type[ii0] add edx, ecx mov eax, [ebp + nb132nf_vdwparam] movlps xmm0, [eax + edx*4] movaps xmm1, xmm0 shufps xmm0, xmm0, 0 shufps xmm1, xmm1, 85 ;# constant 01010101 movaps [esp + nb132nf_c6], xmm0 movaps [esp + nb132nf_c12], xmm1 ;# create constant floating-point factors on stack mov eax, 0x3f000000 ;# constant 0.5 in IEEE (hex) mov [esp + nb132nf_half], eax movss xmm1, [esp + nb132nf_half] shufps xmm1, xmm1, 0 ;# splat to all elements movaps xmm2, xmm1 addps xmm2, xmm2 ;# constant 1.0 movaps xmm3, xmm2 addps xmm2, xmm2 ;# constant 2.0 addps xmm3, xmm2 ;# constant 3.0 movaps [esp + nb132nf_half], xmm1 movaps [esp + nb132nf_three], xmm3.nb132nf_threadloop: mov esi, [ebp + nb132nf_count] ;# pointer to sync counter mov eax, [esi].nb132nf_spinlock: mov ebx, eax ;# ebx=*count=nn0 add ebx, 1 ;# ebx=nn1=nn0+10 lock cmpxchg [esi], ebx ;# write nn1 to *counter, ;# if it hasnt changed. ;# or reread *counter to eax. pause ;# -> better p4 performance jnz .nb132nf_spinlock ;# if(nn1>nri) nn1=nri mov ecx, [esp + nb132nf_nri] mov edx, ecx sub ecx, ebx cmovle ebx, edx ;# if(nn1>nri) nn1=nri ;# Cleared the spinlock if we got here. ;# eax contains nn0, ebx contains nn1. mov [esp + nb132nf_n], eax mov [esp + nb132nf_nn1], ebx sub ebx, eax ;# calc number of outer lists mov esi, eax ;# copy n to esi jg .nb132nf_outerstart jmp .nb132nf_end.nb132nf_outerstart: ;# ebx contains number of outer iterations add ebx, [esp + nb132nf_nouter] mov [esp + nb132nf_nouter], ebx.nb132nf_outer: mov eax, [ebp + nb132nf_shift] ;# eax = pointer into shift[] mov ebx, [eax + esi*4] ;# ebx=shift[n] lea ebx, [ebx + ebx*2] ;# ebx=3*is mov [esp + nb132nf_is3],ebx ;# store is3 mov eax, [ebp + nb132nf_shiftvec] ;# eax = base of shiftvec[] movss xmm0, [eax + ebx*4] movss xmm1, [eax + ebx*4 + 4] movss xmm2, [eax + ebx*4 + 8] mov ecx, [ebp + nb132nf_iinr] ;# ecx = pointer into iinr[] mov ebx, [ecx + esi*4] ;# ebx =ii lea ebx, [ebx + ebx*2] ;# ebx = 3*ii=ii3 mov eax, [ebp + nb132nf_pos] ;# eax = base of pos[] mov [esp + nb132nf_ii3], ebx movaps xmm3, xmm0 movaps xmm4, xmm1 movaps xmm5, xmm2 addss xmm3, [eax + ebx*4] addss xmm4, [eax + ebx*4 + 4] addss xmm5, [eax + ebx*4 + 8] shufps xmm3, xmm3, 0 shufps xmm4, xmm4, 0 shufps xmm5, xmm5, 0 movaps [esp + nb132nf_ixO], xmm3 movaps [esp + nb132nf_iyO], xmm4 movaps [esp + nb132nf_izO], xmm5 movss xmm3, xmm0 movss xmm4, xmm1 movss xmm5, xmm2 addss xmm0, [eax + ebx*4 + 12] addss xmm1, [eax + ebx*4 + 16] addss xmm2, [eax + ebx*4 + 20] addss xmm3, [eax + ebx*4 + 24] addss xmm4, [eax + ebx*4 + 28] addss xmm5, [eax + ebx*4 + 32] shufps xmm0, xmm0, 0 shufps xmm1, xmm1, 0 shufps xmm2, xmm2, 0 shufps xmm3, xmm3, 0 shufps xmm4, xmm4, 0 shufps xmm5, xmm5, 0 movaps [esp + nb132nf_ixH1], xmm0 movaps [esp + nb132nf_iyH1], xmm1 movaps [esp + nb132nf_izH1], xmm2 movaps [esp + nb132nf_ixH2], xmm3 movaps [esp + nb132nf_iyH2], xmm4 movaps [esp + nb132nf_izH2], xmm5 ;# clear vctot xorps xmm4, xmm4 movaps [esp + nb132nf_vctot], xmm4 movaps [esp + nb132nf_Vvdwtot], xmm4 mov eax, [ebp + nb132nf_jindex] mov ecx, [eax + esi*4] ;# jindex[n] mov edx, [eax + esi*4 + 4] ;# jindex[n+1] sub edx, ecx ;# number of innerloop atoms mov esi, [ebp + nb132nf_pos] mov eax, [ebp + nb132nf_jjnr] shl ecx, 2 add eax, ecx mov [esp + nb132nf_innerjjnr], eax ;# pointer to jjnr[nj0] mov ecx, edx sub edx, 4 add ecx, [esp + nb132nf_ninner] mov [esp + nb132nf_ninner], ecx add edx, 0 mov [esp + nb132nf_innerk], edx ;# number of innerloop atoms jge .nb132nf_unroll_loop jmp .nb132nf_single_check.nb132nf_unroll_loop: ;# quad-unroll innerloop here mov edx, [esp + nb132nf_innerjjnr] ;# pointer to jjnr[k] mov eax, [edx] mov ebx, [edx + 4] mov ecx, [edx + 8] mov edx, [edx + 12] ;# eax-edx=jnr1-4 add dword ptr [esp + nb132nf_innerjjnr], 16 ;# advance pointer (unrolled 4) mov esi, [ebp + nb132nf_pos] ;# base of pos[] lea eax, [eax + eax*2] ;# replace jnr with j3 lea ebx, [ebx + ebx*2] lea ecx, [ecx + ecx*2] ;# replace jnr with j3 lea edx, [edx + edx*2] ;# move j coordinates to local temp variables movlps xmm2, [esi + eax*4] movlps xmm3, [esi + eax*4 + 12] movlps xmm4, [esi + eax*4 + 24] movlps xmm5, [esi + ebx*4] movlps xmm6, [esi + ebx*4 + 12] movlps xmm7, [esi + ebx*4 + 24] movhps xmm2, [esi + ecx*4] movhps xmm3, [esi + ecx*4 + 12] movhps xmm4, [esi + ecx*4 + 24] movhps xmm5, [esi + edx*4] movhps xmm6, [esi + edx*4 + 12] movhps xmm7, [esi + edx*4 + 24] ;# current state: ;# xmm2= jxOa jyOa jxOc jyOc ;# xmm3= jxH1a jyH1a jxH1c jyH1c ;# xmm4= jxH2a jyH2a jxH2c jyH2c ;# xmm5= jxOb jyOb jxOd jyOd ;# xmm6= jxH1b jyH1b jxH1d jyH1d ;# xmm7= jxH2b jyH2b jxH2d jyH2d movaps xmm0, xmm2 movaps xmm1, xmm3 unpcklps xmm0, xmm5 ;# xmm0= jxOa jxOb jyOa jyOb unpcklps xmm1, xmm6 ;# xmm1= jxH1a jxH1b jyH1a jyH1b unpckhps xmm2, xmm5 ;# xmm2= jxOc jxOd jyOc jyOd unpckhps xmm3, xmm6 ;# xmm3= jxH1c jxH1d jyH1c jyH1d movaps xmm5, xmm4 movaps xmm6, xmm0 unpcklps xmm4, xmm7 ;# xmm4= jxH2a jxH2b jyH2a jyH2b unpckhps xmm5, xmm7 ;# xmm5= jxH2c jxH2d jyH2c jyH2d movaps xmm7, xmm1 movlhps xmm0, xmm2 ;# xmm0= jxOa jxOb jxOc jxOd movaps [esp + nb132nf_jxO], xmm0 movhlps xmm2, xmm6 ;# xmm2= jyOa jyOb jyOc jyOd movaps [esp + nb132nf_jyO], xmm2 movlhps xmm1, xmm3 movaps [esp + nb132nf_jxH1], xmm1 movhlps xmm3, xmm7 movaps xmm6, xmm4 movaps [esp + nb132nf_jyH1], xmm3 movlhps xmm4, xmm5 movaps [esp + nb132nf_jxH2], xmm4 movhlps xmm5, xmm6 movaps [esp + nb132nf_jyH2], xmm5 movss xmm0, [esi + eax*4 + 8] movss xmm1, [esi + eax*4 + 20] movss xmm2, [esi + eax*4 + 32] movss xmm3, [esi + ecx*4 + 8] movss xmm4, [esi + ecx*4 + 20] movss xmm5, [esi + ecx*4 + 32] movhps xmm0, [esi + ebx*4 + 4] movhps xmm1, [esi + ebx*4 + 16] movhps xmm2, [esi + ebx*4 + 28] movhps xmm3, [esi + edx*4 + 4] movhps xmm4, [esi + edx*4 + 16] movhps xmm5, [esi + edx*4 + 28] shufps xmm0, xmm3, 204 ;# constant 11001100 shufps xmm1, xmm4, 204 ;# constant 11001100 shufps xmm2, xmm5, 204 ;# constant 11001100 movaps [esp + nb132nf_jzO], xmm0 movaps [esp + nb132nf_jzH1], xmm1 movaps [esp + nb132nf_jzH2], xmm2 movaps xmm0, [esp + nb132nf_ixO] movaps xmm1, [esp + nb132nf_iyO] movaps xmm2, [esp + nb132nf_izO] movaps xmm3, [esp + nb132nf_ixO] movaps xmm4, [esp + nb132nf_iyO] movaps xmm5, [esp + nb132nf_izO] subps xmm0, [esp + nb132nf_jxO] subps xmm1, [esp + nb132nf_jyO] subps xmm2, [esp + nb132nf_jzO] subps xmm3, [esp + nb132nf_jxH1] subps xmm4, [esp + nb132nf_jyH1] subps xmm5, [esp + nb132nf_jzH1] mulps xmm0, xmm0 mulps xmm1, xmm1 mulps xmm2, xmm2 mulps xmm3, xmm3 mulps xmm4, xmm4 mulps xmm5, xmm5 addps xmm0, xmm1 addps xmm0, xmm2 addps xmm3, xmm4 addps xmm3, xmm5 movaps [esp + nb132nf_rsqOO], xmm0 movaps [esp + nb132nf_rsqOH1], xmm3 movaps xmm0, [esp + nb132nf_ixO] movaps xmm1, [esp + nb132nf_iyO] movaps xmm2, [esp + nb132nf_izO] movaps xmm3, [esp + nb132nf_ixH1] movaps xmm4, [esp + nb132nf_iyH1] movaps xmm5, [esp + nb132nf_izH1] subps xmm0, [esp + nb132nf_jxH2] subps xmm1, [esp + nb132nf_jyH2] subps xmm2, [esp + nb132nf_jzH2] subps xmm3, [esp + nb132nf_jxO] subps xmm4, [esp + nb132nf_jyO] subps xmm5, [esp + nb132nf_jzO] mulps xmm0, xmm0 mulps xmm1, xmm1 mulps xmm2, xmm2 mulps xmm3, xmm3 mulps xmm4, xmm4 mulps xmm5, xmm5 addps xmm0, xmm1 addps xmm0, xmm2 addps xmm3, xmm4 addps xmm3, xmm5 movaps [esp + nb132nf_rsqOH2], xmm0 movaps [esp + nb132nf_rsqH1O], xmm3
⌨️ 快捷键说明
复制代码Ctrl + C
搜索代码Ctrl + F
全屏模式F11
增大字号Ctrl + =
减小字号Ctrl + -
显示快捷键?