nb_kernel133_ia32_sse.intel_syntax.s
来自「最著名最快的分子模拟软件」· S 代码 · 共 2,185 行 · 第 1/4 页
S
2,185 行
movhlps xmm3, xmm0 movhlps xmm4, xmm1 movhlps xmm5, xmm2 addps xmm3, xmm0 addps xmm4, xmm1 addps xmm5, xmm2 movaps xmm0, xmm3 movaps xmm1, xmm4 movaps xmm2, xmm5 shufps xmm3, xmm3, 0x39 ;# shift right shufps xmm4, xmm4, 0x39 shufps xmm5, xmm5, 0x39 addss xmm0, xmm3 addss xmm1, xmm4 addss xmm2, xmm5 unpcklps xmm0, xmm1 ;# x,y sum in xmm0, z sum in xmm2 subps xmm6, xmm0 subss xmm7, xmm2 movlps [edi + eax*4], xmm6 movss [edi + eax*4 + 8], xmm7 dec dword ptr [esp + nb133_innerk] jz .nb133_updateouterdata jmp .nb133_odd_loop.nb133_updateouterdata: mov ecx, [esp + nb133_ii3] mov edi, [ebp + nb133_faction] mov esi, [ebp + nb133_fshift] mov edx, [esp + nb133_is3] ;# accumulate Oi forces in xmm0, xmm1, xmm2 movaps xmm0, [esp + nb133_fixO] movaps xmm1, [esp + nb133_fiyO] movaps xmm2, [esp + nb133_fizO] movhlps xmm3, xmm0 movhlps xmm4, xmm1 movhlps xmm5, xmm2 addps xmm0, xmm3 addps xmm1, xmm4 addps xmm2, xmm5 ;# sum is in 1/2 in xmm0-xmm2 movaps xmm3, xmm0 movaps xmm4, xmm1 movaps xmm5, xmm2 shufps xmm3, xmm3, 1 shufps xmm4, xmm4, 1 shufps xmm5, xmm5, 1 addss xmm0, xmm3 addss xmm1, xmm4 addss xmm2, xmm5 ;# xmm0-xmm2 has single force in pos0 ;# increment i force movss xmm3, [edi + ecx*4] movss xmm4, [edi + ecx*4 + 4] movss xmm5, [edi + ecx*4 + 8] addss xmm3, xmm0 addss xmm4, xmm1 addss xmm5, xmm2 movss [edi + ecx*4], xmm3 movss [edi + ecx*4 + 4], xmm4 movss [edi + ecx*4 + 8], xmm5 ;# accumulate force in xmm6/xmm7 for fshift movaps xmm6, xmm0 movss xmm7, xmm2 movlhps xmm6, xmm1 shufps xmm6, xmm6, 8 ;# constant 00001000 ;# accumulate H1i forces in xmm0, xmm1, xmm2 movaps xmm0, [esp + nb133_fixH1] movaps xmm1, [esp + nb133_fiyH1] movaps xmm2, [esp + nb133_fizH1] movhlps xmm3, xmm0 movhlps xmm4, xmm1 movhlps xmm5, xmm2 addps xmm0, xmm3 addps xmm1, xmm4 addps xmm2, xmm5 ;# sum is in 1/2 in xmm0-xmm2 movaps xmm3, xmm0 movaps xmm4, xmm1 movaps xmm5, xmm2 shufps xmm3, xmm3, 1 shufps xmm4, xmm4, 1 shufps xmm5, xmm5, 1 addss xmm0, xmm3 addss xmm1, xmm4 addss xmm2, xmm5 ;# xmm0-xmm2 has single force in pos0 ;# increment i force movss xmm3, [edi + ecx*4 + 12] movss xmm4, [edi + ecx*4 + 16] movss xmm5, [edi + ecx*4 + 20] addss xmm3, xmm0 addss xmm4, xmm1 addss xmm5, xmm2 movss [edi + ecx*4 + 12], xmm3 movss [edi + ecx*4 + 16], xmm4 movss [edi + ecx*4 + 20], xmm5 ;# accumulate force in xmm6/xmm7 for fshift addss xmm7, xmm2 movlhps xmm0, xmm1 shufps xmm0, xmm0, 8 ;# constant 00001000 addps xmm6, xmm0 ;# accumulate H2i forces in xmm0, xmm1, xmm2 movaps xmm0, [esp + nb133_fixH2] movaps xmm1, [esp + nb133_fiyH2] movaps xmm2, [esp + nb133_fizH2] movhlps xmm3, xmm0 movhlps xmm4, xmm1 movhlps xmm5, xmm2 addps xmm0, xmm3 addps xmm1, xmm4 addps xmm2, xmm5 ;# sum is in 1/2 in xmm0-xmm2 movaps xmm3, xmm0 movaps xmm4, xmm1 movaps xmm5, xmm2 shufps xmm3, xmm3, 1 shufps xmm4, xmm4, 1 shufps xmm5, xmm5, 1 addss xmm0, xmm3 addss xmm1, xmm4 addss xmm2, xmm5 ;# xmm0-xmm2 has single force in pos0 ;# increment i force movss xmm3, [edi + ecx*4 + 24] movss xmm4, [edi + ecx*4 + 28] movss xmm5, [edi + ecx*4 + 32] addss xmm3, xmm0 addss xmm4, xmm1 addss xmm5, xmm2 movss [edi + ecx*4 + 24], xmm3 movss [edi + ecx*4 + 28], xmm4 movss [edi + ecx*4 + 32], xmm5 ;# accumulate force in xmm6/xmm7 for fshift addss xmm7, xmm2 movlhps xmm0, xmm1 shufps xmm0, xmm0, 8 ;# constant 00001000 addps xmm6, xmm0 ;# accumulate Mi forces in xmm0, xmm1, xmm2 movaps xmm0, [esp + nb133_fixM] movaps xmm1, [esp + nb133_fiyM] movaps xmm2, [esp + nb133_fizM] movhlps xmm3, xmm0 movhlps xmm4, xmm1 movhlps xmm5, xmm2 addps xmm0, xmm3 addps xmm1, xmm4 addps xmm2, xmm5 ;# sum is in 1/2 in xmm0-xmm2 movaps xmm3, xmm0 movaps xmm4, xmm1 movaps xmm5, xmm2 shufps xmm3, xmm3, 1 shufps xmm4, xmm4, 1 shufps xmm5, xmm5, 1 addss xmm0, xmm3 addss xmm1, xmm4 addss xmm2, xmm5 ;# xmm0-xmm2 has single force in pos0 ;# increment i force movss xmm3, [edi + ecx*4 + 36] movss xmm4, [edi + ecx*4 + 40] movss xmm5, [edi + ecx*4 + 44] addss xmm3, xmm0 addss xmm4, xmm1 addss xmm5, xmm2 movss [edi + ecx*4 + 36], xmm3 movss [edi + ecx*4 + 40], xmm4 movss [edi + ecx*4 + 44], xmm5 ;# accumulate force in xmm6/xmm7 for fshift addss xmm7, xmm2 movlhps xmm0, xmm1 shufps xmm0, xmm0, 8 ;# constant 00001000 addps xmm6, xmm0 ;# increment fshift force movlps xmm3, [esi + edx*4] movss xmm4, [esi + edx*4 + 8] addps xmm3, xmm6 addss xmm4, xmm7 movlps [esi + edx*4], xmm3 movss [esi + edx*4 + 8], xmm4 ;# get n from stack mov esi, [esp + nb133_n] ;# get group index for i particle mov edx, [ebp + nb133_gid] ;# base of gid[] mov edx, [edx + esi*4] ;# ggid=gid[n] ;# accumulate total potential energy and update it movaps xmm7, [esp + nb133_vctot] ;# accumulate movhlps xmm6, xmm7 addps xmm7, xmm6 ;# pos 0-1 in xmm7 have the sum now movaps xmm6, xmm7 shufps xmm6, xmm6, 1 addss xmm7, xmm6 ;# add earlier value from mem mov eax, [ebp + nb133_Vc] addss xmm7, [eax + edx*4] ;# move back to mem movss [eax + edx*4], xmm7 ;# accumulate total lj energy and update it movaps xmm7, [esp + nb133_Vvdwtot] ;# accumulate movhlps xmm6, xmm7 addps xmm7, xmm6 ;# pos 0-1 in xmm7 have the sum now movaps xmm6, xmm7 shufps xmm6, xmm6, 1 addss xmm7, xmm6 ;# add earlier value from mem mov eax, [ebp + nb133_Vvdw] addss xmm7, [eax + edx*4] ;# move back to mem movss [eax + edx*4], xmm7 ;# finish if last mov ecx, [esp + nb133_nn1] ;# esi already loaded with n inc esi sub ecx, esi jecxz .nb133_outerend ;# not last, iterate outer loop once more! mov [esp + nb133_n], esi jmp .nb133_outer.nb133_outerend: ;# check if more outer neighborlists remain mov ecx, [esp + nb133_nri] ;# esi already loaded with n above sub ecx, esi jecxz .nb133_end ;# non-zero, do one more workunit jmp .nb133_threadloop.nb133_end: emms mov eax, [esp + nb133_nouter] mov ebx, [esp + nb133_ninner] mov ecx, [ebp + nb133_outeriter] mov edx, [ebp + nb133_inneriter] mov [ecx], eax mov [edx], ebx mov eax, [esp + nb133_salign] add esp, eax add esp, 1004 pop edi pop esi pop edx pop ecx pop ebx pop eax leave ret .globl nb_kernel133nf_ia32_sse.globl _nb_kernel133nf_ia32_ssenb_kernel133nf_ia32_sse: _nb_kernel133nf_ia32_sse: .equiv nb133nf_p_nri, 8.equiv nb133nf_iinr, 12.equiv nb133nf_jindex, 16.equiv nb133nf_jjnr, 20.equiv nb133nf_shift, 24.equiv nb133nf_shiftvec, 28.equiv nb133nf_fshift, 32.equiv nb133nf_gid, 36.equiv nb133nf_pos, 40.equiv nb133nf_faction, 44.equiv nb133nf_charge, 48.equiv nb133nf_p_facel, 52.equiv nb133nf_argkrf, 56.equiv nb133nf_argcrf, 60.equiv nb133nf_Vc, 64.equiv nb133nf_type, 68.equiv nb133nf_p_ntype, 72.equiv nb133nf_vdwparam, 76.equiv nb133nf_Vvdw, 80.equiv nb133nf_p_tabscale, 84.equiv nb133nf_VFtab, 88.equiv nb133nf_invsqrta, 92.equiv nb133nf_dvda, 96.equiv nb133nf_p_gbtabscale, 100.equiv nb133nf_GBtab, 104.equiv nb133nf_p_nthreads, 108.equiv nb133nf_count, 112.equiv nb133nf_mtx, 116.equiv nb133nf_outeriter, 120.equiv nb133nf_inneriter, 124.equiv nb133nf_work, 128 ;# stack offsets for local variables ;# bottom of stack is cache-aligned for sse use .equiv nb133nf_ixO, 0.equiv nb133nf_iyO, 16.equiv nb133nf_izO, 32.equiv nb133nf_ixH1, 48.equiv nb133nf_iyH1, 64.equiv nb133nf_izH1, 80.equiv nb133nf_ixH2, 96.equiv nb133nf_iyH2, 112.equiv nb133nf_izH2, 128.equiv nb133nf_ixM, 144.equiv nb133nf_iyM, 160.equiv nb133nf_izM, 176.equiv nb133nf_iqM, 192.equiv nb133nf_iqH, 208.equiv nb133nf_qqH, 224.equiv nb133nf_rinvH1, 240.equiv nb133nf_rinvH2, 256.equiv nb133nf_rinvM, 272.equiv nb133nf_c6, 288.equiv nb133nf_c12, 304.equiv nb133nf_tsc, 320.equiv nb133nf_vctot, 416.equiv nb133nf_Vvdwtot, 432.equiv nb133nf_half, 448.equiv nb133nf_three, 464.equiv nb133nf_qqM, 480.equiv nb133nf_is3, 496.equiv nb133nf_ii3, 500.equiv nb133nf_ntia, 504.equiv nb133nf_innerjjnr, 508.equiv nb133nf_innerk, 512.equiv nb133nf_n, 516.equiv nb133nf_nn1, 520.equiv nb133nf_nri, 524.equiv nb133nf_nouter, 528.equiv nb133nf_ninner, 532.equiv nb133nf_salign, 536 push ebp mov ebp,esp push eax push ebx push ecx push edx push esi push edi sub esp, 540 ;# local stack space mov eax, esp and eax, 0xf sub esp, eax mov [esp + nb133nf_salign], eax emms ;# Move args passed by reference to stack mov ecx, [ebp + nb133nf_p_nri] mov ecx, [ecx] mov [esp + nb133nf_nri], ecx ;# zero iteration counters mov eax, 0 mov [esp + nb133nf_nouter], eax mov [esp + nb133nf_ninner], eax mov eax, [ebp + nb133nf_p_tabscale] movss xmm3, [eax] shufps xmm3, xmm3, 0 movaps [esp + nb133nf_tsc], xmm3 ;# create constant floating-point factors on stack mov eax, 0x3f000000 ;# constant 0.5 in IEEE (hex) mov [esp + nb133nf_half], eax movss xmm1, [esp + nb133nf_half] shufps xmm1, xmm1, 0 ;# splat to all elements movaps xmm2, xmm1 addps xmm2, xmm2 ;# constant 1.0 movaps xmm3, xmm2 addps xmm2, xmm2 ;# constant 2.0 addps xmm3, xmm2 ;# constant 3.0 movaps [esp + nb133nf_half], xmm1 movaps [esp + nb133nf_three], xmm3 ;# assume we have at least one i particle - start directly mov ecx, [ebp + nb133nf_iinr] ;# ecx = pointer into iinr[] mov ebx, [ecx] ;# ebx =ii mov edx, [ebp + nb133nf_charge] movss xmm4, [edx + ebx*4 + 4] movss xmm3, [edx + ebx*4 + 12] mov esi, [ebp + nb133nf_p_facel] movss xmm5, [esi] mulss xmm3, xmm5 mulss xmm4, xmm5 shufps xmm3, xmm3, 0 shufps xmm4, xmm4, 0 movaps [esp + nb133nf_iqM], xmm3 movaps [esp + nb133nf_iqH], xmm4 mov edx, [ebp + nb133nf_type] mov ecx, [edx + ebx*4] shl ecx, 1 mov edi, [ebp + nb133nf_p_ntype] imul ecx, [edi] ;# ecx = ntia = 2*ntype*type[ii0] mov [esp + nb133nf_ntia], ecx .nb133nf_threadloop: mov esi, [ebp + nb133nf_count] ;# pointer to sync counter mov eax, [esi].nb133nf_spinlock: mov ebx, eax ;# ebx=*count=nn0 add ebx, 1 ;# ebx=nn1=nn0+10 lock cmpxchg [esi], ebx ;# write nn1 to *counter, ;# if it hasnt changed. ;# or reread *counter to eax. pause ;# -> better p4 performance jnz .nb133nf_spinlock ;# if(nn1>nri) nn1=nri mov ecx, [esp + nb133nf_nri] mov edx, ecx sub ecx, ebx cmovle ebx, edx ;# if(nn1>nri) nn1=nri ;# Cleared the spinlock if we got here. ;# eax contains nn0, ebx contains nn1. mov [esp + nb133nf_n], eax mov [esp + nb133nf_nn1], ebx sub ebx, eax ;# calc number of outer lists mov esi, eax ;# copy n to esi jg .nb133nf_outerstart jmp .nb133nf_end .nb133nf_outerstart: ;# ebx contains number of outer iterations add ebx, [esp + nb133nf_nouter] mov [esp + nb133nf_nouter], ebx.nb133nf_outer: mov eax, [ebp + nb133nf_shift] ;# eax = pointer into shift[] mov ebx, [eax + esi*4] ;# ebx=shift[n] lea ebx, [ebx + ebx*2] ;# ebx=3*is mov [esp + nb133nf_is3],ebx ;# store is3 mov eax, [ebp + nb133nf_shiftvec] ;# eax = base of shiftvec[] movss xmm0, [eax + ebx*4] movss xmm1, [eax + ebx*4 + 4] movss xmm2, [eax + ebx*4 + 8] mov ecx, [ebp + nb133nf_iinr] ;# ecx = pointer into iinr[] mov ebx, [ecx + esi*4] ;# ebx =ii movaps xmm3, xmm0 movaps xmm4, xmm1 movaps xmm5, xmm2 movaps xmm6, xmm0 movaps xmm7, xmm1 lea ebx, [ebx + ebx*2] ;# ebx = 3*ii=ii3 mov eax, [ebp + nb133nf_pos] ;# eax = base of pos[] mov [esp + nb133nf_ii3], ebx addss xmm3, [eax + ebx*4] ;# ox addss xmm4, [eax + ebx*4 + 4] ;# oy addss xmm5, [eax + ebx*4 + 8] ;# oz addss xmm6, [eax + ebx*4 + 12] ;# h1x addss xmm7, [eax + ebx*4 + 16] ;# h1y shufps xmm3, xmm3, 0 shufps xmm4, xmm4, 0 shufps xmm5, xmm5, 0 shufps xmm6, xmm6, 0 shufps xmm7, xmm7, 0 movaps [esp + nb133nf_ixO], xmm3 movaps [esp + nb133nf_iyO], xmm4 movaps [esp + nb133nf_izO], xmm5 movaps [esp + nb133nf_ixH1], xmm6 movaps [esp + nb133nf_iyH1], xmm7 movss xmm6, xmm2 movss xmm3, xmm0 movss xmm4, xmm1 movss xmm5, xmm2 addss xmm6, [eax + ebx*4 + 20] ;# h1z addss xmm0, [eax + ebx*4 + 24] ;# h2x addss xmm1, [eax + ebx*4 + 28] ;# h2y addss xmm2, [eax + ebx*4 + 32] ;# h2z addss xmm3, [eax + ebx*4 + 36] ;# mx addss xmm4, [eax + ebx*4 + 40] ;# my addss xmm5, [eax + ebx*4 + 44] ;# mz shufps xmm6, xmm6, 0 shufps xmm0, xmm0, 0 shufps xmm1, xmm1, 0 shufps xmm2, xmm2, 0 shufps xmm3, xmm3, 0 shufps xmm4, xmm4, 0 shufps xmm5, xmm5, 0 movaps [esp + nb133nf_izH1], xmm6 movaps [esp + nb133nf_ixH2], xmm0 movaps [esp + nb133nf_iyH2], xmm1 movaps [esp + nb133nf_izH2], xmm2 movaps [esp + nb133nf_ixM], xmm3 movaps [esp + nb133nf_iyM], xmm4 movaps [esp + nb133nf_izM], xmm5 ;# clear vctot xorps xmm4, xmm4 movaps [esp + nb133nf_vctot], xmm4 movaps [esp + nb133nf_Vvdwtot], xmm4 mov eax, [ebp + nb133nf_jindex] mov ecx, [eax + esi*4] ;# jindex[n] mov edx, [eax + esi*4 + 4] ;# jindex[n+1] sub edx, ecx ;# number of innerloop atoms mov esi, [ebp + nb133nf_pos] mov eax, [ebp + nb133nf_jjnr] shl ecx, 2 add eax, ecx mov [esp + nb133nf_innerjjnr], eax ;# pointer to jjnr[nj0] mov ecx, edx sub edx, 4 add ecx, [esp + nb133nf_ninner] mov [esp + nb133nf_ninner], ecx add edx, 0 mov [esp + nb133nf_innerk], edx ;# number of innerloop atoms jge .nb133nf_unroll_loop jmp .nb133nf_odd_inner.nb133nf_unroll_loop: ;# quad-unroll innerloop here
⌨️ 快捷键说明
复制代码Ctrl + C
搜索代码Ctrl + F
全屏模式F11
增大字号Ctrl + =
减小字号Ctrl + -
显示快捷键?