nb_kernel010_ia64_double.s
来自「最著名最快的分子模拟软件」· S 代码 · 共 892 行 · 第 1/2 页
S
892 行
(pDone) br.cond.spnt.few finish } ;; // THREAD PROLOGUE 4 { .mmi ld4 ggid = [gidPtr], 4 shladd II3 = II, 1, II shladd IS3 = IS, 1, IS } { .mfi ld4 NJ1 = [jindexPtr], 4 nop 0x0 shladd jjnrPtr = NJ0, 2, JJNR } ;;// THREAD PROLOGUE 5 { .mmi cmp.lt pCont, pDone = Tmp2, NRI shladd FShiftIS = IS3, 3, FSHIFT shladd typePtr = II, 2, TYPE } { .mmi shladd posPtr = II3, 3, POSITION shladd FActII = II3, 3, FACTION shladd shiftVPtr = IS3, 3, SHIFTVEC } ;;// THREAD PROLOGUE 6 { .mmi ld4 jnr = [jjnrPtr], 4 (pCont) ld4 IS = [shiftPtr], 4 nop 0x0 } { .mmi (pCont) ld4 II = [iinrPtr], 4 ld4 NTI = [typePtr] (pLast) mov NN1 = NRI } ;;// 12 bundles in thread prologue - still alignedouterLoop: // At this point in the outer loop, the following values are ready // // FActII Pointer to FACTION XYZ for II // FShiftIS Pointer to FSHIFT XYZ for IS // shiftVPtr Pointer to current shift XYZ values // posPtr Pointer to current XYZ position // chargePtr Pointer to current atom charge // ggid Index for Vc array // jjnr Pointer to next neighbor index // jnr Current jnr value // NJ0, NJ1 Bounds of current neighbor list // // Load up all the floating-point values (yes, McKinley can do 4 FP loads // per cycle) and initialize the loop counters and predicates. Compute // the initial position <x, y, z> and charge. If this isn't the last time // through the loop, start loading the next value for NJ1 - we already // moved the previous NJ1 -> NJ0.// OUTER PROLOGUE 1 { .mfi nop 0x0 mov FIX = f0 add Nouter = 1, Nouter } { .mmf ldfd shX = [shiftVPtr], 8 ldfd PosX = [posPtr], 8 mov FIY = f0 } ;;// OUTER PROLOGUE 2 { .mmf setf.sig f32 = NTI ldfd shY = [shiftVPtr], 8 nop 0x0 } { .mfi ldfd PosY = [posPtr], 8 nop 0x0 sub InnerCnt = NJ1, NJ0, 1 } ;; { .mmf ldfd shZ = [shiftVPtr] ldfd PosZ = [posPtr] mov FIZ = f0 } { .mmi ldfd FShiftX = [FShiftIS], 8 ldfd FActIX = [FActII], 8 shladd VNBPtr = ggid, 3, VNB } ;;// OUTER PROLOGUE 4 { .mmf ldfd FShiftY = [FShiftIS], 8 ldfd FActIY = [FActII], 8 xma.l f32 = f32, f33, fZero };;// OUTER PROLOGUE 5 { .mmi ldfd FActIZ = [FActII], -16 ldfd FShiftZ = [FShiftIS], -16 mov NJ0 = NJ1 } ;;// OUTER PROLOGUE 6 { .mfi ldfd VNBTotal = [VNBPtr] fadd IX = shX, PosX add NN0 = 1, NN0 } { .mfi (pCont) ld4 NJ1 = [jindexPtr], 4 // This may seem strange, but we set the first stage of the // pipe to execute this way because setting pr.rot doesn't take // into account how much the predicates have rotated. If this is // the first time through, we cleared all the pipeline predicates // in the initialization. If not, flushing the pipeline set all // the pipeline predicates to 0 cmp.eq pPipe[0], p0 = zero, zero } ;;// OUTER PROLOGUE 7 { .mfi cmp.lt pCont, pDone = NN0, NN1 fadd IY = shY, PosY mov ar.lc = InnerCnt } ;;// OUTER PROLOGUE 8 { .mfi getf.sig NTI = f32 fadd IZ = shZ, PosZ mov ar.ec = PIPE_DEPTH } ;;// 12 bundles in outer loop - still aligned. // The inner loop is a 6-stage pipeline. The serial sequence of float ops // is folded into a 17-cycle loop (17 * 2 = 34 float ops, one empty), // then divided // into 5 stages. innerLoop:// INNER LOOP 1 { .mfi (pPipe[0]) shladd chargePtr = jnr, 3, CHARGE (pPipe[1]) fsub DX[1] = IX, DX[1] (pPipe[0]) shladd jnr3 = jnr, 1, jnr } // We march through jjnr[] sequentially, so it's usually a good idea // to preload the next value. However, we don't want to do this if // (1) we're in the epilogue or (2) this is the last time through and // there are no more atoms to inspect. Thus, we keep track of the loop // trip and use the logic below to see if we should load ahead .pred.rel "mutex", pCont, pDone { .mfi (pCont) cmp.ge pJJNR, p0 = InnerCnt, zero (pPipe[3]) fma RInv2[1] = RInv2[1], RInv2Err[1], RInv2[1] (pDone) cmp.gt pJJNR, p0 = InnerCnt, zero } ;;// INNER LOOP 2 { .mfi nop 0x0 (pPipe[1]) fsub DY[1] = IY, DY[1] (pPipe[0]) add InnerCnt = -1, InnerCnt } { .mfi (pPipe[0]) shladd posPtr = jnr3, 3, POSITION (pPipe[5]) fma FScalar[1] = fTWELVE, C12[3], FScalar[1] (pPipe[0]) shladd FActPtr[0] = jnr3, 3, FACTION } ;;// INNER LOOP 3 { .mfi (pPipe[0]) ldfd JX = [posPtr], 8 (pPipe[1]) fsub DZ[1] = IZ, DZ[1] (pPipe[0]) shladd TypeJ[0] = jnr, 2, TYPE } { .mfi (pJJNR) ld4 jnr = [jjnrPtr], 4 (pPipe[2]) frcpa RInv2[0], p0 = fOne, RSqr[1] (pPipe[2]) add TypeJ[2] = NTI, TypeJ[2] } ;;// INNER LOOP 4 { .mfi (pPipe[0]) ldfd JY = [posPtr], 8 (pPipe[4]) fmpy RInv6[1] = RInv6[1], RInv2[2] (pPipe[0]) add Ninner = 1, Ninner } { .mfi nop 0x0 (pPipe[5]) fsub VNBTotal = VNBTotal, C6[3] nop 0x0 } ;;// INNER LOOP 5 { .mfi (pPipe[0]) ldfd JZ = [posPtr], 8 (pPipe[1]) fmpy RSqr[0] = DX[1], DX[1] (pJJNR) add jjnrPtr = JJNR_PREFETCH_DISTANCE, jjnrPtr } { .mfi nop 0x0 (pPipe[3]) fnma RInv2Err[1] = RSqr[2], RInv2[1], fOne nop 0x0 } ;;// INNER LOOP 6 { .mfi (pPipe[0]) ldfd FActX[0] = [FActPtr[0]], 8 (pPipe[5]) fmpy FScalar[1] = RInv2[3], FScalar[1] nop 0x0 } { .mfi nop 0x0 (pPipe[6]) fma FIX = DX[6], FScalar[2], FIX nop 0x0 } ;;// INNER LOOP 7 { .mfi (pPipe[0]) ldfd FActY[0] = [FActPtr[0]], 8 (pPipe[2]) fnma RInv2Err[0] = RInv2[0], RSqr[1], fOne nop 0x0 } { .mfi (pJJNR) lfetch.nta [jjnrPtr] (pPipe[6]) fma FIY = DY[6], FScalar[2], FIY nop 0x0 } ;;// INNER LOOP 8 { .mfi (pPipe[0]) ldfd FActZ[0] = [FActPtr[0]], -16 (pPipe[4]) fmpy C6[2] = C6[2], RInv6[1] (pPipe[2]) shladd TypeJ[2] = TypeJ[2], 4, NBFP } { .mfi nop 0x0 (pPipe[4]) fmpy RInv6[1] = RInv6[1], RInv6[1] nop 0x0 } ;;// INNER LOOP 9 { .mfi (pPipe[2]) ldfd C6[0] = [TypeJ[2]], 8 (pPipe[1]) fma RSqr[0] = DY[1], DY[1], RSqr[0] (pJJNR) add jjnrPtr = -JJNR_PREFETCH_DISTANCE, jjnrPtr } { .mfi (pPipe[0]) ld4 TypeJ[0] = [TypeJ[0]] (pPipe[3]) fma RInv2[1] = RInv2[1], RInv2Err[1], RInv2[1] nop 0x0 } ;;// INNER LOOP 10 { .mfi (pPipe[2]) ldfd C12[0] = [TypeJ[2]] (pPipe[5]) fadd VNBTotal = VNBTotal, C12[3] nop 0x0 } { .mfi nop 0x0 (pPipe[6]) fma FIZ = DZ[6], FScalar[2], FIZ nop 0x0 } ;;// INNER LOOP 11 { .mfi nop 0x0 (pPipe[2]) fma RInv2Err[0] = RInv2Err[0], RInv2Err[0], RInv2Err[0] nop 0x0 } { .mfi nop 0x0 (pPipe[5]) fnma FActX[5] = FScalar[1], DX[5], FActX[5] nop 0x0 } ;;// INNER LOOP 12 { .mfi (pPipe[6]) stfd [FActPtr[6]] = FActX[6], 8 (pPipe[4]) fmpy C12[2] = C12[2], RInv6[1] nop 0x0 } { .mfi nop 0x0 (pPipe[4]) fnma FScalar[0] = fSIX, C6[2], fZero nop 0x0 } ;;// INNER LOOP 13 { .mfi (pPipe[6]) stfd [FActPtr[6]] = FActY[6], 8 (pPipe[1]) fma RSqr[0] = DZ[1], DZ[1], RSqr[0] nop 0x0 } { .mfi nop 0x0 (pPipe[3]) fmpy RInv6[0] = RInv2[1], RInv2[1] nop 0x0 } ;;// INNER LOOP 14 { .mfi (pPipe[6]) stfd [FActPtr[6]] = FActZ[6], 8 (pPipe[5]) fnma FActY[5] = FScalar[1], DY[5], FActY[5] nop 0x0 } { .mfb nop 0x0 (pPipe[5]) fnma FActZ[5] = FScalar[1], DZ[5], FActZ[5] br.ctop.sptk.many innerLoop } ;;// End of modulo-scheduled inner loop // Having finshed the loop, we now compute various quantities to // store. In paralllel, start computing computing some of the values // for the next loop trip, if we're going there.// OUTER EPILOGUE 1 { .mfi (pCont) shladd typePtr = II, 2, TYPE nop 0x0 (pCont) shladd II3 = II, 1, II } { .mfi nop 0x0 nop 0x0 (pCont) shladd IS3 = IS, 1, IS } ;;// OUTER EPILOGUE 2 { .mfi (pCont) ld4 IS = [shiftPtr], 4 fadd FActIX = FActIX, FIX nop 0x0 } { .mmf (pCont) setf.sig f33 = NTYPE (pCont) ld4 II = [iinrPtr] ,4 fadd FShiftX = FShiftX, FIX } ;;// OUTER EPILOGUE 3 { .mfi (pCont) ld4 NTI = [typePtr] fadd FActIY = FActIY, FIY (pCont) shladd shiftVPtr = IS3, 3, SHIFTVEC } { .mfi nop 0x0 fadd FShiftY = FShiftY, FIY (pCont) shladd posPtr = II3, 3, POSITION } ;;// OUTER EPILOGUE 4 { .mfi nop 0x0 fadd FActIZ = FActIZ, FIZ nop 0x0 } { .mfi nop 0x0 fadd FShiftZ = FShiftZ, FIZ nop 0x0 } ;;// OUTER EPILOGUE 5 { .mmi stfd [FActII] = FActIX, 8 stfd [FShiftIS] = FShiftX, 8 nop 0x0 } { .mmi stfd [VNBPtr] = VNBTotal (pCont) ld4 ggid = [gidPtr], 4 nop 0x0 } ;;// OUTER EPILOGUE 6 { .mmi stfd [FActII] = FActIY, 8 stfd [FShiftIS] = FShiftY, 8 nop 0x0 } ;;// OUTER EPILOGUE 7 { .mmi stfd [FActII] = FActIZ nop 0x0 (pCont) shladd FActII = II3, 3, FACTION } { .mib stfd [FShiftIS] = FShiftZ (pCont) shladd FShiftIS = IS3, 3, FSHIFT (pCont) br.cond.sptk.many outerLoop } ;; // Finish if this was the last chunk, or do another thread-loop iteration// THREAD EPILOGUE 1 { .mib nop 0x0 nop 0x0 (pMore) br.cond.sptk.many threadLoop } ;; // Ready to exit - restore the floating-point registers we saved, the // loop counter, and the predicates, then we're done. Note that the // stack pointer has the address of the last saved FP register.finish:// EXIT 1 { .mmi mov fillP0 = sp add fillP1 = 16, sp mov ar.lc = LCSave } { .mmi st4 [OuterIter] = Nouter st4 [InnerIter] = Ninner nop 0x0 } ;;// EXIT 2 { .mmi ldf.fill fs5 = [fillP0], 32 ldf.fill fs4 = [fillP1], 32 mov pr = PRSave, 0x1ffff } ;;// EXIT 3 { .mmi ldf.fill fs3 = [fillP0], 32 ldf.fill fs2 = [fillP1], 32 add sp = 5 * 16, sp } ;;// EXIT 4 { .mmb ldf.fill fs1 = [fillP0] ldf.fill fs0 = [fillP1] br.ret.sptk.few rp } ;; .endp nb_kernel010_ia64_double
⌨️ 快捷键说明
复制代码Ctrl + C
搜索代码Ctrl + F
全屏模式F11
增大字号Ctrl + =
减小字号Ctrl + -
显示快捷键?