⭐ 欢迎来到虫虫下载站! | 📦 资源下载 📁 资源专辑 ℹ️ 关于我们
⭐ 虫虫下载站

📄 simple_idct_vis.c

📁 ffmpeg的完整源代码和作者自己写的文档。不但有在Linux的工程哦
💻 C
📖 第 1 页 / 共 2 页
字号:
        "fpadd16 %%f24, %%f50, %%f24 \n\t"\        "fpsub16 %%f28, %%f48, %%f28 \n\t"\        "fpadd16 %%f18, %%f52, %%f18 \n\t"\        "fpsub16 %%f22, %%f54, %%f22 \n\t"\        "fpadd16 %%f26, %%f56, %%f26 \n\t"\        "fpsub16 %%f30, %%f58, %%f30 \n\t"\\        "fmul8sux16 %%f12, %%f42, %%f48 \n\t"\        "fmul8sux16 %%f12, %%f34, %%f50 \n\t"\        "fmul8sux16 %%f14, %%f44, %%f52 \n\t"\        "fmul8sux16 %%f14, %%f40, %%f54 \n\t"\        "fmul8sux16 %%f14, %%f36, %%f56 \n\t"\        "fmul8sux16 %%f14, %%f32, %%f58 \n\t"\\        "fpadd16 %%f16, %%f48, %%f16 \n\t"\        "fpsub16 %%f20, %%f50, %%f20 \n\t"\        "fpadd16 %%f24, %%f50, %%f24 \n\t"\        "fpsub16 %%f28, %%f48, %%f28 \n\t"\        "fpadd16 %%f18, %%f52, %%f18 \n\t"\        "fpsub16 %%f22, %%f54, %%f22 \n\t"\        "fpadd16 %%f26, %%f56, %%f26 \n\t"\        "fpsub16 %%f30, %%f58, %%f30 \n\t"\\        "fpsub16 %%f20, %%f12, %%f20 \n\t"\        "fpadd16 %%f24, %%f12, %%f24 \n\t"\        "fpsub16 %%f22, %%f14, %%f22 \n\t"\        "fpadd16 %%f26, %%f14, %%f26 \n\t"\        "fpsub16 %%f30, %%f14, %%f30 \n\t"\    /* final butterfly */\        "5:                          \n\t"\        "fpsub16 %%f16, %%f18, %%f48 \n\t"\        "fpsub16 %%f20, %%f22, %%f50 \n\t"\        "fpsub16 %%f24, %%f26, %%f52 \n\t"\        "fpsub16 %%f28, %%f30, %%f54 \n\t"\        "fpadd16 %%f16, %%f18, %%f16 \n\t"\        "fpadd16 %%f20, %%f22, %%f20 \n\t"\        "fpadd16 %%f24, %%f26, %%f24 \n\t"\        "fpadd16 %%f28, %%f30, %%f28 \n\t"\#define STOREROWS(out) \        "std %%f48, [" out "+112]          \n\t"\        "std %%f50, [" out "+96]           \n\t"\        "std %%f52, [" out "+80]           \n\t"\        "std %%f54, [" out "+64]           \n\t"\        "std %%f16, [" out "]              \n\t"\        "std %%f20, [" out "+16]           \n\t"\        "std %%f24, [" out "+32]           \n\t"\        "std %%f28, [" out "+48]           \n\t"\#define SCALEROWS \        "fmul8sux16 %%f46, %%f48, %%f48 \n\t"\        "fmul8sux16 %%f46, %%f50, %%f50 \n\t"\        "fmul8sux16 %%f46, %%f52, %%f52 \n\t"\        "fmul8sux16 %%f46, %%f54, %%f54 \n\t"\        "fmul8sux16 %%f46, %%f16, %%f16 \n\t"\        "fmul8sux16 %%f46, %%f20, %%f20 \n\t"\        "fmul8sux16 %%f46, %%f24, %%f24 \n\t"\        "fmul8sux16 %%f46, %%f28, %%f28 \n\t"\#define PUTPIXELSCLAMPED(dest) \        "fpack16 %%f48, %%f14 \n\t"\        "fpack16 %%f50, %%f12 \n\t"\        "fpack16 %%f16, %%f0  \n\t"\        "fpack16 %%f20, %%f2  \n\t"\        "fpack16 %%f24, %%f4  \n\t"\        "fpack16 %%f28, %%f6  \n\t"\        "fpack16 %%f54, %%f8  \n\t"\        "fpack16 %%f52, %%f10 \n\t"\        "st %%f0, [%3+" dest "]   \n\t"\        "st %%f2, [%5+" dest "]   \n\t"\        "st %%f4, [%6+" dest "]   \n\t"\        "st %%f6, [%7+" dest "]   \n\t"\        "st %%f8, [%8+" dest "]   \n\t"\        "st %%f10, [%9+" dest "]  \n\t"\        "st %%f12, [%10+" dest "] \n\t"\        "st %%f14, [%11+" dest "] \n\t"\#define ADDPIXELSCLAMPED(dest) \        "ldd [%5], %%f18         \n\t"\        "ld [%3+" dest"], %%f0   \n\t"\        "ld [%6+" dest"], %%f2   \n\t"\        "ld [%7+" dest"], %%f4   \n\t"\        "ld [%8+" dest"], %%f6   \n\t"\        "ld [%9+" dest"], %%f8   \n\t"\        "ld [%10+" dest"], %%f10 \n\t"\        "ld [%11+" dest"], %%f12 \n\t"\        "ld [%12+" dest"], %%f14 \n\t"\        "fmul8x16 %%f0, %%f18, %%f0   \n\t"\        "fmul8x16 %%f2, %%f18, %%f2   \n\t"\        "fmul8x16 %%f4, %%f18, %%f4   \n\t"\        "fmul8x16 %%f6, %%f18, %%f6   \n\t"\        "fmul8x16 %%f8, %%f18, %%f8   \n\t"\        "fmul8x16 %%f10, %%f18, %%f10 \n\t"\        "fmul8x16 %%f12, %%f18, %%f12 \n\t"\        "fmul8x16 %%f14, %%f18, %%f14 \n\t"\        "fpadd16 %%f0, %%f16, %%f0    \n\t"\        "fpadd16 %%f2, %%f20, %%f2    \n\t"\        "fpadd16 %%f4, %%f24, %%f4    \n\t"\        "fpadd16 %%f6, %%f28, %%f6    \n\t"\        "fpadd16 %%f8, %%f54, %%f8    \n\t"\        "fpadd16 %%f10, %%f52, %%f10  \n\t"\        "fpadd16 %%f12, %%f50, %%f12  \n\t"\        "fpadd16 %%f14, %%f48, %%f14  \n\t"\        "fpack16 %%f0, %%f0   \n\t"\        "fpack16 %%f2, %%f2   \n\t"\        "fpack16 %%f4, %%f4   \n\t"\        "fpack16 %%f6, %%f6   \n\t"\        "fpack16 %%f8, %%f8   \n\t"\        "fpack16 %%f10, %%f10 \n\t"\        "fpack16 %%f12, %%f12 \n\t"\        "fpack16 %%f14, %%f14 \n\t"\        "st %%f0, [%3+" dest "]   \n\t"\        "st %%f2, [%6+" dest "]   \n\t"\        "st %%f4, [%7+" dest "]   \n\t"\        "st %%f6, [%8+" dest "]   \n\t"\        "st %%f8, [%9+" dest "]   \n\t"\        "st %%f10, [%10+" dest "] \n\t"\        "st %%f12, [%11+" dest "] \n\t"\        "st %%f14, [%12+" dest "] \n\t"\inline void ff_simple_idct_vis(DCTELEM *data) {    int out1, out2, out3, out4;    DECLARE_ALIGNED_8(int16_t, temp[8*8]);    asm volatile(        INIT_IDCT#define ADDROUNDER        // shift right 16-4=12        LOADSCALE("%2+8")        IDCT4ROWS        STOREROWS("%3+8")        LOADSCALE("%2+0")        IDCT4ROWS        "std %%f48, [%3+112] \n\t"        "std %%f50, [%3+96]  \n\t"        "std %%f52, [%3+80]  \n\t"        "std %%f54, [%3+64]  \n\t"        // shift right 16+4        "ldd [%3+8], %%f18  \n\t"        "ldd [%3+24], %%f22 \n\t"        "ldd [%3+40], %%f26 \n\t"        "ldd [%3+56], %%f30 \n\t"        TRANSPOSE        IDCT4ROWS        SCALEROWS        STOREROWS("%2+0")        LOAD("%3+64")        TRANSPOSE        IDCT4ROWS        SCALEROWS        STOREROWS("%2+8")        : "=r" (out1), "=r" (out2), "=r" (out3), "=r" (out4)        : "0" (scale), "1" (coeffs), "2" (data), "3" (temp)    );}void ff_simple_idct_put_vis(uint8_t *dest, int line_size, DCTELEM *data) {    int out1, out2, out3, out4, out5;    int r1, r2, r3, r4, r5, r6, r7;    asm volatile(        "wr %%g0, 0x8, %%gsr \n\t"        INIT_IDCT        "add %3, %4, %5   \n\t"        "add %5, %4, %6   \n\t"        "add %6, %4, %7   \n\t"        "add %7, %4, %8   \n\t"        "add %8, %4, %9   \n\t"        "add %9, %4, %10  \n\t"        "add %10, %4, %11 \n\t"        // shift right 16-4=12        LOADSCALE("%2+8")        IDCT4ROWS        STOREROWS("%2+8")        LOADSCALE("%2+0")        IDCT4ROWS        "std %%f48, [%2+112] \n\t"        "std %%f50, [%2+96]  \n\t"        "std %%f52, [%2+80]  \n\t"        "std %%f54, [%2+64]  \n\t"#undef ADDROUNDER#define ADDROUNDER "fpadd16 %%f28, %%f46, %%f28 \n\t"        // shift right 16+4        "ldd [%2+8], %%f18  \n\t"        "ldd [%2+24], %%f22 \n\t"        "ldd [%2+40], %%f26 \n\t"        "ldd [%2+56], %%f30 \n\t"        TRANSPOSE        IDCT4ROWS        PUTPIXELSCLAMPED("0")        LOAD("%2+64")        TRANSPOSE        IDCT4ROWS        PUTPIXELSCLAMPED("4")        : "=r" (out1), "=r" (out2), "=r" (out3), "=r" (out4), "=r" (out5),          "=r" (r1), "=r" (r2), "=r" (r3), "=r" (r4), "=r" (r5), "=r" (r6), "=r" (r7)        : "0" (rounder), "1" (coeffs), "2" (data), "3" (dest), "4" (line_size)    );}void ff_simple_idct_add_vis(uint8_t *dest, int line_size, DCTELEM *data) {    int out1, out2, out3, out4, out5, out6;    int r1, r2, r3, r4, r5, r6, r7;    asm volatile(        "wr %%g0, 0x8, %%gsr \n\t"        INIT_IDCT        "add %3, %4, %6   \n\t"        "add %6, %4, %7   \n\t"        "add %7, %4, %8   \n\t"        "add %8, %4, %9   \n\t"        "add %9, %4, %10  \n\t"        "add %10, %4, %11 \n\t"        "add %11, %4, %12 \n\t"#undef ADDROUNDER#define ADDROUNDER        // shift right 16-4=12        LOADSCALE("%2+8")        IDCT4ROWS        STOREROWS("%2+8")        LOADSCALE("%2+0")        IDCT4ROWS        "std %%f48, [%2+112] \n\t"        "std %%f50, [%2+96]  \n\t"        "std %%f52, [%2+80]  \n\t"        "std %%f54, [%2+64]  \n\t"#undef ADDROUNDER#define ADDROUNDER "fpadd16 %%f28, %%f46, %%f28 \n\t"        // shift right 16+4        "ldd [%2+8], %%f18  \n\t"        "ldd [%2+24], %%f22 \n\t"        "ldd [%2+40], %%f26 \n\t"        "ldd [%2+56], %%f30 \n\t"        TRANSPOSE        IDCT4ROWS        ADDPIXELSCLAMPED("0")        LOAD("%2+64")        TRANSPOSE        IDCT4ROWS        ADDPIXELSCLAMPED("4")        : "=r" (out1), "=r" (out2), "=r" (out3), "=r" (out4), "=r" (out5), "=r" (out6),          "=r" (r1), "=r" (r2), "=r" (r3), "=r" (r4), "=r" (r5), "=r" (r6), "=r" (r7)        : "0" (rounder), "1" (coeffs), "2" (data), "3" (dest), "4" (line_size), "5" (expand)    );}

⌨️ 快捷键说明

复制代码 Ctrl + C
搜索代码 Ctrl + F
全屏模式 F11
切换主题 Ctrl + Shift + D
显示快捷键 ?
增大字号 Ctrl + =
减小字号 Ctrl + -