|
|
|
@ -522,6 +522,50 @@ static void OPNAME ## h264_qpel4_h_lowpass_ ## MMX(uint8_t *dst, uint8_t *src, i |
|
|
|
|
: "memory"\
|
|
|
|
|
);\
|
|
|
|
|
}\
|
|
|
|
|
static void OPNAME ## h264_qpel4_h_lowpass_l2_ ## MMX(uint8_t *dst, uint8_t *src, uint8_t *src2, int dstStride, int src2Stride){\
|
|
|
|
|
int h=4;\
|
|
|
|
|
\
|
|
|
|
|
asm volatile(\
|
|
|
|
|
"pxor %%mm7, %%mm7 \n\t"\
|
|
|
|
|
"movq %6, %%mm4 \n\t"\
|
|
|
|
|
"movq %7, %%mm5 \n\t"\
|
|
|
|
|
"1: \n\t"\
|
|
|
|
|
"movd -1(%0), %%mm1 \n\t"\
|
|
|
|
|
"movd (%0), %%mm2 \n\t"\
|
|
|
|
|
"movd 1(%0), %%mm3 \n\t"\
|
|
|
|
|
"movd 2(%0), %%mm0 \n\t"\
|
|
|
|
|
"punpcklbw %%mm7, %%mm1 \n\t"\
|
|
|
|
|
"punpcklbw %%mm7, %%mm2 \n\t"\
|
|
|
|
|
"punpcklbw %%mm7, %%mm3 \n\t"\
|
|
|
|
|
"punpcklbw %%mm7, %%mm0 \n\t"\
|
|
|
|
|
"paddw %%mm0, %%mm1 \n\t"\
|
|
|
|
|
"paddw %%mm3, %%mm2 \n\t"\
|
|
|
|
|
"movd -2(%0), %%mm0 \n\t"\
|
|
|
|
|
"movd 3(%0), %%mm3 \n\t"\
|
|
|
|
|
"punpcklbw %%mm7, %%mm0 \n\t"\
|
|
|
|
|
"punpcklbw %%mm7, %%mm3 \n\t"\
|
|
|
|
|
"paddw %%mm3, %%mm0 \n\t"\
|
|
|
|
|
"psllw $2, %%mm2 \n\t"\
|
|
|
|
|
"psubw %%mm1, %%mm2 \n\t"\
|
|
|
|
|
"pmullw %%mm4, %%mm2 \n\t"\
|
|
|
|
|
"paddw %%mm5, %%mm0 \n\t"\
|
|
|
|
|
"paddw %%mm2, %%mm0 \n\t"\
|
|
|
|
|
"movd (%2), %%mm3 \n\t"\
|
|
|
|
|
"psraw $5, %%mm0 \n\t"\
|
|
|
|
|
"packuswb %%mm0, %%mm0 \n\t"\
|
|
|
|
|
PAVGB" %%mm3, %%mm0 \n\t"\
|
|
|
|
|
OP(%%mm0, (%1),%%mm6, d)\
|
|
|
|
|
"add %5, %0 \n\t"\
|
|
|
|
|
"add %5, %1 \n\t"\
|
|
|
|
|
"add %4, %2 \n\t"\
|
|
|
|
|
"decl %3 \n\t"\
|
|
|
|
|
" jnz 1b \n\t"\
|
|
|
|
|
: "+a"(src), "+c"(dst), "+d"(src2), "+m"(h)\
|
|
|
|
|
: "D"((long)src2Stride), "S"((long)dstStride),\
|
|
|
|
|
"m"(ff_pw_5), "m"(ff_pw_16)\
|
|
|
|
|
: "memory"\
|
|
|
|
|
);\
|
|
|
|
|
}\
|
|
|
|
|
static void OPNAME ## h264_qpel4_v_lowpass_ ## MMX(uint8_t *dst, uint8_t *src, int dstStride, int srcStride){\
|
|
|
|
|
src -= 2*srcStride;\
|
|
|
|
|
asm volatile(\
|
|
|
|
@ -672,6 +716,67 @@ static void OPNAME ## h264_qpel8_h_lowpass_ ## MMX(uint8_t *dst, uint8_t *src, i |
|
|
|
|
);\
|
|
|
|
|
}\
|
|
|
|
|
\
|
|
|
|
|
static void OPNAME ## h264_qpel8_h_lowpass_l2_ ## MMX(uint8_t *dst, uint8_t *src, uint8_t *src2, int dstStride, int src2Stride){\
|
|
|
|
|
int h=8;\
|
|
|
|
|
asm volatile(\
|
|
|
|
|
"pxor %%mm7, %%mm7 \n\t"\
|
|
|
|
|
"movq %6, %%mm6 \n\t"\
|
|
|
|
|
"1: \n\t"\
|
|
|
|
|
"movq (%0), %%mm0 \n\t"\
|
|
|
|
|
"movq 1(%0), %%mm2 \n\t"\
|
|
|
|
|
"movq %%mm0, %%mm1 \n\t"\
|
|
|
|
|
"movq %%mm2, %%mm3 \n\t"\
|
|
|
|
|
"punpcklbw %%mm7, %%mm0 \n\t"\
|
|
|
|
|
"punpckhbw %%mm7, %%mm1 \n\t"\
|
|
|
|
|
"punpcklbw %%mm7, %%mm2 \n\t"\
|
|
|
|
|
"punpckhbw %%mm7, %%mm3 \n\t"\
|
|
|
|
|
"paddw %%mm2, %%mm0 \n\t"\
|
|
|
|
|
"paddw %%mm3, %%mm1 \n\t"\
|
|
|
|
|
"psllw $2, %%mm0 \n\t"\
|
|
|
|
|
"psllw $2, %%mm1 \n\t"\
|
|
|
|
|
"movq -1(%0), %%mm2 \n\t"\
|
|
|
|
|
"movq 2(%0), %%mm4 \n\t"\
|
|
|
|
|
"movq %%mm2, %%mm3 \n\t"\
|
|
|
|
|
"movq %%mm4, %%mm5 \n\t"\
|
|
|
|
|
"punpcklbw %%mm7, %%mm2 \n\t"\
|
|
|
|
|
"punpckhbw %%mm7, %%mm3 \n\t"\
|
|
|
|
|
"punpcklbw %%mm7, %%mm4 \n\t"\
|
|
|
|
|
"punpckhbw %%mm7, %%mm5 \n\t"\
|
|
|
|
|
"paddw %%mm4, %%mm2 \n\t"\
|
|
|
|
|
"paddw %%mm3, %%mm5 \n\t"\
|
|
|
|
|
"psubw %%mm2, %%mm0 \n\t"\
|
|
|
|
|
"psubw %%mm5, %%mm1 \n\t"\
|
|
|
|
|
"pmullw %%mm6, %%mm0 \n\t"\
|
|
|
|
|
"pmullw %%mm6, %%mm1 \n\t"\
|
|
|
|
|
"movd -2(%0), %%mm2 \n\t"\
|
|
|
|
|
"movd 7(%0), %%mm5 \n\t"\
|
|
|
|
|
"punpcklbw %%mm7, %%mm2 \n\t"\
|
|
|
|
|
"punpcklbw %%mm7, %%mm5 \n\t"\
|
|
|
|
|
"paddw %%mm3, %%mm2 \n\t"\
|
|
|
|
|
"paddw %%mm5, %%mm4 \n\t"\
|
|
|
|
|
"movq %7, %%mm5 \n\t"\
|
|
|
|
|
"paddw %%mm5, %%mm2 \n\t"\
|
|
|
|
|
"paddw %%mm5, %%mm4 \n\t"\
|
|
|
|
|
"paddw %%mm2, %%mm0 \n\t"\
|
|
|
|
|
"paddw %%mm4, %%mm1 \n\t"\
|
|
|
|
|
"psraw $5, %%mm0 \n\t"\
|
|
|
|
|
"psraw $5, %%mm1 \n\t"\
|
|
|
|
|
"movq (%2), %%mm4 \n\t"\
|
|
|
|
|
"packuswb %%mm1, %%mm0 \n\t"\
|
|
|
|
|
PAVGB" %%mm4, %%mm0 \n\t"\
|
|
|
|
|
OP(%%mm0, (%1),%%mm5, q)\
|
|
|
|
|
"add %5, %0 \n\t"\
|
|
|
|
|
"add %5, %1 \n\t"\
|
|
|
|
|
"add %4, %2 \n\t"\
|
|
|
|
|
"decl %3 \n\t"\
|
|
|
|
|
" jnz 1b \n\t"\
|
|
|
|
|
: "+a"(src), "+c"(dst), "+d"(src2), "+m"(h)\
|
|
|
|
|
: "D"((long)src2Stride), "S"((long)dstStride),\
|
|
|
|
|
"m"(ff_pw_5), "m"(ff_pw_16)\
|
|
|
|
|
: "memory"\
|
|
|
|
|
);\
|
|
|
|
|
}\
|
|
|
|
|
\
|
|
|
|
|
static inline void OPNAME ## h264_qpel8or16_v_lowpass_ ## MMX(uint8_t *dst, uint8_t *src, int dstStride, int srcStride, int h){\
|
|
|
|
|
int w= 2;\
|
|
|
|
|
src -= 2*srcStride;\
|
|
|
|
@ -846,6 +951,16 @@ static void OPNAME ## h264_qpel16_h_lowpass_ ## MMX(uint8_t *dst, uint8_t *src, |
|
|
|
|
OPNAME ## h264_qpel8_h_lowpass_ ## MMX(dst+8, src+8, dstStride, srcStride);\
|
|
|
|
|
}\
|
|
|
|
|
\
|
|
|
|
|
static void OPNAME ## h264_qpel16_h_lowpass_l2_ ## MMX(uint8_t *dst, uint8_t *src, uint8_t *src2, int dstStride, int src2Stride){\
|
|
|
|
|
OPNAME ## h264_qpel8_h_lowpass_l2_ ## MMX(dst , src , src2 , dstStride, src2Stride);\
|
|
|
|
|
OPNAME ## h264_qpel8_h_lowpass_l2_ ## MMX(dst+8, src+8, src2+8, dstStride, src2Stride);\
|
|
|
|
|
src += 8*dstStride;\
|
|
|
|
|
dst += 8*dstStride;\
|
|
|
|
|
src2 += 8*src2Stride;\
|
|
|
|
|
OPNAME ## h264_qpel8_h_lowpass_l2_ ## MMX(dst , src , src2 , dstStride, src2Stride);\
|
|
|
|
|
OPNAME ## h264_qpel8_h_lowpass_l2_ ## MMX(dst+8, src+8, src2+8, dstStride, src2Stride);\
|
|
|
|
|
}\
|
|
|
|
|
\
|
|
|
|
|
static void OPNAME ## h264_qpel8_hv_lowpass_ ## MMX(uint8_t *dst, int16_t *tmp, uint8_t *src, int dstStride, int tmpStride, int srcStride){\
|
|
|
|
|
OPNAME ## h264_qpel8or16_hv_lowpass_ ## MMX(dst , tmp , src , dstStride, tmpStride, srcStride, 8);\
|
|
|
|
|
}\
|
|
|
|
@ -933,10 +1048,7 @@ static void OPNAME ## h264_qpel ## SIZE ## _mc00_ ## MMX (uint8_t *dst, uint8_t |
|
|
|
|
}\
|
|
|
|
|
\
|
|
|
|
|
static void OPNAME ## h264_qpel ## SIZE ## _mc10_ ## MMX(uint8_t *dst, uint8_t *src, int stride){\
|
|
|
|
|
uint64_t temp[SIZE*SIZE/8];\
|
|
|
|
|
uint8_t * const half= (uint8_t*)temp;\
|
|
|
|
|
put_h264_qpel ## SIZE ## _h_lowpass_ ## MMX(half, src, SIZE, stride);\
|
|
|
|
|
OPNAME ## pixels ## SIZE ## _l2_ ## MMX(dst, src, half, stride, stride, SIZE);\
|
|
|
|
|
OPNAME ## h264_qpel ## SIZE ## _h_lowpass_l2_ ## MMX(dst, src, src, stride, stride);\
|
|
|
|
|
}\
|
|
|
|
|
\
|
|
|
|
|
static void OPNAME ## h264_qpel ## SIZE ## _mc20_ ## MMX(uint8_t *dst, uint8_t *src, int stride){\
|
|
|
|
@ -944,10 +1056,7 @@ static void OPNAME ## h264_qpel ## SIZE ## _mc20_ ## MMX(uint8_t *dst, uint8_t * |
|
|
|
|
}\
|
|
|
|
|
\
|
|
|
|
|
static void OPNAME ## h264_qpel ## SIZE ## _mc30_ ## MMX(uint8_t *dst, uint8_t *src, int stride){\
|
|
|
|
|
uint64_t temp[SIZE*SIZE/8];\
|
|
|
|
|
uint8_t * const half= (uint8_t*)temp;\
|
|
|
|
|
put_h264_qpel ## SIZE ## _h_lowpass_ ## MMX(half, src, SIZE, stride);\
|
|
|
|
|
OPNAME ## pixels ## SIZE ## _l2_ ## MMX(dst, src+1, half, stride, stride, SIZE);\
|
|
|
|
|
OPNAME ## h264_qpel ## SIZE ## _h_lowpass_l2_ ## MMX(dst, src, src+1, stride, stride);\
|
|
|
|
|
}\
|
|
|
|
|
\
|
|
|
|
|
static void OPNAME ## h264_qpel ## SIZE ## _mc01_ ## MMX(uint8_t *dst, uint8_t *src, int stride){\
|
|
|
|
@ -969,39 +1078,31 @@ static void OPNAME ## h264_qpel ## SIZE ## _mc03_ ## MMX(uint8_t *dst, uint8_t * |
|
|
|
|
}\
|
|
|
|
|
\
|
|
|
|
|
static void OPNAME ## h264_qpel ## SIZE ## _mc11_ ## MMX(uint8_t *dst, uint8_t *src, int stride){\
|
|
|
|
|
uint64_t temp[SIZE*SIZE/4];\
|
|
|
|
|
uint8_t * const halfH= (uint8_t*)temp;\
|
|
|
|
|
uint8_t * const halfV= ((uint8_t*)temp) + SIZE*SIZE;\
|
|
|
|
|
put_h264_qpel ## SIZE ## _h_lowpass_ ## MMX(halfH, src, SIZE, stride);\
|
|
|
|
|
uint64_t temp[SIZE*SIZE/8];\
|
|
|
|
|
uint8_t * const halfV= (uint8_t*)temp;\
|
|
|
|
|
put_h264_qpel ## SIZE ## _v_lowpass_ ## MMX(halfV, src, SIZE, stride);\
|
|
|
|
|
OPNAME ## pixels ## SIZE ## _l2_ ## MMX(dst, halfH, halfV, stride, SIZE, SIZE);\
|
|
|
|
|
OPNAME ## h264_qpel ## SIZE ## _h_lowpass_l2_ ## MMX(dst, src, halfV, stride, SIZE);\
|
|
|
|
|
}\
|
|
|
|
|
\
|
|
|
|
|
static void OPNAME ## h264_qpel ## SIZE ## _mc31_ ## MMX(uint8_t *dst, uint8_t *src, int stride){\
|
|
|
|
|
uint64_t temp[SIZE*SIZE/4];\
|
|
|
|
|
uint8_t * const halfH= (uint8_t*)temp;\
|
|
|
|
|
uint8_t * const halfV= ((uint8_t*)temp) + SIZE*SIZE;\
|
|
|
|
|
put_h264_qpel ## SIZE ## _h_lowpass_ ## MMX(halfH, src, SIZE, stride);\
|
|
|
|
|
uint64_t temp[SIZE*SIZE/8];\
|
|
|
|
|
uint8_t * const halfV= (uint8_t*)temp;\
|
|
|
|
|
put_h264_qpel ## SIZE ## _v_lowpass_ ## MMX(halfV, src+1, SIZE, stride);\
|
|
|
|
|
OPNAME ## pixels ## SIZE ## _l2_ ## MMX(dst, halfH, halfV, stride, SIZE, SIZE);\
|
|
|
|
|
OPNAME ## h264_qpel ## SIZE ## _h_lowpass_l2_ ## MMX(dst, src, halfV, stride, SIZE);\
|
|
|
|
|
}\
|
|
|
|
|
\
|
|
|
|
|
static void OPNAME ## h264_qpel ## SIZE ## _mc13_ ## MMX(uint8_t *dst, uint8_t *src, int stride){\
|
|
|
|
|
uint64_t temp[SIZE*SIZE/4];\
|
|
|
|
|
uint8_t * const halfH= (uint8_t*)temp;\
|
|
|
|
|
uint8_t * const halfV= ((uint8_t*)temp) + SIZE*SIZE;\
|
|
|
|
|
put_h264_qpel ## SIZE ## _h_lowpass_ ## MMX(halfH, src + stride, SIZE, stride);\
|
|
|
|
|
uint64_t temp[SIZE*SIZE/8];\
|
|
|
|
|
uint8_t * const halfV= (uint8_t*)temp;\
|
|
|
|
|
put_h264_qpel ## SIZE ## _v_lowpass_ ## MMX(halfV, src, SIZE, stride);\
|
|
|
|
|
OPNAME ## pixels ## SIZE ## _l2_ ## MMX(dst, halfH, halfV, stride, SIZE, SIZE);\
|
|
|
|
|
OPNAME ## h264_qpel ## SIZE ## _h_lowpass_l2_ ## MMX(dst, src+stride, halfV, stride, SIZE);\
|
|
|
|
|
}\
|
|
|
|
|
\
|
|
|
|
|
static void OPNAME ## h264_qpel ## SIZE ## _mc33_ ## MMX(uint8_t *dst, uint8_t *src, int stride){\
|
|
|
|
|
uint64_t temp[SIZE*SIZE/4];\
|
|
|
|
|
uint8_t * const halfH= (uint8_t*)temp;\
|
|
|
|
|
uint8_t * const halfV= ((uint8_t*)temp) + SIZE*SIZE;\
|
|
|
|
|
put_h264_qpel ## SIZE ## _h_lowpass_ ## MMX(halfH, src + stride, SIZE, stride);\
|
|
|
|
|
uint64_t temp[SIZE*SIZE/8];\
|
|
|
|
|
uint8_t * const halfV= (uint8_t*)temp;\
|
|
|
|
|
put_h264_qpel ## SIZE ## _v_lowpass_ ## MMX(halfV, src+1, SIZE, stride);\
|
|
|
|
|
OPNAME ## pixels ## SIZE ## _l2_ ## MMX(dst, halfH, halfV, stride, SIZE, SIZE);\
|
|
|
|
|
OPNAME ## h264_qpel ## SIZE ## _h_lowpass_l2_ ## MMX(dst, src+stride, halfV, stride, SIZE);\
|
|
|
|
|
}\
|
|
|
|
|
\
|
|
|
|
|
static void OPNAME ## h264_qpel ## SIZE ## _mc22_ ## MMX(uint8_t *dst, uint8_t *src, int stride){\
|
|
|
|
@ -1011,23 +1112,19 @@ static void OPNAME ## h264_qpel ## SIZE ## _mc22_ ## MMX(uint8_t *dst, uint8_t * |
|
|
|
|
}\
|
|
|
|
|
\
|
|
|
|
|
static void OPNAME ## h264_qpel ## SIZE ## _mc21_ ## MMX(uint8_t *dst, uint8_t *src, int stride){\
|
|
|
|
|
uint64_t temp[SIZE*(SIZE<8?12:24)/4 + SIZE*SIZE/4];\
|
|
|
|
|
uint8_t * const halfH= (uint8_t*)temp;\
|
|
|
|
|
uint8_t * const halfHV= ((uint8_t*)temp) + SIZE*SIZE;\
|
|
|
|
|
int16_t * const tmp= ((int16_t*)temp) + SIZE*SIZE;\
|
|
|
|
|
put_h264_qpel ## SIZE ## _h_lowpass_ ## MMX(halfH, src, SIZE, stride);\
|
|
|
|
|
uint64_t temp[SIZE*(SIZE<8?12:24)/4 + SIZE*SIZE/8];\
|
|
|
|
|
uint8_t * const halfHV= (uint8_t*)temp;\
|
|
|
|
|
int16_t * const tmp= ((int16_t*)temp) + SIZE*SIZE/2;\
|
|
|
|
|
put_h264_qpel ## SIZE ## _hv_lowpass_ ## MMX(halfHV, tmp, src, SIZE, SIZE, stride);\
|
|
|
|
|
OPNAME ## pixels ## SIZE ## _l2_ ## MMX(dst, halfH, halfHV, stride, SIZE, SIZE);\
|
|
|
|
|
OPNAME ## h264_qpel ## SIZE ## _h_lowpass_l2_ ## MMX(dst, src, halfHV, stride, SIZE);\
|
|
|
|
|
}\
|
|
|
|
|
\
|
|
|
|
|
static void OPNAME ## h264_qpel ## SIZE ## _mc23_ ## MMX(uint8_t *dst, uint8_t *src, int stride){\
|
|
|
|
|
uint64_t temp[SIZE*(SIZE<8?12:24)/4 + SIZE*SIZE/4];\
|
|
|
|
|
uint8_t * const halfH= (uint8_t*)temp;\
|
|
|
|
|
uint8_t * const halfHV= ((uint8_t*)temp) + SIZE*SIZE;\
|
|
|
|
|
int16_t * const tmp= ((int16_t*)temp) + SIZE*SIZE;\
|
|
|
|
|
put_h264_qpel ## SIZE ## _h_lowpass_ ## MMX(halfH, src + stride, SIZE, stride);\
|
|
|
|
|
uint64_t temp[SIZE*(SIZE<8?12:24)/4 + SIZE*SIZE/8];\
|
|
|
|
|
uint8_t * const halfHV= (uint8_t*)temp;\
|
|
|
|
|
int16_t * const tmp= ((int16_t*)temp) + SIZE*SIZE/2;\
|
|
|
|
|
put_h264_qpel ## SIZE ## _hv_lowpass_ ## MMX(halfHV, tmp, src, SIZE, SIZE, stride);\
|
|
|
|
|
OPNAME ## pixels ## SIZE ## _l2_ ## MMX(dst, halfH, halfHV, stride, SIZE, SIZE);\
|
|
|
|
|
OPNAME ## h264_qpel ## SIZE ## _h_lowpass_l2_ ## MMX(dst, src+stride, halfHV, stride, SIZE);\
|
|
|
|
|
}\
|
|
|
|
|
\
|
|
|
|
|
static void OPNAME ## h264_qpel ## SIZE ## _mc12_ ## MMX(uint8_t *dst, uint8_t *src, int stride){\
|
|
|
|
|