|
|
|
@ -32,7 +32,7 @@ static av_unused void OPNAME ## rv40_qpel8_h_lowpass(uint8_t *dst, uint8_t *src, |
|
|
|
|
const int h, const int C1, const int C2, const int SHIFT){\
|
|
|
|
|
uint8_t *cm = ff_cropTbl + MAX_NEG_CROP;\
|
|
|
|
|
int i;\
|
|
|
|
|
for(i=0; i<h; i++)\
|
|
|
|
|
for(i = 0; i < h; i++)\
|
|
|
|
|
{\
|
|
|
|
|
OP(dst[0], (src[-2] + src[ 3] - 5*(src[-1]+src[2]) + src[0]*C1 + src[1]*C2 + (1<<(SHIFT-1))) >> SHIFT);\
|
|
|
|
|
OP(dst[1], (src[-1] + src[ 4] - 5*(src[ 0]+src[3]) + src[1]*C1 + src[2]*C2 + (1<<(SHIFT-1))) >> SHIFT);\
|
|
|
|
@ -42,8 +42,8 @@ static av_unused void OPNAME ## rv40_qpel8_h_lowpass(uint8_t *dst, uint8_t *src, |
|
|
|
|
OP(dst[5], (src[ 3] + src[ 8] - 5*(src[ 4]+src[7]) + src[5]*C1 + src[6]*C2 + (1<<(SHIFT-1))) >> SHIFT);\
|
|
|
|
|
OP(dst[6], (src[ 4] + src[ 9] - 5*(src[ 5]+src[8]) + src[6]*C1 + src[7]*C2 + (1<<(SHIFT-1))) >> SHIFT);\
|
|
|
|
|
OP(dst[7], (src[ 5] + src[10] - 5*(src[ 6]+src[9]) + src[7]*C1 + src[8]*C2 + (1<<(SHIFT-1))) >> SHIFT);\
|
|
|
|
|
dst+=dstStride;\
|
|
|
|
|
src+=srcStride;\
|
|
|
|
|
dst += dstStride;\
|
|
|
|
|
src += srcStride;\
|
|
|
|
|
}\
|
|
|
|
|
}\
|
|
|
|
|
\
|
|
|
|
@ -51,7 +51,7 @@ static void OPNAME ## rv40_qpel8_v_lowpass(uint8_t *dst, uint8_t *src, int dstSt |
|
|
|
|
const int w, const int C1, const int C2, const int SHIFT){\
|
|
|
|
|
uint8_t *cm = ff_cropTbl + MAX_NEG_CROP;\
|
|
|
|
|
int i;\
|
|
|
|
|
for(i=0; i<w; i++)\
|
|
|
|
|
for(i = 0; i < w; i++)\
|
|
|
|
|
{\
|
|
|
|
|
const int srcB = src[-2*srcStride];\
|
|
|
|
|
const int srcA = src[-1*srcStride];\
|
|
|
|
@ -65,7 +65,7 @@ static void OPNAME ## rv40_qpel8_v_lowpass(uint8_t *dst, uint8_t *src, int dstSt |
|
|
|
|
const int src7 = src[7 *srcStride];\
|
|
|
|
|
const int src8 = src[8 *srcStride];\
|
|
|
|
|
const int src9 = src[9 *srcStride];\
|
|
|
|
|
const int src10= src[10*srcStride];\
|
|
|
|
|
const int src10 = src[10*srcStride];\
|
|
|
|
|
OP(dst[0*dstStride], (srcB + src3 - 5*(srcA+src2) + src0*C1 + src1*C2 + (1<<(SHIFT-1))) >> SHIFT);\
|
|
|
|
|
OP(dst[1*dstStride], (srcA + src4 - 5*(src0+src3) + src1*C1 + src2*C2 + (1<<(SHIFT-1))) >> SHIFT);\
|
|
|
|
|
OP(dst[2*dstStride], (src0 + src5 - 5*(src1+src4) + src2*C1 + src3*C2 + (1<<(SHIFT-1))) >> SHIFT);\
|
|
|
|
@ -119,21 +119,21 @@ static void OPNAME ## rv40_qpel ## SIZE ## _mc01_c(uint8_t *dst, uint8_t *src, i |
|
|
|
|
\
|
|
|
|
|
static void OPNAME ## rv40_qpel ## SIZE ## _mc11_c(uint8_t *dst, uint8_t *src, int stride){\
|
|
|
|
|
uint8_t full[SIZE*(SIZE+5)];\
|
|
|
|
|
uint8_t * const full_mid= full + SIZE*2;\
|
|
|
|
|
uint8_t * const full_mid = full + SIZE*2;\
|
|
|
|
|
put_rv40_qpel ## SIZE ## _h_lowpass(full, src - 2*stride, SIZE, stride, SIZE+5, 52, 20, 6);\
|
|
|
|
|
OPNAME ## rv40_qpel ## SIZE ## _v_lowpass(dst, full_mid, stride, SIZE, SIZE, 52, 20, 6);\
|
|
|
|
|
}\
|
|
|
|
|
\
|
|
|
|
|
static void OPNAME ## rv40_qpel ## SIZE ## _mc21_c(uint8_t *dst, uint8_t *src, int stride){\
|
|
|
|
|
uint8_t full[SIZE*(SIZE+5)];\
|
|
|
|
|
uint8_t * const full_mid= full + SIZE*2;\
|
|
|
|
|
uint8_t * const full_mid = full + SIZE*2;\
|
|
|
|
|
put_rv40_qpel ## SIZE ## _h_lowpass(full, src - 2*stride, SIZE, stride, SIZE+5, 20, 20, 5);\
|
|
|
|
|
OPNAME ## rv40_qpel ## SIZE ## _v_lowpass(dst, full_mid, stride, SIZE, SIZE, 52, 20, 6);\
|
|
|
|
|
}\
|
|
|
|
|
\
|
|
|
|
|
static void OPNAME ## rv40_qpel ## SIZE ## _mc31_c(uint8_t *dst, uint8_t *src, int stride){\
|
|
|
|
|
uint8_t full[SIZE*(SIZE+5)];\
|
|
|
|
|
uint8_t * const full_mid= full + SIZE*2;\
|
|
|
|
|
uint8_t * const full_mid = full + SIZE*2;\
|
|
|
|
|
put_rv40_qpel ## SIZE ## _h_lowpass(full, src - 2*stride, SIZE, stride, SIZE+5, 20, 52, 6);\
|
|
|
|
|
OPNAME ## rv40_qpel ## SIZE ## _v_lowpass(dst, full_mid, stride, SIZE, SIZE, 52, 20, 6);\
|
|
|
|
|
}\
|
|
|
|
@ -144,21 +144,21 @@ static void OPNAME ## rv40_qpel ## SIZE ## _mc02_c(uint8_t *dst, uint8_t *src, i |
|
|
|
|
\
|
|
|
|
|
static void OPNAME ## rv40_qpel ## SIZE ## _mc12_c(uint8_t *dst, uint8_t *src, int stride){\
|
|
|
|
|
uint8_t full[SIZE*(SIZE+5)];\
|
|
|
|
|
uint8_t * const full_mid= full + SIZE*2;\
|
|
|
|
|
uint8_t * const full_mid = full + SIZE*2;\
|
|
|
|
|
put_rv40_qpel ## SIZE ## _h_lowpass(full, src - 2*stride, SIZE, stride, SIZE+5, 52, 20, 6);\
|
|
|
|
|
OPNAME ## rv40_qpel ## SIZE ## _v_lowpass(dst, full_mid, stride, SIZE, SIZE, 20, 20, 5);\
|
|
|
|
|
}\
|
|
|
|
|
\
|
|
|
|
|
static void OPNAME ## rv40_qpel ## SIZE ## _mc22_c(uint8_t *dst, uint8_t *src, int stride){\
|
|
|
|
|
uint8_t full[SIZE*(SIZE+5)];\
|
|
|
|
|
uint8_t * const full_mid= full + SIZE*2;\
|
|
|
|
|
uint8_t * const full_mid = full + SIZE*2;\
|
|
|
|
|
put_rv40_qpel ## SIZE ## _h_lowpass(full, src - 2*stride, SIZE, stride, SIZE+5, 20, 20, 5);\
|
|
|
|
|
OPNAME ## rv40_qpel ## SIZE ## _v_lowpass(dst, full_mid, stride, SIZE, SIZE, 20, 20, 5);\
|
|
|
|
|
}\
|
|
|
|
|
\
|
|
|
|
|
static void OPNAME ## rv40_qpel ## SIZE ## _mc32_c(uint8_t *dst, uint8_t *src, int stride){\
|
|
|
|
|
uint8_t full[SIZE*(SIZE+5)];\
|
|
|
|
|
uint8_t * const full_mid= full + SIZE*2;\
|
|
|
|
|
uint8_t * const full_mid = full + SIZE*2;\
|
|
|
|
|
put_rv40_qpel ## SIZE ## _h_lowpass(full, src - 2*stride, SIZE, stride, SIZE+5, 20, 52, 6);\
|
|
|
|
|
OPNAME ## rv40_qpel ## SIZE ## _v_lowpass(dst, full_mid, stride, SIZE, SIZE, 20, 20, 5);\
|
|
|
|
|
}\
|
|
|
|
@ -169,14 +169,14 @@ static void OPNAME ## rv40_qpel ## SIZE ## _mc03_c(uint8_t *dst, uint8_t *src, i |
|
|
|
|
\
|
|
|
|
|
static void OPNAME ## rv40_qpel ## SIZE ## _mc13_c(uint8_t *dst, uint8_t *src, int stride){\
|
|
|
|
|
uint8_t full[SIZE*(SIZE+5)];\
|
|
|
|
|
uint8_t * const full_mid= full + SIZE*2;\
|
|
|
|
|
uint8_t * const full_mid = full + SIZE*2;\
|
|
|
|
|
put_rv40_qpel ## SIZE ## _h_lowpass(full, src - 2*stride, SIZE, stride, SIZE+5, 52, 20, 6);\
|
|
|
|
|
OPNAME ## rv40_qpel ## SIZE ## _v_lowpass(dst, full_mid, stride, SIZE, SIZE, 20, 52, 6);\
|
|
|
|
|
}\
|
|
|
|
|
\
|
|
|
|
|
static void OPNAME ## rv40_qpel ## SIZE ## _mc23_c(uint8_t *dst, uint8_t *src, int stride){\
|
|
|
|
|
uint8_t full[SIZE*(SIZE+5)];\
|
|
|
|
|
uint8_t * const full_mid= full + SIZE*2;\
|
|
|
|
|
uint8_t * const full_mid = full + SIZE*2;\
|
|
|
|
|
put_rv40_qpel ## SIZE ## _h_lowpass(full, src - 2*stride, SIZE, stride, SIZE+5, 20, 20, 5);\
|
|
|
|
|
OPNAME ## rv40_qpel ## SIZE ## _v_lowpass(dst, full_mid, stride, SIZE, SIZE, 20, 52, 6);\
|
|
|
|
|
}\
|
|
|
|
@ -205,50 +205,50 @@ static const int rv40_bias[4][4] = { |
|
|
|
|
|
|
|
|
|
#define RV40_CHROMA_MC(OPNAME, OP)\ |
|
|
|
|
static void OPNAME ## rv40_chroma_mc4_c(uint8_t *dst/*align 8*/, uint8_t *src/*align 1*/, int stride, int h, int x, int y){\
|
|
|
|
|
const int A=(8-x)*(8-y);\
|
|
|
|
|
const int B=( x)*(8-y);\
|
|
|
|
|
const int C=(8-x)*( y);\
|
|
|
|
|
const int D=( x)*( y);\
|
|
|
|
|
const int A = (8-x) * (8-y);\
|
|
|
|
|
const int B = ( x) * (8-y);\
|
|
|
|
|
const int C = (8-x) * ( y);\
|
|
|
|
|
const int D = ( x) * ( y);\
|
|
|
|
|
int i;\
|
|
|
|
|
int bias = rv40_bias[y>>1][x>>1];\
|
|
|
|
|
\
|
|
|
|
|
assert(x<8 && y<8 && x>=0 && y>=0);\
|
|
|
|
|
\
|
|
|
|
|
if(D){\
|
|
|
|
|
for(i=0; i<h; i++){\
|
|
|
|
|
for(i = 0; i < h; i++){\
|
|
|
|
|
OP(dst[0], (A*src[0] + B*src[1] + C*src[stride+0] + D*src[stride+1] + bias));\
|
|
|
|
|
OP(dst[1], (A*src[1] + B*src[2] + C*src[stride+1] + D*src[stride+2] + bias));\
|
|
|
|
|
OP(dst[2], (A*src[2] + B*src[3] + C*src[stride+2] + D*src[stride+3] + bias));\
|
|
|
|
|
OP(dst[3], (A*src[3] + B*src[4] + C*src[stride+3] + D*src[stride+4] + bias));\
|
|
|
|
|
dst+= stride;\
|
|
|
|
|
src+= stride;\
|
|
|
|
|
dst += stride;\
|
|
|
|
|
src += stride;\
|
|
|
|
|
}\
|
|
|
|
|
}else{\
|
|
|
|
|
const int E= B+C;\
|
|
|
|
|
const int step= C ? stride : 1;\
|
|
|
|
|
for(i=0; i<h; i++){\
|
|
|
|
|
const int E = B + C;\
|
|
|
|
|
const int step = C ? stride : 1;\
|
|
|
|
|
for(i = 0; i < h; i++){\
|
|
|
|
|
OP(dst[0], (A*src[0] + E*src[step+0] + bias));\
|
|
|
|
|
OP(dst[1], (A*src[1] + E*src[step+1] + bias));\
|
|
|
|
|
OP(dst[2], (A*src[2] + E*src[step+2] + bias));\
|
|
|
|
|
OP(dst[3], (A*src[3] + E*src[step+3] + bias));\
|
|
|
|
|
dst+= stride;\
|
|
|
|
|
src+= stride;\
|
|
|
|
|
dst += stride;\
|
|
|
|
|
src += stride;\
|
|
|
|
|
}\
|
|
|
|
|
}\
|
|
|
|
|
}\
|
|
|
|
|
\
|
|
|
|
|
static void OPNAME ## rv40_chroma_mc8_c(uint8_t *dst/*align 8*/, uint8_t *src/*align 1*/, int stride, int h, int x, int y){\
|
|
|
|
|
const int A=(8-x)*(8-y);\
|
|
|
|
|
const int B=( x)*(8-y);\
|
|
|
|
|
const int C=(8-x)*( y);\
|
|
|
|
|
const int D=( x)*( y);\
|
|
|
|
|
const int A = (8-x) * (8-y);\
|
|
|
|
|
const int B = ( x) * (8-y);\
|
|
|
|
|
const int C = (8-x) * ( y);\
|
|
|
|
|
const int D = ( x) * ( y);\
|
|
|
|
|
int i;\
|
|
|
|
|
int bias = rv40_bias[y>>1][x>>1];\
|
|
|
|
|
\
|
|
|
|
|
assert(x<8 && y<8 && x>=0 && y>=0);\
|
|
|
|
|
\
|
|
|
|
|
if(D){\
|
|
|
|
|
for(i=0; i<h; i++){\
|
|
|
|
|
for(i = 0; i < h; i++){\
|
|
|
|
|
OP(dst[0], (A*src[0] + B*src[1] + C*src[stride+0] + D*src[stride+1] + bias));\
|
|
|
|
|
OP(dst[1], (A*src[1] + B*src[2] + C*src[stride+1] + D*src[stride+2] + bias));\
|
|
|
|
|
OP(dst[2], (A*src[2] + B*src[3] + C*src[stride+2] + D*src[stride+3] + bias));\
|
|
|
|
@ -257,13 +257,13 @@ static void OPNAME ## rv40_chroma_mc8_c(uint8_t *dst/*align 8*/, uint8_t *src/*a |
|
|
|
|
OP(dst[5], (A*src[5] + B*src[6] + C*src[stride+5] + D*src[stride+6] + bias));\
|
|
|
|
|
OP(dst[6], (A*src[6] + B*src[7] + C*src[stride+6] + D*src[stride+7] + bias));\
|
|
|
|
|
OP(dst[7], (A*src[7] + B*src[8] + C*src[stride+7] + D*src[stride+8] + bias));\
|
|
|
|
|
dst+= stride;\
|
|
|
|
|
src+= stride;\
|
|
|
|
|
dst += stride;\
|
|
|
|
|
src += stride;\
|
|
|
|
|
}\
|
|
|
|
|
}else{\
|
|
|
|
|
const int E= B+C;\
|
|
|
|
|
const int step= C ? stride : 1;\
|
|
|
|
|
for(i=0; i<h; i++){\
|
|
|
|
|
const int E = B + C;\
|
|
|
|
|
const int step = C ? stride : 1;\
|
|
|
|
|
for(i = 0; i < h; i++){\
|
|
|
|
|
OP(dst[0], (A*src[0] + E*src[step+0] + bias));\
|
|
|
|
|
OP(dst[1], (A*src[1] + E*src[step+1] + bias));\
|
|
|
|
|
OP(dst[2], (A*src[2] + E*src[step+2] + bias));\
|
|
|
|
@ -272,8 +272,8 @@ static void OPNAME ## rv40_chroma_mc8_c(uint8_t *dst/*align 8*/, uint8_t *src/*a |
|
|
|
|
OP(dst[5], (A*src[5] + E*src[step+5] + bias));\
|
|
|
|
|
OP(dst[6], (A*src[6] + E*src[step+6] + bias));\
|
|
|
|
|
OP(dst[7], (A*src[7] + E*src[step+7] + bias));\
|
|
|
|
|
dst+= stride;\
|
|
|
|
|
src+= stride;\
|
|
|
|
|
dst += stride;\
|
|
|
|
|
src += stride;\
|
|
|
|
|
}\
|
|
|
|
|
}\
|
|
|
|
|
} |
|
|
|
@ -346,8 +346,8 @@ void ff_rv40dsp_init(DSPContext* c, AVCodecContext *avctx) { |
|
|
|
|
c->avg_rv40_qpel_pixels_tab[1][13] = avg_rv40_qpel8_mc13_c; |
|
|
|
|
c->avg_rv40_qpel_pixels_tab[1][14] = avg_rv40_qpel8_mc23_c; |
|
|
|
|
|
|
|
|
|
c->put_rv40_chroma_pixels_tab[0]= put_rv40_chroma_mc8_c; |
|
|
|
|
c->put_rv40_chroma_pixels_tab[1]= put_rv40_chroma_mc4_c; |
|
|
|
|
c->avg_rv40_chroma_pixels_tab[0]= avg_rv40_chroma_mc8_c; |
|
|
|
|
c->avg_rv40_chroma_pixels_tab[1]= avg_rv40_chroma_mc4_c; |
|
|
|
|
c->put_rv40_chroma_pixels_tab[0] = put_rv40_chroma_mc8_c; |
|
|
|
|
c->put_rv40_chroma_pixels_tab[1] = put_rv40_chroma_mc4_c; |
|
|
|
|
c->avg_rv40_chroma_pixels_tab[0] = avg_rv40_chroma_mc8_c; |
|
|
|
|
c->avg_rv40_chroma_pixels_tab[1] = avg_rv40_chroma_mc4_c; |
|
|
|
|
} |
|
|
|
|