00001
00002
00003
00004
00005
00006
00007
00008
00009
00010
00011
00012
00013
00014
00015
00016
00017
00018
00019
00020
00021 #include "libavutil/cpu.h"
00022 #include "libavutil/x86/asm.h"
00023 #include "libavcodec/h264dsp.h"
00024 #include "dsputil_mmx.h"
00025
00026
00027
00028 #define IDCT_ADD_FUNC(NUM, DEPTH, OPT) \
00029 void ff_h264_idct ## NUM ## _add_ ## DEPTH ## _ ## OPT(uint8_t *dst, \
00030 int16_t *block, \
00031 int stride);
00032
00033 IDCT_ADD_FUNC(, 8, mmx)
00034 IDCT_ADD_FUNC(, 10, sse2)
00035 IDCT_ADD_FUNC(_dc, 8, mmx2)
00036 IDCT_ADD_FUNC(_dc, 10, mmx2)
00037 IDCT_ADD_FUNC(8_dc, 8, mmx2)
00038 IDCT_ADD_FUNC(8_dc, 10, sse2)
00039 IDCT_ADD_FUNC(8, 8, mmx)
00040 IDCT_ADD_FUNC(8, 8, sse2)
00041 IDCT_ADD_FUNC(8, 10, sse2)
00042 #if HAVE_AVX
00043 IDCT_ADD_FUNC(, 10, avx)
00044 IDCT_ADD_FUNC(8_dc, 10, avx)
00045 IDCT_ADD_FUNC(8, 10, avx)
00046 #endif
00047
00048
00049 #define IDCT_ADD_REP_FUNC(NUM, REP, DEPTH, OPT) \
00050 void ff_h264_idct ## NUM ## _add ## REP ## _ ## DEPTH ## _ ## OPT \
00051 (uint8_t *dst, const int *block_offset, \
00052 DCTELEM *block, int stride, const uint8_t nnzc[6 * 8]);
00053
00054 IDCT_ADD_REP_FUNC(8, 4, 8, mmx)
00055 IDCT_ADD_REP_FUNC(8, 4, 8, mmx2)
00056 IDCT_ADD_REP_FUNC(8, 4, 8, sse2)
00057 IDCT_ADD_REP_FUNC(8, 4, 10, sse2)
00058 IDCT_ADD_REP_FUNC(8, 4, 10, avx)
00059 IDCT_ADD_REP_FUNC(, 16, 8, mmx)
00060 IDCT_ADD_REP_FUNC(, 16, 8, mmx2)
00061 IDCT_ADD_REP_FUNC(, 16, 8, sse2)
00062 IDCT_ADD_REP_FUNC(, 16, 10, sse2)
00063 IDCT_ADD_REP_FUNC(, 16intra, 8, mmx)
00064 IDCT_ADD_REP_FUNC(, 16intra, 8, mmx2)
00065 IDCT_ADD_REP_FUNC(, 16intra, 8, sse2)
00066 IDCT_ADD_REP_FUNC(, 16intra, 10, sse2)
00067 #if HAVE_AVX
00068 IDCT_ADD_REP_FUNC(, 16, 10, avx)
00069 IDCT_ADD_REP_FUNC(, 16intra, 10, avx)
00070 #endif
00071
00072
00073 #define IDCT_ADD_REP_FUNC2(NUM, REP, DEPTH, OPT) \
00074 void ff_h264_idct ## NUM ## _add ## REP ## _ ## DEPTH ## _ ## OPT \
00075 (uint8_t **dst, const int *block_offset, \
00076 DCTELEM *block, int stride, const uint8_t nnzc[6 * 8]);
00077
00078 IDCT_ADD_REP_FUNC2(, 8, 8, mmx)
00079 IDCT_ADD_REP_FUNC2(, 8, 8, mmx2)
00080 IDCT_ADD_REP_FUNC2(, 8, 8, sse2)
00081 IDCT_ADD_REP_FUNC2(, 8, 10, sse2)
00082 #if HAVE_AVX
00083 IDCT_ADD_REP_FUNC2(, 8, 10, avx)
00084 #endif
00085
00086 void ff_h264_luma_dc_dequant_idct_mmx(DCTELEM *output, DCTELEM *input, int qmul);
00087 void ff_h264_luma_dc_dequant_idct_sse2(DCTELEM *output, DCTELEM *input, int qmul);
00088
00089
00090
00091
00092 void ff_h264_loop_filter_strength_mmx2(int16_t bS[2][4][4], uint8_t nnz[40],
00093 int8_t ref[2][40], int16_t mv[2][40][2],
00094 int bidir, int edges, int step,
00095 int mask_mv0, int mask_mv1, int field);
00096
00097 #define LF_FUNC(DIR, TYPE, DEPTH, OPT) \
00098 void ff_deblock_ ## DIR ## _ ## TYPE ## _ ## DEPTH ## _ ## OPT(uint8_t *pix, \
00099 int stride, \
00100 int alpha, \
00101 int beta, \
00102 int8_t *tc0);
00103 #define LF_IFUNC(DIR, TYPE, DEPTH, OPT) \
00104 void ff_deblock_ ## DIR ## _ ## TYPE ## _ ## DEPTH ## _ ## OPT(uint8_t *pix, \
00105 int stride, \
00106 int alpha, \
00107 int beta);
00108
00109 #define LF_FUNCS(type, depth) \
00110 LF_FUNC(h, chroma, depth, mmx2) \
00111 LF_IFUNC(h, chroma_intra, depth, mmx2) \
00112 LF_FUNC(v, chroma, depth, mmx2) \
00113 LF_IFUNC(v, chroma_intra, depth, mmx2) \
00114 LF_FUNC(h, luma, depth, mmx2) \
00115 LF_IFUNC(h, luma_intra, depth, mmx2) \
00116 LF_FUNC(h, luma, depth, sse2) \
00117 LF_IFUNC(h, luma_intra, depth, sse2) \
00118 LF_FUNC(v, luma, depth, sse2) \
00119 LF_IFUNC(v, luma_intra, depth, sse2) \
00120 LF_FUNC(h, chroma, depth, sse2) \
00121 LF_IFUNC(h, chroma_intra, depth, sse2) \
00122 LF_FUNC(v, chroma, depth, sse2) \
00123 LF_IFUNC(v, chroma_intra, depth, sse2) \
00124 LF_FUNC(h, luma, depth, avx) \
00125 LF_IFUNC(h, luma_intra, depth, avx) \
00126 LF_FUNC(v, luma, depth, avx) \
00127 LF_IFUNC(v, luma_intra, depth, avx) \
00128 LF_FUNC(h, chroma, depth, avx) \
00129 LF_IFUNC(h, chroma_intra, depth, avx) \
00130 LF_FUNC(v, chroma, depth, avx) \
00131 LF_IFUNC(v, chroma_intra, depth, avx)
00132
00133 LF_FUNCS(uint8_t, 8)
00134 LF_FUNCS(uint16_t, 10)
00135
00136 #if ARCH_X86_32 && HAVE_YASM
00137 LF_FUNC(v8, luma, 8, mmx2)
00138 static void ff_deblock_v_luma_8_mmx2(uint8_t *pix, int stride, int alpha,
00139 int beta, int8_t *tc0)
00140 {
00141 if ((tc0[0] & tc0[1]) >= 0)
00142 ff_deblock_v8_luma_8_mmx2(pix + 0, stride, alpha, beta, tc0);
00143 if ((tc0[2] & tc0[3]) >= 0)
00144 ff_deblock_v8_luma_8_mmx2(pix + 8, stride, alpha, beta, tc0 + 2);
00145 }
00146
00147 LF_IFUNC(v8, luma_intra, 8, mmx2)
00148 static void ff_deblock_v_luma_intra_8_mmx2(uint8_t *pix, int stride,
00149 int alpha, int beta)
00150 {
00151 ff_deblock_v8_luma_intra_8_mmx2(pix + 0, stride, alpha, beta);
00152 ff_deblock_v8_luma_intra_8_mmx2(pix + 8, stride, alpha, beta);
00153 }
00154 #endif
00155
00156 LF_FUNC(v, luma, 10, mmx2)
00157 LF_IFUNC(v, luma_intra, 10, mmx2)
00158
00159
00160
00161
00162 #define H264_WEIGHT(W, OPT) \
00163 void ff_h264_weight_ ## W ## _ ## OPT(uint8_t *dst, int stride, \
00164 int height, int log2_denom, \
00165 int weight, int offset);
00166
00167 #define H264_BIWEIGHT(W, OPT) \
00168 void ff_h264_biweight_ ## W ## _ ## OPT(uint8_t *dst, uint8_t *src, \
00169 int stride, int height, \
00170 int log2_denom, int weightd, \
00171 int weights, int offset);
00172
00173 #define H264_BIWEIGHT_MMX(W) \
00174 H264_WEIGHT(W, mmx2) \
00175 H264_BIWEIGHT(W, mmx2)
00176
00177 #define H264_BIWEIGHT_MMX_SSE(W) \
00178 H264_BIWEIGHT_MMX(W) \
00179 H264_WEIGHT(W, sse2) \
00180 H264_BIWEIGHT(W, sse2) \
00181 H264_BIWEIGHT(W, ssse3)
00182
00183 H264_BIWEIGHT_MMX_SSE(16)
00184 H264_BIWEIGHT_MMX_SSE(8)
00185 H264_BIWEIGHT_MMX(4)
00186
00187 #define H264_WEIGHT_10(W, DEPTH, OPT) \
00188 void ff_h264_weight_ ## W ## _ ## DEPTH ## _ ## OPT(uint8_t *dst, \
00189 int stride, \
00190 int height, \
00191 int log2_denom, \
00192 int weight, \
00193 int offset);
00194
00195 #define H264_BIWEIGHT_10(W, DEPTH, OPT) \
00196 void ff_h264_biweight_ ## W ## _ ## DEPTH ## _ ## OPT(uint8_t *dst, \
00197 uint8_t *src, \
00198 int stride, \
00199 int height, \
00200 int log2_denom, \
00201 int weightd, \
00202 int weights, \
00203 int offset);
00204
00205 #define H264_BIWEIGHT_10_SSE(W, DEPTH) \
00206 H264_WEIGHT_10(W, DEPTH, sse2) \
00207 H264_WEIGHT_10(W, DEPTH, sse4) \
00208 H264_BIWEIGHT_10(W, DEPTH, sse2) \
00209 H264_BIWEIGHT_10(W, DEPTH, sse4)
00210
00211 H264_BIWEIGHT_10_SSE(16, 10)
00212 H264_BIWEIGHT_10_SSE(8, 10)
00213 H264_BIWEIGHT_10_SSE(4, 10)
00214
00215 void ff_h264dsp_init_x86(H264DSPContext *c, const int bit_depth,
00216 const int chroma_format_idc)
00217 {
00218 #if HAVE_YASM
00219 int mm_flags = av_get_cpu_flags();
00220
00221 if (chroma_format_idc == 1 && mm_flags & AV_CPU_FLAG_MMXEXT)
00222 c->h264_loop_filter_strength = ff_h264_loop_filter_strength_mmx2;
00223
00224 if (bit_depth == 8) {
00225 if (mm_flags & AV_CPU_FLAG_MMX) {
00226 c->h264_idct_dc_add =
00227 c->h264_idct_add = ff_h264_idct_add_8_mmx;
00228 c->h264_idct8_dc_add =
00229 c->h264_idct8_add = ff_h264_idct8_add_8_mmx;
00230
00231 c->h264_idct_add16 = ff_h264_idct_add16_8_mmx;
00232 c->h264_idct8_add4 = ff_h264_idct8_add4_8_mmx;
00233 if (chroma_format_idc == 1)
00234 c->h264_idct_add8 = ff_h264_idct_add8_8_mmx;
00235 c->h264_idct_add16intra = ff_h264_idct_add16intra_8_mmx;
00236 if (mm_flags & AV_CPU_FLAG_CMOV)
00237 c->h264_luma_dc_dequant_idct = ff_h264_luma_dc_dequant_idct_mmx;
00238
00239 if (mm_flags & AV_CPU_FLAG_MMXEXT) {
00240 c->h264_idct_dc_add = ff_h264_idct_dc_add_8_mmx2;
00241 c->h264_idct8_dc_add = ff_h264_idct8_dc_add_8_mmx2;
00242 c->h264_idct_add16 = ff_h264_idct_add16_8_mmx2;
00243 c->h264_idct8_add4 = ff_h264_idct8_add4_8_mmx2;
00244 if (chroma_format_idc == 1)
00245 c->h264_idct_add8 = ff_h264_idct_add8_8_mmx2;
00246 c->h264_idct_add16intra = ff_h264_idct_add16intra_8_mmx2;
00247
00248 c->h264_v_loop_filter_chroma = ff_deblock_v_chroma_8_mmx2;
00249 c->h264_v_loop_filter_chroma_intra = ff_deblock_v_chroma_intra_8_mmx2;
00250 if (chroma_format_idc == 1) {
00251 c->h264_h_loop_filter_chroma = ff_deblock_h_chroma_8_mmx2;
00252 c->h264_h_loop_filter_chroma_intra = ff_deblock_h_chroma_intra_8_mmx2;
00253 }
00254 #if ARCH_X86_32
00255 c->h264_v_loop_filter_luma = ff_deblock_v_luma_8_mmx2;
00256 c->h264_h_loop_filter_luma = ff_deblock_h_luma_8_mmx2;
00257 c->h264_v_loop_filter_luma_intra = ff_deblock_v_luma_intra_8_mmx2;
00258 c->h264_h_loop_filter_luma_intra = ff_deblock_h_luma_intra_8_mmx2;
00259 #endif
00260 c->weight_h264_pixels_tab[0] = ff_h264_weight_16_mmx2;
00261 c->weight_h264_pixels_tab[1] = ff_h264_weight_8_mmx2;
00262 c->weight_h264_pixels_tab[2] = ff_h264_weight_4_mmx2;
00263
00264 c->biweight_h264_pixels_tab[0] = ff_h264_biweight_16_mmx2;
00265 c->biweight_h264_pixels_tab[1] = ff_h264_biweight_8_mmx2;
00266 c->biweight_h264_pixels_tab[2] = ff_h264_biweight_4_mmx2;
00267
00268 if (mm_flags & AV_CPU_FLAG_SSE2) {
00269 c->h264_idct8_add = ff_h264_idct8_add_8_sse2;
00270
00271 c->h264_idct_add16 = ff_h264_idct_add16_8_sse2;
00272 c->h264_idct8_add4 = ff_h264_idct8_add4_8_sse2;
00273 if (chroma_format_idc == 1)
00274 c->h264_idct_add8 = ff_h264_idct_add8_8_sse2;
00275 c->h264_idct_add16intra = ff_h264_idct_add16intra_8_sse2;
00276 c->h264_luma_dc_dequant_idct = ff_h264_luma_dc_dequant_idct_sse2;
00277
00278 c->weight_h264_pixels_tab[0] = ff_h264_weight_16_sse2;
00279 c->weight_h264_pixels_tab[1] = ff_h264_weight_8_sse2;
00280
00281 c->biweight_h264_pixels_tab[0] = ff_h264_biweight_16_sse2;
00282 c->biweight_h264_pixels_tab[1] = ff_h264_biweight_8_sse2;
00283
00284 #if HAVE_ALIGNED_STACK
00285 c->h264_v_loop_filter_luma = ff_deblock_v_luma_8_sse2;
00286 c->h264_h_loop_filter_luma = ff_deblock_h_luma_8_sse2;
00287 c->h264_v_loop_filter_luma_intra = ff_deblock_v_luma_intra_8_sse2;
00288 c->h264_h_loop_filter_luma_intra = ff_deblock_h_luma_intra_8_sse2;
00289 #endif
00290 }
00291 if (mm_flags & AV_CPU_FLAG_SSSE3) {
00292 c->biweight_h264_pixels_tab[0] = ff_h264_biweight_16_ssse3;
00293 c->biweight_h264_pixels_tab[1] = ff_h264_biweight_8_ssse3;
00294 }
00295 if (HAVE_AVX && mm_flags & AV_CPU_FLAG_AVX) {
00296 #if HAVE_ALIGNED_STACK
00297 c->h264_v_loop_filter_luma = ff_deblock_v_luma_8_avx;
00298 c->h264_h_loop_filter_luma = ff_deblock_h_luma_8_avx;
00299 c->h264_v_loop_filter_luma_intra = ff_deblock_v_luma_intra_8_avx;
00300 c->h264_h_loop_filter_luma_intra = ff_deblock_h_luma_intra_8_avx;
00301 #endif
00302 }
00303 }
00304 }
00305 } else if (bit_depth == 10) {
00306 if (mm_flags & AV_CPU_FLAG_MMX) {
00307 if (mm_flags & AV_CPU_FLAG_MMXEXT) {
00308 #if ARCH_X86_32
00309 c->h264_v_loop_filter_chroma = ff_deblock_v_chroma_10_mmx2;
00310 c->h264_v_loop_filter_chroma_intra = ff_deblock_v_chroma_intra_10_mmx2;
00311 c->h264_v_loop_filter_luma = ff_deblock_v_luma_10_mmx2;
00312 c->h264_h_loop_filter_luma = ff_deblock_h_luma_10_mmx2;
00313 c->h264_v_loop_filter_luma_intra = ff_deblock_v_luma_intra_10_mmx2;
00314 c->h264_h_loop_filter_luma_intra = ff_deblock_h_luma_intra_10_mmx2;
00315 #endif
00316 c->h264_idct_dc_add = ff_h264_idct_dc_add_10_mmx2;
00317 if (mm_flags & AV_CPU_FLAG_SSE2) {
00318 c->h264_idct_add = ff_h264_idct_add_10_sse2;
00319 c->h264_idct8_dc_add = ff_h264_idct8_dc_add_10_sse2;
00320
00321 c->h264_idct_add16 = ff_h264_idct_add16_10_sse2;
00322 if (chroma_format_idc == 1)
00323 c->h264_idct_add8 = ff_h264_idct_add8_10_sse2;
00324 c->h264_idct_add16intra = ff_h264_idct_add16intra_10_sse2;
00325 #if HAVE_ALIGNED_STACK
00326 c->h264_idct8_add = ff_h264_idct8_add_10_sse2;
00327 c->h264_idct8_add4 = ff_h264_idct8_add4_10_sse2;
00328 #endif
00329
00330 c->weight_h264_pixels_tab[0] = ff_h264_weight_16_10_sse2;
00331 c->weight_h264_pixels_tab[1] = ff_h264_weight_8_10_sse2;
00332 c->weight_h264_pixels_tab[2] = ff_h264_weight_4_10_sse2;
00333
00334 c->biweight_h264_pixels_tab[0] = ff_h264_biweight_16_10_sse2;
00335 c->biweight_h264_pixels_tab[1] = ff_h264_biweight_8_10_sse2;
00336 c->biweight_h264_pixels_tab[2] = ff_h264_biweight_4_10_sse2;
00337
00338 c->h264_v_loop_filter_chroma = ff_deblock_v_chroma_10_sse2;
00339 c->h264_v_loop_filter_chroma_intra = ff_deblock_v_chroma_intra_10_sse2;
00340 #if HAVE_ALIGNED_STACK
00341 c->h264_v_loop_filter_luma = ff_deblock_v_luma_10_sse2;
00342 c->h264_h_loop_filter_luma = ff_deblock_h_luma_10_sse2;
00343 c->h264_v_loop_filter_luma_intra = ff_deblock_v_luma_intra_10_sse2;
00344 c->h264_h_loop_filter_luma_intra = ff_deblock_h_luma_intra_10_sse2;
00345 #endif
00346 }
00347 if (mm_flags & AV_CPU_FLAG_SSE4) {
00348 c->weight_h264_pixels_tab[0] = ff_h264_weight_16_10_sse4;
00349 c->weight_h264_pixels_tab[1] = ff_h264_weight_8_10_sse4;
00350 c->weight_h264_pixels_tab[2] = ff_h264_weight_4_10_sse4;
00351
00352 c->biweight_h264_pixels_tab[0] = ff_h264_biweight_16_10_sse4;
00353 c->biweight_h264_pixels_tab[1] = ff_h264_biweight_8_10_sse4;
00354 c->biweight_h264_pixels_tab[2] = ff_h264_biweight_4_10_sse4;
00355 }
00356 #if HAVE_AVX
00357 if (mm_flags & AV_CPU_FLAG_AVX) {
00358 c->h264_idct_dc_add =
00359 c->h264_idct_add = ff_h264_idct_add_10_avx;
00360 c->h264_idct8_dc_add = ff_h264_idct8_dc_add_10_avx;
00361
00362 c->h264_idct_add16 = ff_h264_idct_add16_10_avx;
00363 if (chroma_format_idc == 1)
00364 c->h264_idct_add8 = ff_h264_idct_add8_10_avx;
00365 c->h264_idct_add16intra = ff_h264_idct_add16intra_10_avx;
00366 #if HAVE_ALIGNED_STACK
00367 c->h264_idct8_add = ff_h264_idct8_add_10_avx;
00368 c->h264_idct8_add4 = ff_h264_idct8_add4_10_avx;
00369 #endif
00370
00371 c->h264_v_loop_filter_chroma = ff_deblock_v_chroma_10_avx;
00372 c->h264_v_loop_filter_chroma_intra = ff_deblock_v_chroma_intra_10_avx;
00373 #if HAVE_ALIGNED_STACK
00374 c->h264_v_loop_filter_luma = ff_deblock_v_luma_10_avx;
00375 c->h264_h_loop_filter_luma = ff_deblock_h_luma_10_avx;
00376 c->h264_v_loop_filter_luma_intra = ff_deblock_v_luma_intra_10_avx;
00377 c->h264_h_loop_filter_luma_intra = ff_deblock_h_luma_intra_10_avx;
00378 #endif
00379 }
00380 #endif
00381 }
00382 }
00383 }
00384 #endif
00385 }