00001
00002
00003
00004
00005
00006
00007
00008
00009
00010
00011
00012
00013
00014
00015
00016
00017
00018
00019
00020
00021
00022 #include "libavutil/cpu.h"
00023 #include "libavutil/x86/asm.h"
00024 #include "libavcodec/dsputil.h"
00025 #include "libavcodec/mpegaudiodsp.h"
00026
00027 void ff_imdct36_float_sse(float *out, float *buf, float *in, float *win);
00028 void ff_imdct36_float_sse2(float *out, float *buf, float *in, float *win);
00029 void ff_imdct36_float_sse3(float *out, float *buf, float *in, float *win);
00030 void ff_imdct36_float_ssse3(float *out, float *buf, float *in, float *win);
00031 void ff_imdct36_float_avx(float *out, float *buf, float *in, float *win);
00032 void ff_four_imdct36_float_sse(float *out, float *buf, float *in, float *win,
00033 float *tmpbuf);
00034 void ff_four_imdct36_float_avx(float *out, float *buf, float *in, float *win,
00035 float *tmpbuf);
00036
00037 DECLARE_ALIGNED(16, static float, mdct_win_sse)[2][4][4*40];
00038
00039 #if HAVE_INLINE_ASM
00040
00041 #define MACS(rt, ra, rb) rt+=(ra)*(rb)
00042 #define MLSS(rt, ra, rb) rt-=(ra)*(rb)
00043
00044 #define SUM8(op, sum, w, p) \
00045 { \
00046 op(sum, (w)[0 * 64], (p)[0 * 64]); \
00047 op(sum, (w)[1 * 64], (p)[1 * 64]); \
00048 op(sum, (w)[2 * 64], (p)[2 * 64]); \
00049 op(sum, (w)[3 * 64], (p)[3 * 64]); \
00050 op(sum, (w)[4 * 64], (p)[4 * 64]); \
00051 op(sum, (w)[5 * 64], (p)[5 * 64]); \
00052 op(sum, (w)[6 * 64], (p)[6 * 64]); \
00053 op(sum, (w)[7 * 64], (p)[7 * 64]); \
00054 }
00055
00056 static void apply_window(const float *buf, const float *win1,
00057 const float *win2, float *sum1, float *sum2, int len)
00058 {
00059 x86_reg count = - 4*len;
00060 const float *win1a = win1+len;
00061 const float *win2a = win2+len;
00062 const float *bufa = buf+len;
00063 float *sum1a = sum1+len;
00064 float *sum2a = sum2+len;
00065
00066
00067 #define MULT(a, b) \
00068 "movaps " #a "(%1,%0), %%xmm1 \n\t" \
00069 "movaps " #a "(%3,%0), %%xmm2 \n\t" \
00070 "mulps %%xmm2, %%xmm1 \n\t" \
00071 "subps %%xmm1, %%xmm0 \n\t" \
00072 "mulps " #b "(%2,%0), %%xmm2 \n\t" \
00073 "subps %%xmm2, %%xmm4 \n\t" \
00074
00075 __asm__ volatile(
00076 "1: \n\t"
00077 "xorps %%xmm0, %%xmm0 \n\t"
00078 "xorps %%xmm4, %%xmm4 \n\t"
00079
00080 MULT( 0, 0)
00081 MULT( 256, 64)
00082 MULT( 512, 128)
00083 MULT( 768, 192)
00084 MULT(1024, 256)
00085 MULT(1280, 320)
00086 MULT(1536, 384)
00087 MULT(1792, 448)
00088
00089 "movaps %%xmm0, (%4,%0) \n\t"
00090 "movaps %%xmm4, (%5,%0) \n\t"
00091 "add $16, %0 \n\t"
00092 "jl 1b \n\t"
00093 :"+&r"(count)
00094 :"r"(win1a), "r"(win2a), "r"(bufa), "r"(sum1a), "r"(sum2a)
00095 );
00096
00097 #undef MULT
00098 }
00099
00100 static void apply_window_mp3(float *in, float *win, int *unused, float *out,
00101 int incr)
00102 {
00103 LOCAL_ALIGNED_16(float, suma, [17]);
00104 LOCAL_ALIGNED_16(float, sumb, [17]);
00105 LOCAL_ALIGNED_16(float, sumc, [17]);
00106 LOCAL_ALIGNED_16(float, sumd, [17]);
00107
00108 float sum;
00109
00110
00111 __asm__ volatile(
00112 "movaps 0(%0), %%xmm0 \n\t" \
00113 "movaps 16(%0), %%xmm1 \n\t" \
00114 "movaps 32(%0), %%xmm2 \n\t" \
00115 "movaps 48(%0), %%xmm3 \n\t" \
00116 "movaps %%xmm0, 0(%1) \n\t" \
00117 "movaps %%xmm1, 16(%1) \n\t" \
00118 "movaps %%xmm2, 32(%1) \n\t" \
00119 "movaps %%xmm3, 48(%1) \n\t" \
00120 "movaps 64(%0), %%xmm0 \n\t" \
00121 "movaps 80(%0), %%xmm1 \n\t" \
00122 "movaps 96(%0), %%xmm2 \n\t" \
00123 "movaps 112(%0), %%xmm3 \n\t" \
00124 "movaps %%xmm0, 64(%1) \n\t" \
00125 "movaps %%xmm1, 80(%1) \n\t" \
00126 "movaps %%xmm2, 96(%1) \n\t" \
00127 "movaps %%xmm3, 112(%1) \n\t"
00128 ::"r"(in), "r"(in+512)
00129 :"memory"
00130 );
00131
00132 apply_window(in + 16, win , win + 512, suma, sumc, 16);
00133 apply_window(in + 32, win + 48, win + 640, sumb, sumd, 16);
00134
00135 SUM8(MACS, suma[0], win + 32, in + 48);
00136
00137 sumc[ 0] = 0;
00138 sumb[16] = 0;
00139 sumd[16] = 0;
00140
00141 #define SUMS(suma, sumb, sumc, sumd, out1, out2) \
00142 "movups " #sumd "(%4), %%xmm0 \n\t" \
00143 "shufps $0x1b, %%xmm0, %%xmm0 \n\t" \
00144 "subps " #suma "(%1), %%xmm0 \n\t" \
00145 "movaps %%xmm0," #out1 "(%0) \n\t" \
00146 \
00147 "movups " #sumc "(%3), %%xmm0 \n\t" \
00148 "shufps $0x1b, %%xmm0, %%xmm0 \n\t" \
00149 "addps " #sumb "(%2), %%xmm0 \n\t" \
00150 "movaps %%xmm0," #out2 "(%0) \n\t"
00151
00152 if (incr == 1) {
00153 __asm__ volatile(
00154 SUMS( 0, 48, 4, 52, 0, 112)
00155 SUMS(16, 32, 20, 36, 16, 96)
00156 SUMS(32, 16, 36, 20, 32, 80)
00157 SUMS(48, 0, 52, 4, 48, 64)
00158
00159 :"+&r"(out)
00160 :"r"(&suma[0]), "r"(&sumb[0]), "r"(&sumc[0]), "r"(&sumd[0])
00161 :"memory"
00162 );
00163 out += 16*incr;
00164 } else {
00165 int j;
00166 float *out2 = out + 32 * incr;
00167 out[0 ] = -suma[ 0];
00168 out += incr;
00169 out2 -= incr;
00170 for(j=1;j<16;j++) {
00171 *out = -suma[ j] + sumd[16-j];
00172 *out2 = sumb[16-j] + sumc[ j];
00173 out += incr;
00174 out2 -= incr;
00175 }
00176 }
00177
00178 sum = 0;
00179 SUM8(MLSS, sum, win + 16 + 32, in + 32);
00180 *out = sum;
00181 }
00182
00183 #endif
00184
00185 #define DECL_IMDCT_BLOCKS(CPU1, CPU2) \
00186 static void imdct36_blocks_ ## CPU1(float *out, float *buf, float *in, \
00187 int count, int switch_point, int block_type) \
00188 { \
00189 int align_end = count - (count & 3); \
00190 int j; \
00191 for (j = 0; j < align_end; j+= 4) { \
00192 LOCAL_ALIGNED_16(float, tmpbuf, [1024]); \
00193 float *win = mdct_win_sse[switch_point && j < 4][block_type]; \
00194 \
00195 \
00196 \
00197 ff_four_imdct36_float_ ## CPU2(out, buf, in, win, tmpbuf); \
00198 in += 4*18; \
00199 buf += 4*18; \
00200 out += 4; \
00201 } \
00202 for (; j < count; j++) { \
00203 \
00204 \
00205 \
00206 int win_idx = (switch_point && j < 2) ? 0 : block_type; \
00207 float *win = ff_mdct_win_float[win_idx + (4 & -(j & 1))]; \
00208 \
00209 ff_imdct36_float_ ## CPU1(out, buf, in, win); \
00210 \
00211 in += 18; \
00212 buf++; \
00213 out++; \
00214 } \
00215 }
00216
00217 #if HAVE_YASM
00218 #if HAVE_SSE
00219 DECL_IMDCT_BLOCKS(sse,sse)
00220 DECL_IMDCT_BLOCKS(sse2,sse)
00221 DECL_IMDCT_BLOCKS(sse3,sse)
00222 DECL_IMDCT_BLOCKS(ssse3,sse)
00223 #endif
00224 #if HAVE_AVX
00225 DECL_IMDCT_BLOCKS(avx,avx)
00226 #endif
00227 #endif
00228
00229 void ff_mpadsp_init_mmx(MPADSPContext *s)
00230 {
00231 int mm_flags = av_get_cpu_flags();
00232
00233 int i, j;
00234 for (j = 0; j < 4; j++) {
00235 for (i = 0; i < 40; i ++) {
00236 mdct_win_sse[0][j][4*i ] = ff_mdct_win_float[j ][i];
00237 mdct_win_sse[0][j][4*i + 1] = ff_mdct_win_float[j + 4][i];
00238 mdct_win_sse[0][j][4*i + 2] = ff_mdct_win_float[j ][i];
00239 mdct_win_sse[0][j][4*i + 3] = ff_mdct_win_float[j + 4][i];
00240 mdct_win_sse[1][j][4*i ] = ff_mdct_win_float[0 ][i];
00241 mdct_win_sse[1][j][4*i + 1] = ff_mdct_win_float[4 ][i];
00242 mdct_win_sse[1][j][4*i + 2] = ff_mdct_win_float[j ][i];
00243 mdct_win_sse[1][j][4*i + 3] = ff_mdct_win_float[j + 4][i];
00244 }
00245 }
00246
00247 #if HAVE_INLINE_ASM
00248 if (mm_flags & AV_CPU_FLAG_SSE2) {
00249 s->apply_window_float = apply_window_mp3;
00250 }
00251 #endif
00252 #if HAVE_YASM
00253 if (0) {
00254 #if HAVE_AVX
00255 } else if (mm_flags & AV_CPU_FLAG_AVX && HAVE_AVX) {
00256 s->imdct36_blocks_float = imdct36_blocks_avx;
00257 #endif
00258 #if HAVE_SSE
00259 } else if (mm_flags & AV_CPU_FLAG_SSSE3) {
00260 s->imdct36_blocks_float = imdct36_blocks_ssse3;
00261 } else if (mm_flags & AV_CPU_FLAG_SSE3) {
00262 s->imdct36_blocks_float = imdct36_blocks_sse3;
00263 } else if (mm_flags & AV_CPU_FLAG_SSE2) {
00264 s->imdct36_blocks_float = imdct36_blocks_sse2;
00265 } else if (mm_flags & AV_CPU_FLAG_SSE) {
00266 s->imdct36_blocks_float = imdct36_blocks_sse;
00267 #endif
00268 }
00269 #endif
00270 }