libavcodec.hg: x86/dsputil

comparison x86/dsputil_mmx.c @ 10104:0fa3d21b317e libavcodec

SSE optimized vector_clipf(). 10% faster TwinVQ decoding.

author	vitor
date	Thu, 27 Aug 2009 14:49:36 +0000
parents	3141f69e3905
children	7775f6627612

comparison

equal deleted inserted replaced

-:2066cbe806ef
+:0fa3d21b317e
 :"+r"(i)
 :"r"(dst+len), "r"(src+len), "m"(mul)
 );
 }
+static void vector_clipf_sse(float *dst, float *src, float min, float max,
+int len)
+{
+x86_reg i = (len-16)*4;
+__asm__ volatile(
+"movss  %3, %%xmm4 \n"
+"movss  %4, %%xmm5 \n"
+"shufps $0, %%xmm4, %%xmm4 \n"
+"shufps $0, %%xmm5, %%xmm5 \n"
+"1: \n\t"
+"movaps    (%2,%0), %%xmm0 \n\t" // 3/1 on intel
+"movaps  16(%2,%0), %%xmm1 \n\t"
+"movaps  32(%2,%0), %%xmm2 \n\t"
+"movaps  48(%2,%0), %%xmm3 \n\t"
+"maxps      %%xmm4, %%xmm0 \n\t"
+"maxps      %%xmm4, %%xmm1 \n\t"
+"maxps      %%xmm4, %%xmm2 \n\t"
+"maxps      %%xmm4, %%xmm3 \n\t"
+"minps      %%xmm5, %%xmm0 \n\t"
+"minps      %%xmm5, %%xmm1 \n\t"
+"minps      %%xmm5, %%xmm2 \n\t"
+"minps      %%xmm5, %%xmm3 \n\t"
+"movaps  %%xmm0,   (%1,%0) \n\t"
+"movaps  %%xmm1, 16(%1,%0) \n\t"
+"movaps  %%xmm2, 32(%1,%0) \n\t"
+"movaps  %%xmm3, 48(%1,%0) \n\t"
+"sub  $64, %0 \n\t"
+"jge 1b \n\t"
+:"+r"(i)
+:"r"(dst), "r"(src), "m"(min), "m"(max)
+:"memory"
+);
+}
 static void float_to_int16_3dnow(int16_t *dst, const float *src, long len){
 x86_reg reglen = len;
 // not bit-exact: pf2id uses different rounding than C and SSE
 __asm__ volatile(
 "add        %0          , %0        \n\t"
 c->vector_fmul = vector_fmul_sse;
 c->vector_fmul_reverse = vector_fmul_reverse_sse;
 c->vector_fmul_add_add = vector_fmul_add_add_sse;
 c->vector_fmul_window = vector_fmul_window_sse;
 c->int32_to_float_fmul_scalar = int32_to_float_fmul_scalar_sse;
+c->vector_clipf = vector_clipf_sse;
 c->float_to_int16 = float_to_int16_sse;
 c->float_to_int16_interleave = float_to_int16_interleave_sse;
 }
 if(mm_flags & FF_MM_3DNOW)
 c->vector_fmul_add_add = vector_fmul_add_add_3dnow; // faster than sse

Mercurial > libavcodec.hg

comparison x86/dsputil_mmx.c @ 10104:0fa3d21b317e libavcodec