Mercurial > libavcodec.hg
changeset 8340:834a77844ba3 libavcodec
ARM: NEON optimised h264_idct_dc_add
author | mru |
---|---|
date | Mon, 15 Dec 2008 22:12:54 +0000 |
parents | a561ec6d1bf6 |
children | 750453ac8c90 |
files | armv4l/dsputil_neon.c armv4l/h264idct_neon.S |
diffstat | 2 files changed, 21 insertions(+), 0 deletions(-) [+] |
line wrap: on
line diff
--- a/armv4l/dsputil_neon.c Mon Dec 15 22:12:51 2008 +0000 +++ b/armv4l/dsputil_neon.c Mon Dec 15 22:12:54 2008 +0000 @@ -93,6 +93,7 @@ int beta, int8_t *tc0); void ff_h264_idct_add_neon(uint8_t *dst, DCTELEM *block, int stride); +void ff_h264_idct_dc_add_neon(uint8_t *dst, DCTELEM *block, int stride); void ff_dsputil_init_neon(DSPContext *c, AVCodecContext *avctx) { @@ -164,4 +165,5 @@ c->h264_h_loop_filter_chroma = ff_h264_h_loop_filter_chroma_neon; c->h264_idct_add = ff_h264_idct_add_neon; + c->h264_idct_dc_add = ff_h264_idct_dc_add_neon; }
--- a/armv4l/h264idct_neon.S Mon Dec 15 22:12:51 2008 +0000 +++ b/armv4l/h264idct_neon.S Mon Dec 15 22:12:54 2008 +0000 @@ -75,3 +75,22 @@ bx lr .endfunc + +function ff_h264_idct_dc_add_neon, export=1 + vld1.16 {d2[],d3[]}, [r1,:16] + vrshr.s16 q1, q1, #6 + vld1.32 {d0[0]}, [r0,:32], r2 + vld1.32 {d0[1]}, [r0,:32], r2 + vaddw.u8 q2, q1, d0 + vld1.32 {d1[0]}, [r0,:32], r2 + vld1.32 {d1[1]}, [r0,:32], r2 + vaddw.u8 q1, q1, d1 + vqmovun.s16 d0, q2 + vqmovun.s16 d1, q1 + sub r0, r0, r2, lsl #2 + vst1.32 {d0[0]}, [r0,:32], r2 + vst1.32 {d0[1]}, [r0,:32], r2 + vst1.32 {d1[0]}, [r0,:32], r2 + vst1.32 {d1[1]}, [r0,:32], r2 + bx lr + .endfunc