[FFmpeg-cvslog] r16151 - in trunk/libavcodec/armv4l: dsputil_neon.c h264idct_neon.S

mru subversion
Mon Dec 15 23:12:54 CET 2008


Author: mru
Date: Mon Dec 15 23:12:54 2008
New Revision: 16151

Log:
ARM: NEON optimised h264_idct_dc_add

Modified:
   trunk/libavcodec/armv4l/dsputil_neon.c
   trunk/libavcodec/armv4l/h264idct_neon.S

Modified: trunk/libavcodec/armv4l/dsputil_neon.c
==============================================================================
--- trunk/libavcodec/armv4l/dsputil_neon.c	(original)
+++ trunk/libavcodec/armv4l/dsputil_neon.c	Mon Dec 15 23:12:54 2008
@@ -93,6 +93,7 @@ void ff_h264_h_loop_filter_chroma_neon(u
                                        int beta, int8_t *tc0);
 
 void ff_h264_idct_add_neon(uint8_t *dst, DCTELEM *block, int stride);
+void ff_h264_idct_dc_add_neon(uint8_t *dst, DCTELEM *block, int stride);
 
 void ff_dsputil_init_neon(DSPContext *c, AVCodecContext *avctx)
 {
@@ -164,4 +165,5 @@ void ff_dsputil_init_neon(DSPContext *c,
     c->h264_h_loop_filter_chroma = ff_h264_h_loop_filter_chroma_neon;
 
     c->h264_idct_add = ff_h264_idct_add_neon;
+    c->h264_idct_dc_add = ff_h264_idct_dc_add_neon;
 }

Modified: trunk/libavcodec/armv4l/h264idct_neon.S
==============================================================================
--- trunk/libavcodec/armv4l/h264idct_neon.S	(original)
+++ trunk/libavcodec/armv4l/h264idct_neon.S	Mon Dec 15 23:12:54 2008
@@ -75,3 +75,22 @@ function ff_h264_idct_add_neon, export=1
 
         bx              lr
         .endfunc
+
+function ff_h264_idct_dc_add_neon, export=1
+        vld1.16         {d2[],d3[]}, [r1,:16]
+        vrshr.s16       q1,  q1,  #6
+        vld1.32         {d0[0]},  [r0,:32], r2
+        vld1.32         {d0[1]},  [r0,:32], r2
+        vaddw.u8        q2,  q1,  d0
+        vld1.32         {d1[0]},  [r0,:32], r2
+        vld1.32         {d1[1]},  [r0,:32], r2
+        vaddw.u8        q1,  q1,  d1
+        vqmovun.s16     d0,  q2
+        vqmovun.s16     d1,  q1
+        sub             r0,  r0,  r2, lsl #2
+        vst1.32         {d0[0]},  [r0,:32], r2
+        vst1.32         {d0[1]},  [r0,:32], r2
+        vst1.32         {d1[0]},  [r0,:32], r2
+        vst1.32         {d1[1]},  [r0,:32], r2
+        bx              lr
+        .endfunc




More information about the ffmpeg-cvslog mailing list