ARM: NEON optimised h264_idct_dc_add
Originally committed as revision 16151 to svn://svn.ffmpeg.org/ffmpeg/trunk
This commit is contained in:
@@ -93,6 +93,7 @@ void ff_h264_h_loop_filter_chroma_neon(uint8_t *pix, int stride, int alpha,
|
|||||||
int beta, int8_t *tc0);
|
int beta, int8_t *tc0);
|
||||||
|
|
||||||
void ff_h264_idct_add_neon(uint8_t *dst, DCTELEM *block, int stride);
|
void ff_h264_idct_add_neon(uint8_t *dst, DCTELEM *block, int stride);
|
||||||
|
void ff_h264_idct_dc_add_neon(uint8_t *dst, DCTELEM *block, int stride);
|
||||||
|
|
||||||
void ff_dsputil_init_neon(DSPContext *c, AVCodecContext *avctx)
|
void ff_dsputil_init_neon(DSPContext *c, AVCodecContext *avctx)
|
||||||
{
|
{
|
||||||
@@ -164,4 +165,5 @@ void ff_dsputil_init_neon(DSPContext *c, AVCodecContext *avctx)
|
|||||||
c->h264_h_loop_filter_chroma = ff_h264_h_loop_filter_chroma_neon;
|
c->h264_h_loop_filter_chroma = ff_h264_h_loop_filter_chroma_neon;
|
||||||
|
|
||||||
c->h264_idct_add = ff_h264_idct_add_neon;
|
c->h264_idct_add = ff_h264_idct_add_neon;
|
||||||
|
c->h264_idct_dc_add = ff_h264_idct_dc_add_neon;
|
||||||
}
|
}
|
||||||
|
@@ -75,3 +75,22 @@ function ff_h264_idct_add_neon, export=1
|
|||||||
|
|
||||||
bx lr
|
bx lr
|
||||||
.endfunc
|
.endfunc
|
||||||
|
|
||||||
|
function ff_h264_idct_dc_add_neon, export=1
|
||||||
|
vld1.16 {d2[],d3[]}, [r1,:16]
|
||||||
|
vrshr.s16 q1, q1, #6
|
||||||
|
vld1.32 {d0[0]}, [r0,:32], r2
|
||||||
|
vld1.32 {d0[1]}, [r0,:32], r2
|
||||||
|
vaddw.u8 q2, q1, d0
|
||||||
|
vld1.32 {d1[0]}, [r0,:32], r2
|
||||||
|
vld1.32 {d1[1]}, [r0,:32], r2
|
||||||
|
vaddw.u8 q1, q1, d1
|
||||||
|
vqmovun.s16 d0, q2
|
||||||
|
vqmovun.s16 d1, q1
|
||||||
|
sub r0, r0, r2, lsl #2
|
||||||
|
vst1.32 {d0[0]}, [r0,:32], r2
|
||||||
|
vst1.32 {d0[1]}, [r0,:32], r2
|
||||||
|
vst1.32 {d1[0]}, [r0,:32], r2
|
||||||
|
vst1.32 {d1[1]}, [r0,:32], r2
|
||||||
|
bx lr
|
||||||
|
.endfunc
|
||||||
|
Reference in New Issue
Block a user