ppc: h predictor 4x4

2x faster.

Change-Id: I0583dec353299c6797401b646099f18db4e0420d
This commit is contained in:
Luca Barbato 2017-04-09 13:44:41 +00:00 committed by James Zern
parent 58245d7050
commit 2904eb5800
3 changed files with 26 additions and 1 deletions

View File

@ -313,6 +313,10 @@ INTRA_PRED_TEST(MSA, TestIntraPred32, vpx_dc_predictor_32x32_msa,
#endif // HAVE_MSA
#if HAVE_VSX
INTRA_PRED_TEST(VSX, TestIntraPred4, NULL, NULL, NULL, NULL, NULL,
vpx_h_predictor_4x4_vsx, NULL, NULL, NULL, NULL, NULL, NULL,
NULL)
INTRA_PRED_TEST(VSX, TestIntraPred8, vpx_dc_predictor_8x8_vsx, NULL, NULL, NULL,
NULL, NULL, vpx_d45_predictor_8x8_vsx, NULL, NULL, NULL, NULL,
NULL, vpx_tm_predictor_8x8_vsx)

View File

@ -35,6 +35,27 @@ void vpx_v_predictor_32x32_vsx(uint8_t *dst, ptrdiff_t stride,
}
}
static const uint32x4_t mask4 = { 0, 0xFFFFFFFF, 0xFFFFFFFF, 0xFFFFFFFF };
void vpx_h_predictor_4x4_vsx(uint8_t *dst, ptrdiff_t stride,
const uint8_t *above, const uint8_t *left) {
const uint8x16_t d = vec_vsx_ld(0, left);
const uint8x16_t v0 = vec_splat(d, 0);
const uint8x16_t v1 = vec_splat(d, 1);
const uint8x16_t v2 = vec_splat(d, 2);
const uint8x16_t v3 = vec_splat(d, 3);
(void)above;
vec_vsx_st(vec_sel(v0, vec_vsx_ld(0, dst), (uint8x16_t)mask4), 0, dst);
dst += stride;
vec_vsx_st(vec_sel(v1, vec_vsx_ld(0, dst), (uint8x16_t)mask4), 0, dst);
dst += stride;
vec_vsx_st(vec_sel(v2, vec_vsx_ld(0, dst), (uint8x16_t)mask4), 0, dst);
dst += stride;
vec_vsx_st(vec_sel(v3, vec_vsx_ld(0, dst), (uint8x16_t)mask4), 0, dst);
}
void vpx_h_predictor_16x16_vsx(uint8_t *dst, ptrdiff_t stride,
const uint8_t *above, const uint8_t *left) {
const uint8x16_t d = vec_vsx_ld(0, left);

View File

@ -39,7 +39,7 @@ specialize qw/vpx_d63_predictor_4x4 ssse3/;
add_proto qw/void vpx_d63e_predictor_4x4/, "uint8_t *dst, ptrdiff_t y_stride, const uint8_t *above, const uint8_t *left";
add_proto qw/void vpx_h_predictor_4x4/, "uint8_t *dst, ptrdiff_t y_stride, const uint8_t *above, const uint8_t *left";
specialize qw/vpx_h_predictor_4x4 neon dspr2 msa sse2/;
specialize qw/vpx_h_predictor_4x4 neon dspr2 msa sse2 vsx/;
add_proto qw/void vpx_he_predictor_4x4/, "uint8_t *dst, ptrdiff_t y_stride, const uint8_t *above, const uint8_t *left";