2010-05-18 17:58:33 +02:00
|
|
|
/*
|
2010-09-09 14:16:39 +02:00
|
|
|
* Copyright (c) 2010 The WebM project authors. All Rights Reserved.
|
2010-05-18 17:58:33 +02:00
|
|
|
*
|
2010-06-18 18:39:21 +02:00
|
|
|
* Use of this source code is governed by a BSD-style license
|
2010-06-04 22:19:40 +02:00
|
|
|
* that can be found in the LICENSE file in the root of the source
|
|
|
|
* tree. An additional intellectual property rights grant can be found
|
2010-06-18 18:39:21 +02:00
|
|
|
* in the file PATENTS. All contributing project authors may
|
2010-06-04 22:19:40 +02:00
|
|
|
* be found in the AUTHORS file in the root of the source tree.
|
2010-05-18 17:58:33 +02:00
|
|
|
*/
|
|
|
|
|
|
|
|
|
2011-09-15 14:34:12 +02:00
|
|
|
#include "vpx_config.h"
|
2010-05-18 17:58:33 +02:00
|
|
|
#include "vpx_ports/x86.h"
|
2011-02-10 20:41:38 +01:00
|
|
|
#include "vp8/encoder/variance.h"
|
|
|
|
#include "vp8/encoder/onyx_int.h"
|
2010-05-18 17:58:33 +02:00
|
|
|
|
|
|
|
|
|
|
|
#if HAVE_MMX
|
2011-06-14 17:31:50 +02:00
|
|
|
void vp8_short_fdct8x4_mmx(short *input, short *output, int pitch)
|
2010-05-18 17:58:33 +02:00
|
|
|
{
|
2010-10-21 19:53:15 +02:00
|
|
|
vp8_short_fdct4x4_mmx(input, output, pitch);
|
|
|
|
vp8_short_fdct4x4_mmx(input + 4, output + 16, pitch);
|
2010-05-18 17:58:33 +02:00
|
|
|
}
|
|
|
|
|
|
|
|
int vp8_fast_quantize_b_impl_mmx(short *coeff_ptr, short *zbin_ptr,
|
|
|
|
short *qcoeff_ptr, short *dequant_ptr,
|
|
|
|
short *scan_mask, short *round_ptr,
|
|
|
|
short *quant_ptr, short *dqcoeff_ptr);
|
2011-06-14 17:31:50 +02:00
|
|
|
void vp8_fast_quantize_b_mmx(BLOCK *b, BLOCKD *d)
|
2010-05-18 17:58:33 +02:00
|
|
|
{
|
2010-10-22 02:04:30 +02:00
|
|
|
short *scan_mask = vp8_default_zig_zag_mask;//d->scan_order_mask_ptr;
|
|
|
|
short *coeff_ptr = b->coeff;
|
|
|
|
short *zbin_ptr = b->zbin;
|
|
|
|
short *round_ptr = b->round;
|
2010-12-28 20:51:46 +01:00
|
|
|
short *quant_ptr = b->quant_fast;
|
2010-10-22 02:04:30 +02:00
|
|
|
short *qcoeff_ptr = d->qcoeff;
|
2010-05-18 17:58:33 +02:00
|
|
|
short *dqcoeff_ptr = d->dqcoeff;
|
2010-10-22 02:04:30 +02:00
|
|
|
short *dequant_ptr = d->dequant;
|
2010-05-18 17:58:33 +02:00
|
|
|
|
2011-10-25 13:25:11 +02:00
|
|
|
*d->eob = (char)vp8_fast_quantize_b_impl_mmx(
|
|
|
|
coeff_ptr,
|
|
|
|
zbin_ptr,
|
|
|
|
qcoeff_ptr,
|
|
|
|
dequant_ptr,
|
|
|
|
scan_mask,
|
|
|
|
|
|
|
|
round_ptr,
|
|
|
|
quant_ptr,
|
|
|
|
dqcoeff_ptr
|
|
|
|
);
|
2010-05-18 17:58:33 +02:00
|
|
|
}
|
|
|
|
|
|
|
|
int vp8_mbblock_error_mmx_impl(short *coeff_ptr, short *dcoef_ptr, int dc);
|
2011-06-14 17:31:50 +02:00
|
|
|
int vp8_mbblock_error_mmx(MACROBLOCK *mb, int dc)
|
2010-05-18 17:58:33 +02:00
|
|
|
{
|
|
|
|
short *coeff_ptr = mb->block[0].coeff;
|
|
|
|
short *dcoef_ptr = mb->e_mbd.block[0].dqcoeff;
|
|
|
|
return vp8_mbblock_error_mmx_impl(coeff_ptr, dcoef_ptr, dc);
|
|
|
|
}
|
|
|
|
|
|
|
|
int vp8_mbuverror_mmx_impl(short *s_ptr, short *d_ptr);
|
2011-06-14 17:31:50 +02:00
|
|
|
int vp8_mbuverror_mmx(MACROBLOCK *mb)
|
2010-05-18 17:58:33 +02:00
|
|
|
{
|
|
|
|
short *s_ptr = &mb->coeff[256];
|
|
|
|
short *d_ptr = &mb->e_mbd.dqcoeff[256];
|
|
|
|
return vp8_mbuverror_mmx_impl(s_ptr, d_ptr);
|
|
|
|
}
|
|
|
|
|
|
|
|
void vp8_subtract_b_mmx_impl(unsigned char *z, int src_stride,
|
|
|
|
short *diff, unsigned char *predictor,
|
|
|
|
int pitch);
|
2011-06-14 17:31:50 +02:00
|
|
|
void vp8_subtract_b_mmx(BLOCK *be, BLOCKD *bd, int pitch)
|
2010-05-18 17:58:33 +02:00
|
|
|
{
|
|
|
|
unsigned char *z = *(be->base_src) + be->src;
|
|
|
|
unsigned int src_stride = be->src_stride;
|
|
|
|
short *diff = &be->src_diff[0];
|
|
|
|
unsigned char *predictor = &bd->predictor[0];
|
|
|
|
vp8_subtract_b_mmx_impl(z, src_stride, diff, predictor, pitch);
|
|
|
|
}
|
|
|
|
|
|
|
|
#endif
|
|
|
|
|
|
|
|
#if HAVE_SSE2
|
|
|
|
int vp8_mbblock_error_xmm_impl(short *coeff_ptr, short *dcoef_ptr, int dc);
|
2011-06-14 17:31:50 +02:00
|
|
|
int vp8_mbblock_error_xmm(MACROBLOCK *mb, int dc)
|
2010-05-18 17:58:33 +02:00
|
|
|
{
|
|
|
|
short *coeff_ptr = mb->block[0].coeff;
|
|
|
|
short *dcoef_ptr = mb->e_mbd.block[0].dqcoeff;
|
|
|
|
return vp8_mbblock_error_xmm_impl(coeff_ptr, dcoef_ptr, dc);
|
|
|
|
}
|
|
|
|
|
|
|
|
int vp8_mbuverror_xmm_impl(short *s_ptr, short *d_ptr);
|
2011-06-14 17:31:50 +02:00
|
|
|
int vp8_mbuverror_xmm(MACROBLOCK *mb)
|
2010-05-18 17:58:33 +02:00
|
|
|
{
|
|
|
|
short *s_ptr = &mb->coeff[256];
|
|
|
|
short *d_ptr = &mb->e_mbd.dqcoeff[256];
|
|
|
|
return vp8_mbuverror_xmm_impl(s_ptr, d_ptr);
|
|
|
|
}
|
|
|
|
|
2010-10-18 20:15:15 +02:00
|
|
|
void vp8_subtract_b_sse2_impl(unsigned char *z, int src_stride,
|
|
|
|
short *diff, unsigned char *predictor,
|
|
|
|
int pitch);
|
2011-06-14 17:31:50 +02:00
|
|
|
void vp8_subtract_b_sse2(BLOCK *be, BLOCKD *bd, int pitch)
|
2010-10-18 20:15:15 +02:00
|
|
|
{
|
|
|
|
unsigned char *z = *(be->base_src) + be->src;
|
|
|
|
unsigned int src_stride = be->src_stride;
|
|
|
|
short *diff = &be->src_diff[0];
|
|
|
|
unsigned char *predictor = &bd->predictor[0];
|
|
|
|
vp8_subtract_b_sse2_impl(z, src_stride, diff, predictor, pitch);
|
|
|
|
}
|
|
|
|
|
2010-05-18 17:58:33 +02:00
|
|
|
#endif
|
|
|
|
|
|
|
|
void vp8_arch_x86_encoder_init(VP8_COMP *cpi)
|
|
|
|
{
|
|
|
|
#if CONFIG_RUNTIME_CPU_DETECT
|
|
|
|
int flags = x86_simd_caps();
|
|
|
|
|
|
|
|
/* Note:
|
|
|
|
*
|
|
|
|
* This platform can be built without runtime CPU detection as well. If
|
|
|
|
* you modify any of the function mappings present in this file, be sure
|
|
|
|
* to also update them in static mapings (<arch>/filename_<arch>.h)
|
|
|
|
*/
|
|
|
|
|
|
|
|
/* Override default functions with fastest ones for this CPU. */
|
|
|
|
#if HAVE_MMX
|
2011-05-09 17:16:31 +02:00
|
|
|
if (flags & HAS_MMX)
|
2010-05-18 17:58:33 +02:00
|
|
|
{
|
|
|
|
cpi->rtcd.encodemb.berr = vp8_block_error_mmx;
|
2011-06-14 17:31:50 +02:00
|
|
|
cpi->rtcd.encodemb.mberr = vp8_mbblock_error_mmx;
|
|
|
|
cpi->rtcd.encodemb.mbuverr = vp8_mbuverror_mmx;
|
|
|
|
cpi->rtcd.encodemb.subb = vp8_subtract_b_mmx;
|
2010-05-18 17:58:33 +02:00
|
|
|
cpi->rtcd.encodemb.submby = vp8_subtract_mby_mmx;
|
|
|
|
cpi->rtcd.encodemb.submbuv = vp8_subtract_mbuv_mmx;
|
|
|
|
|
2011-06-14 17:31:50 +02:00
|
|
|
/*cpi->rtcd.quantize.fastquantb = vp8_fast_quantize_b_mmx;*/
|
2010-05-18 17:58:33 +02:00
|
|
|
}
|
|
|
|
#endif
|
|
|
|
|
2010-10-27 14:45:24 +02:00
|
|
|
#if HAVE_SSE2
|
2011-05-09 17:16:31 +02:00
|
|
|
if (flags & HAS_SSE2)
|
2010-05-18 17:58:33 +02:00
|
|
|
{
|
|
|
|
cpi->rtcd.encodemb.berr = vp8_block_error_xmm;
|
2011-06-14 17:31:50 +02:00
|
|
|
cpi->rtcd.encodemb.mberr = vp8_mbblock_error_xmm;
|
|
|
|
cpi->rtcd.encodemb.mbuverr = vp8_mbuverror_xmm;
|
|
|
|
cpi->rtcd.encodemb.subb = vp8_subtract_b_sse2;
|
2010-10-18 20:15:15 +02:00
|
|
|
cpi->rtcd.encodemb.submby = vp8_subtract_mby_sse2;
|
|
|
|
cpi->rtcd.encodemb.submbuv = vp8_subtract_mbuv_sse2;
|
2010-05-18 17:58:33 +02:00
|
|
|
|
2011-02-10 20:57:43 +01:00
|
|
|
cpi->rtcd.quantize.quantb = vp8_regular_quantize_b_sse2;
|
2011-03-24 18:31:10 +01:00
|
|
|
cpi->rtcd.quantize.fastquantb = vp8_fast_quantize_b_sse2;
|
2010-12-22 17:23:51 +01:00
|
|
|
|
2011-02-22 09:29:23 +01:00
|
|
|
#if !(CONFIG_REALTIME_ONLY)
|
2010-12-22 17:23:51 +01:00
|
|
|
cpi->rtcd.temporal.apply = vp8_temporal_filter_apply_sse2;
|
2011-08-22 21:36:28 +02:00
|
|
|
#endif
|
|
|
|
|
2010-05-18 17:58:33 +02:00
|
|
|
}
|
|
|
|
#endif
|
|
|
|
|
2010-10-27 14:45:24 +02:00
|
|
|
#if HAVE_SSE3
|
2011-05-09 17:16:31 +02:00
|
|
|
if (flags & HAS_SSE3)
|
2010-05-18 17:58:33 +02:00
|
|
|
{
|
|
|
|
cpi->rtcd.search.full_search = vp8_full_search_sadx3;
|
|
|
|
cpi->rtcd.search.diamond_search = vp8_diamond_search_sadx4;
|
2011-05-06 18:51:31 +02:00
|
|
|
cpi->rtcd.search.refining_search = vp8_refining_search_sadx4;
|
2010-05-18 17:58:33 +02:00
|
|
|
}
|
|
|
|
#endif
|
|
|
|
|
2010-10-27 14:45:24 +02:00
|
|
|
#if HAVE_SSSE3
|
2011-05-09 17:16:31 +02:00
|
|
|
if (flags & HAS_SSSE3)
|
2010-05-18 17:58:33 +02:00
|
|
|
{
|
2011-04-07 22:40:05 +02:00
|
|
|
cpi->rtcd.quantize.fastquantb = vp8_fast_quantize_b_ssse3;
|
2010-05-18 17:58:33 +02:00
|
|
|
}
|
2010-10-27 14:45:24 +02:00
|
|
|
#endif
|
2010-05-18 17:58:33 +02:00
|
|
|
|
2011-03-08 15:05:18 +01:00
|
|
|
|
|
|
|
|
2010-10-27 14:45:24 +02:00
|
|
|
#if HAVE_SSE4_1
|
2011-05-09 17:16:31 +02:00
|
|
|
if (flags & HAS_SSE4_1)
|
2010-10-27 14:45:24 +02:00
|
|
|
{
|
|
|
|
cpi->rtcd.search.full_search = vp8_full_search_sadx8;
|
2011-04-13 22:38:02 +02:00
|
|
|
|
|
|
|
cpi->rtcd.quantize.quantb = vp8_regular_quantize_b_sse4;
|
2010-10-27 14:45:24 +02:00
|
|
|
}
|
2010-05-18 17:58:33 +02:00
|
|
|
#endif
|
2010-10-27 14:45:24 +02:00
|
|
|
|
2010-05-18 17:58:33 +02:00
|
|
|
#endif
|
|
|
|
}
|