763 lines
24 KiB
C++
763 lines
24 KiB
C++
/*
|
|
* Copyright (c) 2010 The WebM project authors. All Rights Reserved.
|
|
*
|
|
* Use of this source code is governed by a BSD-style license
|
|
* that can be found in the LICENSE file in the root of the source
|
|
* tree. An additional intellectual property rights grant can be found
|
|
* in the file PATENTS. All contributing project authors may
|
|
* be found in the AUTHORS file in the root of the source tree.
|
|
*/
|
|
|
|
#include "vpx_config.h"
|
|
|
|
#include "vp8/common/optimisation_vecops.h"
|
|
#if defined(VECOPS_ENABLED)
|
|
|
|
#include <assert.h>
|
|
|
|
#include "vp8/common/optimisation_profiling.h"
|
|
|
|
#define DEBUG_DEQUANT_IDCT 0
|
|
|
|
#if DEBUG_DEQUANT_IDCT
|
|
#include "vp8/common/optimisation_debug.h"
|
|
#endif
|
|
|
|
DECLARE_ALIGNED(32, const float, k_sincosPi8Sqrt2[8]) =
|
|
{0.5411987f, 0.5411987f, 0.5411987f, 0.5411987f, //sin // use values equal to the integer approximations
|
|
1.3065643f, 1.3065643f, 1.3065643f, 1.3065643f}; //cos // rather than the exact ones.
|
|
|
|
template <int count>
|
|
VPX_FORCEINLINE static void idct_CacheLines(const void* __restrict pAddr, int iStride)
|
|
{
|
|
for (int i=0; i<count; ++i)
|
|
{
|
|
CacheTouch(pAddr, i*iStride);
|
|
}
|
|
}
|
|
|
|
VPX_FORCEINLINE static void idct_LoadAndUnpack16I16s(
|
|
short* __restrict pIn,
|
|
v128i_t* __restrict pvOut)
|
|
{
|
|
pvOut[1] = VecLoadAlignedI32(pIn, 0);
|
|
pvOut[3] = VecLoadAlignedI32(pIn, 16);
|
|
pvOut[0] = VecUnpackLoSignedHalf(pvOut[1]);
|
|
pvOut[1] = VecUnpackHiSignedHalf(pvOut[1]);
|
|
pvOut[2] = VecUnpackLoSignedHalf(pvOut[3]);
|
|
pvOut[3] = VecUnpackHiSignedHalf(pvOut[3]);
|
|
}
|
|
|
|
VPX_FORCEINLINE static void idct_LoadUnpackAndConvert16I16sToF32s(
|
|
short* __restrict pIn,
|
|
v128f_t* __restrict pvOut
|
|
)
|
|
{
|
|
v128i_t vI32[4];
|
|
idct_LoadAndUnpack16I16s(pIn, vI32);
|
|
pvOut[0] = VecConvertI32ToF32(vI32[0]);
|
|
pvOut[1] = VecConvertI32ToF32(vI32[1]);
|
|
pvOut[2] = VecConvertI32ToF32(vI32[2]);
|
|
pvOut[3] = VecConvertI32ToF32(vI32[3]);
|
|
}
|
|
|
|
VPX_FORCEINLINE static v128i_t idct_CalcQDQ(
|
|
v128i_t vQ,
|
|
v128f_t vDQ
|
|
)
|
|
{
|
|
v128f_t vfQ;
|
|
vfQ = VecConvertI32ToF32(vQ);
|
|
vfQ = VecMulFloat(vfQ, vDQ);
|
|
return VecConvertF32ToI32(vfQ);
|
|
}
|
|
|
|
VPX_FORCEINLINE static void idct_Calculate(v128i_t* __restrict vInOutQDQ)
|
|
{
|
|
v128i_t vP, vQ, vR, vS;
|
|
v128f_t vBs, vBc, vDs, vDc;
|
|
v128f_t vSinFactor = VecLoadAlignedF32(k_sincosPi8Sqrt2, 0);
|
|
v128f_t vCosFactor = VecLoadAlignedF32(k_sincosPi8Sqrt2, 16);
|
|
|
|
vBs = VecConvertI32ToF32(vInOutQDQ[1]);
|
|
vDs = VecConvertI32ToF32(vInOutQDQ[3]);
|
|
vBc = VecMulFloat(vBs, vCosFactor);
|
|
vDc = VecMulFloat(vDs, vCosFactor);
|
|
vBs = VecMulFloat(vBs, vSinFactor);
|
|
vDs = VecMulFloat(vDs, vSinFactor);
|
|
vBc = VecRoundFloatNegInfinity(vBc);
|
|
vDc = VecRoundFloatNegInfinity(vDc);
|
|
vBs = VecRoundFloatNegInfinity(vBs);
|
|
vDs = VecRoundFloatNegInfinity(vDs);
|
|
|
|
vP = VecAddSignedWordSaturate(vInOutQDQ[0], vInOutQDQ[2]);
|
|
vQ = VecSubSignedWordSaturate(vInOutQDQ[0], vInOutQDQ[2]);
|
|
vR = VecSubSignedWordSaturate(VecConvertF32ToI32(vBs), VecConvertF32ToI32(vDc));
|
|
vS = VecAddSignedWordSaturate(VecConvertF32ToI32(vBc), VecConvertF32ToI32(vDs));
|
|
|
|
vInOutQDQ[0] = VecAddSignedWordSaturate(vP, vS);
|
|
vInOutQDQ[1] = VecAddSignedWordSaturate(vQ, vR);
|
|
vInOutQDQ[2] = VecSubSignedWordSaturate(vQ, vR);
|
|
vInOutQDQ[3] = VecSubSignedWordSaturate(vP, vS);
|
|
}
|
|
|
|
VPX_FORCEINLINE static void idct_Transpose4x4(v128i_t* __restrict pvRows)
|
|
{
|
|
v128i_t vTemp[4];
|
|
vTemp[0] = VecInterleaveLoWord(pvRows[0], pvRows[2]);
|
|
vTemp[1] = VecInterleaveHiWord(pvRows[0], pvRows[2]);
|
|
vTemp[2] = VecInterleaveLoWord(pvRows[1], pvRows[3]);
|
|
vTemp[3] = VecInterleaveHiWord(pvRows[1], pvRows[3]);
|
|
pvRows[0] = VecInterleaveLoWord( vTemp[0], vTemp[2]);
|
|
pvRows[1] = VecInterleaveHiWord( vTemp[0], vTemp[2]);
|
|
pvRows[2] = VecInterleaveLoWord( vTemp[1], vTemp[3]);
|
|
pvRows[3] = VecInterleaveHiWord( vTemp[1], vTemp[3]);
|
|
}
|
|
|
|
VPX_FORCEINLINE static void idct_AddResultLo(unsigned char* __restrict pDst, v128i_t vVal)
|
|
{
|
|
v128i_t vCur, vLo;
|
|
vCur = VecLoadUnalignedI32(pDst, 0);
|
|
vLo = VecUnpackLoUnsignedByte(vCur);
|
|
vLo = VecAddSignedHalfSaturate(vLo, vVal);
|
|
vCur = VecPackSignedHalfToUnsignedByteSaturate(vLo, VecUnpackHiUnsignedByte(vCur));
|
|
VecStoreUnalignedI32(vCur, pDst, 0);
|
|
}
|
|
|
|
VPX_FORCEINLINE static void idct_AddResult(unsigned char* pDst, v128i_t vLoR, v128i_t vHiR)
|
|
{
|
|
v128i_t vCur, vLo, vHi;
|
|
vCur = VecLoadUnalignedI32(pDst, 0);
|
|
vLo = VecUnpackLoUnsignedByte(vCur);
|
|
vHi = VecUnpackHiUnsignedByte(vCur);
|
|
vLo = VecAddSignedHalfSaturate(vLo, vLoR);
|
|
vHi = VecAddSignedHalfSaturate(vHi, vHiR);
|
|
vCur = VecPackSignedHalfToUnsignedByteSaturate(vLo, vHi);
|
|
VecStoreUnalignedI32(vCur, pDst, 0);
|
|
}
|
|
|
|
VPX_FORCEINLINE void idct_Calc4x4(
|
|
short* __restrict pQ,
|
|
v128f_t* __restrict pvDQ,
|
|
v128i_t* __restrict pvOut)
|
|
{
|
|
v128i_t vQ[4], vQDQ[4];
|
|
idct_LoadAndUnpack16I16s(pQ, vQ);
|
|
vQDQ[0] = idct_CalcQDQ(vQ[0], pvDQ[0]);
|
|
vQDQ[1] = idct_CalcQDQ(vQ[1], pvDQ[1]);
|
|
vQDQ[2] = idct_CalcQDQ(vQ[2], pvDQ[2]);
|
|
vQDQ[3] = idct_CalcQDQ(vQ[3], pvDQ[3]);
|
|
idct_Calculate(vQDQ);
|
|
idct_Transpose4x4(vQDQ);
|
|
idct_Calculate(vQDQ);
|
|
idct_Transpose4x4(vQDQ);
|
|
pvOut[0] = VecShiftRightArithmeticWordImmediate<3>(VecAddSignedWordSaturate(vQDQ[0], VecSplatImmediateWord<4>()));
|
|
pvOut[1] = VecShiftRightArithmeticWordImmediate<3>(VecAddSignedWordSaturate(vQDQ[1], VecSplatImmediateWord<4>()));
|
|
pvOut[2] = VecShiftRightArithmeticWordImmediate<3>(VecAddSignedWordSaturate(vQDQ[2], VecSplatImmediateWord<4>()));
|
|
pvOut[3] = VecShiftRightArithmeticWordImmediate<3>(VecAddSignedWordSaturate(vQDQ[3], VecSplatImmediateWord<4>()));
|
|
}
|
|
|
|
void vp8_idct_dequant_0_1x_vecops(
|
|
short* __restrict q
|
|
, short* __restrict dq
|
|
, unsigned char* __restrict dst
|
|
, int dst_stride)
|
|
{
|
|
PRF_Scoped("vp8_idct_dequant_0_1x_vecops");
|
|
|
|
// the block will have ((q[0] * dq[0]) + 4) >> 3 added to it
|
|
v128i_t vRes;
|
|
{
|
|
v128i_t vQ, vDQ;
|
|
v128f_t vQf, vDQf;
|
|
vQ = VecLoadAlignedI32(q, 0);
|
|
vDQ = VecLoadUnalignedI32(dq, 0);
|
|
|
|
vQ = VecUnpackLoSignedHalf(vQ);
|
|
vDQ = VecUnpackLoSignedHalf(vDQ);
|
|
vQf = VecConvertI32ToF32(vQ);
|
|
vDQf = VecConvertI32ToF32(vDQ);
|
|
vQf = VecMulFloat(vQf, vDQf);
|
|
vQ = VecConvertF32ToI32(vQf);
|
|
vQ = VecAddSignedWordSaturate(vQ, VecSplatImmediateWord<4>());
|
|
vQ = VecShiftRightArithmeticWordImmediate<3>(vQ);
|
|
vQ = VecShuffleWord<0,0,0,0>(vQ);
|
|
vRes = VecPackSignedWordToSignedHalfSaturate(vQ, VecSplatImmediateWord<0>());
|
|
}
|
|
|
|
idct_AddResultLo(dst , vRes);
|
|
idct_AddResultLo(dst + dst_stride, vRes);
|
|
idct_AddResultLo(dst + 2*dst_stride, vRes);
|
|
idct_AddResultLo(dst + 3*dst_stride, vRes);
|
|
q[0] = 0;
|
|
}
|
|
|
|
void vp8_idct_dequant_0_2x_vecops(
|
|
short* __restrict q
|
|
, short* __restrict dq
|
|
, unsigned char* __restrict dst
|
|
, int dst_stride)
|
|
{
|
|
PRF_Scoped("vp8_idct_dequant_0_2x_vecops");
|
|
// calculate the values to add to the two 4x4 blocks at dst and dst+4.
|
|
// the first block will have ((q[0] * dq[0]) + 4) >> 3 added to it
|
|
// the second block will have ((q[16] * dq[0]) + 4) >> 3 added to it
|
|
v128i_t vRes;
|
|
{
|
|
v128i_t vQ[2], vDQ;
|
|
v128f_t vQf[2], vDQf;
|
|
vQ[0] = VecLoadAlignedI32(q, 0);
|
|
vQ[1] = VecLoadAlignedI32(q, 32);
|
|
vDQ = VecLoadUnalignedI32(dq, 0);
|
|
|
|
vQ[0] = VecUnpackLoSignedHalf(vQ[0]);
|
|
vQ[1] = VecUnpackLoSignedHalf(vQ[1]);
|
|
vDQ = VecUnpackLoSignedHalf(vDQ);
|
|
vQf[0] = VecConvertI32ToF32(vQ[0]);
|
|
vQf[1] = VecConvertI32ToF32(vQ[1]);
|
|
vDQf = VecConvertI32ToF32(vDQ);
|
|
vQf[0] = VecMulFloat(vQf[0], vDQf);
|
|
vQf[1] = VecMulFloat(vQf[1], vDQf);
|
|
vQ[0] = VecConvertF32ToI32(vQf[0]);
|
|
vQ[1] = VecConvertF32ToI32(vQf[1]);
|
|
|
|
vQ[0] = VecAddSignedWordSaturate(vQ[0], VecSplatImmediateWord<4>());
|
|
vQ[1] = VecAddSignedWordSaturate(vQ[1], VecSplatImmediateWord<4>());
|
|
vQ[0] = VecShiftRightArithmeticWordImmediate<3>(vQ[0]);
|
|
vQ[1] = VecShiftRightArithmeticWordImmediate<3>(vQ[1]);
|
|
|
|
vQ[0] = VecShuffleWord<0,0,0,0>(vQ[0]);
|
|
vQ[1] = VecShuffleWord<0,0,0,0>(vQ[1]);
|
|
vRes = VecPackSignedWordToSignedHalfSaturate(vQ[0], vQ[1]);
|
|
}
|
|
|
|
idct_AddResultLo(dst , vRes);
|
|
idct_AddResultLo(dst + dst_stride, vRes);
|
|
idct_AddResultLo(dst + 2*dst_stride, vRes);
|
|
idct_AddResultLo(dst + 3*dst_stride, vRes);
|
|
q[0] = 0;
|
|
q[16] = 0;
|
|
}
|
|
|
|
void vp8_idct_dequant_0_4x_vecops(
|
|
short* __restrict q
|
|
, short* __restrict dq
|
|
, unsigned char* __restrict dst
|
|
, int dst_stride)
|
|
{
|
|
// calculate the values to add to the two 4x4 blocks at dst and dst+4.
|
|
// the first block will have ((q[0] * dq[0]) + 4) >> 3 added to it
|
|
// the second block will have ((q[16] * dq[0]) + 4) >> 3 added to it
|
|
// the third block will have ((q[32] * dq[0]) + 4) >> 3 added to it
|
|
// the fourth block will have ((q[48] * dq[0]) + 4) >> 3 added to it
|
|
v128i_t vResLo, vResHi;
|
|
{
|
|
v128i_t vQ[4], vDQ;
|
|
v128f_t vQf[4], vDQf;
|
|
vQ[0] = VecLoadAlignedI32(q, 0);
|
|
vQ[1] = VecLoadAlignedI32(q, 32);
|
|
vQ[2] = VecLoadAlignedI32(q, 64);
|
|
vQ[3] = VecLoadAlignedI32(q, 96);
|
|
vDQ = VecLoadUnalignedI32(dq, 0);
|
|
|
|
vQ[0] = VecUnpackLoSignedHalf(vQ[0]);
|
|
vQ[1] = VecUnpackLoSignedHalf(vQ[1]);
|
|
vQ[2] = VecUnpackLoSignedHalf(vQ[2]);
|
|
vQ[3] = VecUnpackLoSignedHalf(vQ[3]);
|
|
vDQ = VecUnpackLoSignedHalf(vDQ);
|
|
vQf[0] = VecConvertI32ToF32(vQ[0]);
|
|
vQf[1] = VecConvertI32ToF32(vQ[1]);
|
|
vQf[2] = VecConvertI32ToF32(vQ[2]);
|
|
vQf[3] = VecConvertI32ToF32(vQ[3]);
|
|
vDQf = VecConvertI32ToF32(vDQ);
|
|
vQf[0] = VecMulFloat(vQf[0], vDQf);
|
|
vQf[1] = VecMulFloat(vQf[1], vDQf);
|
|
vQf[2] = VecMulFloat(vQf[2], vDQf);
|
|
vQf[3] = VecMulFloat(vQf[3], vDQf);
|
|
vQ[0] = VecConvertF32ToI32(vQf[0]);
|
|
vQ[1] = VecConvertF32ToI32(vQf[1]);
|
|
vQ[2] = VecConvertF32ToI32(vQf[2]);
|
|
vQ[3] = VecConvertF32ToI32(vQf[3]);
|
|
|
|
vQ[0] = VecAddSignedWordSaturate(vQ[0], VecSplatImmediateWord<4>());
|
|
vQ[1] = VecAddSignedWordSaturate(vQ[1], VecSplatImmediateWord<4>());
|
|
vQ[2] = VecAddSignedWordSaturate(vQ[2], VecSplatImmediateWord<4>());
|
|
vQ[3] = VecAddSignedWordSaturate(vQ[3], VecSplatImmediateWord<4>());
|
|
vQ[0] = VecShiftRightArithmeticWordImmediate<3>(vQ[0]);
|
|
vQ[1] = VecShiftRightArithmeticWordImmediate<3>(vQ[1]);
|
|
vQ[2] = VecShiftRightArithmeticWordImmediate<3>(vQ[2]);
|
|
vQ[3] = VecShiftRightArithmeticWordImmediate<3>(vQ[3]);
|
|
|
|
vQ[0] = VecShuffleWord<0,0,0,0>(vQ[0]);
|
|
vQ[1] = VecShuffleWord<0,0,0,0>(vQ[1]);
|
|
vQ[2] = VecShuffleWord<0,0,0,0>(vQ[2]);
|
|
vQ[3] = VecShuffleWord<0,0,0,0>(vQ[3]);
|
|
vResLo = VecPackSignedWordToSignedHalfSaturate(vQ[0], vQ[1]);
|
|
vResHi = VecPackSignedWordToSignedHalfSaturate(vQ[2], vQ[3]);
|
|
}
|
|
idct_AddResult(dst , vResLo, vResHi);
|
|
idct_AddResult(dst + dst_stride, vResLo, vResHi);
|
|
idct_AddResult(dst + 2*dst_stride, vResLo, vResHi);
|
|
idct_AddResult(dst + 3*dst_stride, vResLo, vResHi);
|
|
q[0] = 0;
|
|
q[16] = 0;
|
|
q[32] = 0;
|
|
q[48] = 0;
|
|
}
|
|
|
|
void vp8_idct_dequant_full_2x_vecops(
|
|
short* __restrict q
|
|
, short* __restrict dq
|
|
, unsigned char* __restrict dst
|
|
, int dst_stride)
|
|
{
|
|
PRF_Scoped("vp8_idct_dequant_full_2x_vecops");
|
|
|
|
//
|
|
// This function dequantises 32 q values (16 per 4x4 output block),
|
|
// performs the IDCT function and adds the result to the current values in dst.
|
|
// Each 16 values of Q are multiplied by the 16 values of DQ.
|
|
//
|
|
v128i_t vResults[8];
|
|
|
|
v128f_t vDQ[4];
|
|
idct_LoadUnpackAndConvert16I16sToF32s(dq, vDQ);
|
|
idct_Calc4x4(q , vDQ, &vResults[0]);
|
|
idct_Calc4x4(q+16, vDQ, &vResults[4]);
|
|
|
|
vResults[0] = VecPackSignedWordToSignedHalfSaturate(vResults[0], vResults[4]);
|
|
vResults[1] = VecPackSignedWordToSignedHalfSaturate(vResults[1], vResults[5]);
|
|
vResults[2] = VecPackSignedWordToSignedHalfSaturate(vResults[2], vResults[6]);
|
|
vResults[3] = VecPackSignedWordToSignedHalfSaturate(vResults[3], vResults[7]);
|
|
|
|
idct_AddResultLo(dst , vResults[0]);
|
|
idct_AddResultLo(dst + dst_stride, vResults[1]);
|
|
idct_AddResultLo(dst + 2*dst_stride, vResults[2]);
|
|
idct_AddResultLo(dst + 3*dst_stride, vResults[3]);
|
|
|
|
v128i_t vZero = VecSplatImmediateWord<0>();
|
|
VecStoreAlignedI32(vZero, q, 0);
|
|
VecStoreAlignedI32(vZero, q, 0x10);
|
|
VecStoreAlignedI32(vZero, q, 0x20);
|
|
VecStoreAlignedI32(vZero, q, 0x30);
|
|
}
|
|
|
|
void vp8_idct_dequant_full_4x_vecops(
|
|
short* __restrict q
|
|
, short* __restrict dq
|
|
, unsigned char* __restrict dst
|
|
, int dst_stride)
|
|
{
|
|
//
|
|
// This function dequantises/performs the idct on 64 q values (16 per 4x4 output block, i.e. a whole block high row of a macroblock)
|
|
// and adds the result to the current values in dst.
|
|
// Each 16 values of Q are multiplied by the 16 values of DQ.
|
|
// While it is more efficient than the 2x version (as it writes out the data more efficiently)
|
|
// it is situational - when the whole row of blocks have more than one dct coefficient.
|
|
// Also, it might be beneficial to provide a 4x version of the dequant_0.
|
|
//
|
|
v128i_t vResults[16];
|
|
|
|
v128f_t vDQ[4];
|
|
idct_LoadUnpackAndConvert16I16sToF32s(dq, vDQ);
|
|
idct_Calc4x4(q , vDQ, &vResults[ 0]);
|
|
idct_Calc4x4(q+16, vDQ, &vResults[ 4]);
|
|
idct_Calc4x4(q+32, vDQ, &vResults[ 8]);
|
|
idct_Calc4x4(q+48, vDQ, &vResults[12]);
|
|
|
|
vResults[0] = VecPackSignedWordToSignedHalfSaturate(vResults[ 0], vResults[ 4]);
|
|
vResults[1] = VecPackSignedWordToSignedHalfSaturate(vResults[ 1], vResults[ 5]);
|
|
vResults[2] = VecPackSignedWordToSignedHalfSaturate(vResults[ 2], vResults[ 6]);
|
|
vResults[3] = VecPackSignedWordToSignedHalfSaturate(vResults[ 3], vResults[ 7]);
|
|
vResults[4] = VecPackSignedWordToSignedHalfSaturate(vResults[ 8], vResults[12]);
|
|
vResults[5] = VecPackSignedWordToSignedHalfSaturate(vResults[ 9], vResults[13]);
|
|
vResults[6] = VecPackSignedWordToSignedHalfSaturate(vResults[10], vResults[14]);
|
|
vResults[7] = VecPackSignedWordToSignedHalfSaturate(vResults[11], vResults[15]);
|
|
|
|
idct_AddResult(dst , vResults[0], vResults[4]);
|
|
idct_AddResult(dst + dst_stride, vResults[1], vResults[5]);
|
|
idct_AddResult(dst + 2*dst_stride, vResults[2], vResults[6]);
|
|
idct_AddResult(dst + 3*dst_stride, vResults[3], vResults[7]);
|
|
|
|
v128i_t vZero = VecSplatImmediateWord<0>();
|
|
VecStoreAlignedI32(vZero, q, 0);
|
|
VecStoreAlignedI32(vZero, q, 0x10);
|
|
VecStoreAlignedI32(vZero, q, 0x20);
|
|
VecStoreAlignedI32(vZero, q, 0x30);
|
|
VecStoreAlignedI32(vZero, q, 0x40);
|
|
VecStoreAlignedI32(vZero, q, 0x50);
|
|
VecStoreAlignedI32(vZero, q, 0x60);
|
|
VecStoreAlignedI32(vZero, q, 0x70);
|
|
}
|
|
|
|
#if DEBUG_DEQUANT_IDCT
|
|
extern "C" void vp8_dequant_idct_add_c(short *input, short *dq, unsigned char *dest, int stride);
|
|
extern "C" void vp8_dc_only_idct_add_c(short input_dc, unsigned char * pred, int pred_stride, unsigned char *dst_ptr, int dst_stride);
|
|
|
|
template <int kCols, int kRows>
|
|
int idctdebug_compare_block(short* q, short* dq, unsigned char* dst, int stride, const char* eobs, const unsigned char* to_compare, int to_compare_stride)
|
|
{
|
|
int error_row = -1;
|
|
|
|
for (int i = 0; i < kRows; i++)
|
|
{
|
|
for (int j = 0; j < kCols; j++)
|
|
{
|
|
if (*eobs++ > 1)
|
|
{
|
|
vp8_dequant_idct_add_c (q, dq, dst, stride);
|
|
}
|
|
else
|
|
{
|
|
vp8_dc_only_idct_add_c (q[0]*dq[0], dst, stride, dst, stride);
|
|
((int *)q)[0] = 0;
|
|
}
|
|
|
|
if (!compare_b(dst, stride, to_compare, to_compare_stride))
|
|
{
|
|
error_row = i;
|
|
}
|
|
|
|
q += 16;
|
|
dst += 4;
|
|
to_compare += 4;
|
|
}
|
|
|
|
dst += 4*stride - (kCols*4);
|
|
to_compare += 4*to_compare_stride - (kCols*4);
|
|
}
|
|
|
|
return error_row;
|
|
}
|
|
|
|
template <int kCols, int kRows>
|
|
void idctdebug_process_single_row(int error_row, short* q, short* dq, char* eobs, unsigned char* dst, int dst_stride )
|
|
{
|
|
int qofst = error_row*16*kCols;
|
|
int dstofst = error_row*dst_stride*4;
|
|
eobs += error_row*kRows;
|
|
for (int i=0; i<kCols; i+=2)
|
|
{
|
|
if (((short *)(eobs))[i/2])
|
|
{
|
|
if (((short *)(eobs))[i/2] & 0xfefe)
|
|
{
|
|
vp8_idct_dequant_full_2x_vecops(q+qofst, dq, dst+dstofst, dst_stride);
|
|
}
|
|
else
|
|
{
|
|
vp8_idct_dequant_0_2x_vecops(q+qofst, dq, dst+dstofst, dst_stride);
|
|
}
|
|
}
|
|
qofst += 32;
|
|
dstofst += 8;
|
|
}
|
|
}
|
|
#endif
|
|
|
|
extern "C"
|
|
void vp8_dequant_idct_add_vecops(
|
|
short* __restrict q
|
|
, short* __restrict dq
|
|
, unsigned char* __restrict dst
|
|
, int dst_stride)
|
|
{
|
|
PRF_Scoped("vp8_dequant_idct_add_vecops");
|
|
v128i_t vResults[4];
|
|
|
|
v128f_t vDQ[4];
|
|
idct_LoadUnpackAndConvert16I16sToF32s(dq, vDQ);
|
|
idct_Calc4x4(q, vDQ, vResults);
|
|
|
|
vResults[0] = VecPackSignedWordToSignedHalfSaturate(vResults[0], VecSplatImmediateWord<0>());
|
|
vResults[1] = VecPackSignedWordToSignedHalfSaturate(vResults[1], VecSplatImmediateWord<0>());
|
|
vResults[2] = VecPackSignedWordToSignedHalfSaturate(vResults[2], VecSplatImmediateWord<0>());
|
|
vResults[3] = VecPackSignedWordToSignedHalfSaturate(vResults[3], VecSplatImmediateWord<0>());
|
|
|
|
idct_AddResultLo(dst , vResults[0]);
|
|
idct_AddResultLo(dst + dst_stride, vResults[1]);
|
|
idct_AddResultLo(dst + 2*dst_stride, vResults[2]);
|
|
idct_AddResultLo(dst + 3*dst_stride, vResults[3]);
|
|
|
|
v128i_t vZero = VecSplatImmediateWord<0>();
|
|
VecStoreAlignedI32(vZero, q, 0);
|
|
VecStoreAlignedI32(vZero, q, 0x10);
|
|
}
|
|
|
|
extern "C"
|
|
void vp8_dequant_idct_add_y_block_vecops(
|
|
short* __restrict q
|
|
, short* __restrict dq
|
|
, unsigned char* __restrict dst
|
|
, int stride
|
|
, char* eobs)
|
|
{
|
|
#if DEBUG_DEQUANT_IDCT
|
|
unsigned char* orig_dst = dst;
|
|
char* orig_eobs = eobs;
|
|
char* orig_eobs2 = eobs;
|
|
int temp_pitch = 16;
|
|
|
|
unsigned char temp_buffer[256];
|
|
short temp_q[256];
|
|
duplicate_buffer(16, 16, dst, stride, temp_buffer, temp_pitch);
|
|
duplicate_buffer(sizeof(short)*256, 1, (unsigned char*)q, sizeof(short)*256, (unsigned char*)temp_q, sizeof(short)*256);
|
|
|
|
unsigned char temp_buffer2[256];
|
|
short temp_q2[256];
|
|
duplicate_buffer(16, 16, dst, stride, temp_buffer2, temp_pitch);
|
|
duplicate_buffer(sizeof(short)*256, 1, (unsigned char*)q, sizeof(short)*256, (unsigned char*)temp_q2, sizeof(short)*256);
|
|
#endif
|
|
// PIXBeginNamedEvent(0, "Dequantize IDCT add Y block");
|
|
#if 0
|
|
// effectively what the section below does, but without the cache prefetching
|
|
for (int i = 0; i < 4; i++)
|
|
{
|
|
if (((short *)(eobs))[0])
|
|
{
|
|
if (((short *)(eobs))[0] & 0xfefe)
|
|
{
|
|
vp8_idct_dequant_full_2x_vecops(q, dq, dst, stride);
|
|
}
|
|
else
|
|
{
|
|
vp8_idct_dequant_0_2x_vecops(q, dq, dst, stride);
|
|
}
|
|
}
|
|
if (((short *)(eobs))[1])
|
|
{
|
|
if (((short *)(eobs))[1] & 0xfefe)
|
|
{
|
|
vp8_idct_dequant_full_2x_vecops(q+32, dq, dst+8, stride);
|
|
}
|
|
else
|
|
{
|
|
vp8_idct_dequant_0_2x_vecops(q+32, dq, dst+8, stride);
|
|
}
|
|
}
|
|
q += 64;
|
|
dst += stride*4;
|
|
eobs += 4;
|
|
}
|
|
#elif 1
|
|
|
|
idct_CacheLines<8>(dst, stride);
|
|
|
|
if (((short *)(eobs))[0])
|
|
{
|
|
if (((short *)(eobs))[0] & 0xfefe)
|
|
{
|
|
vp8_idct_dequant_full_2x_vecops(q, dq, dst, stride);
|
|
}
|
|
else
|
|
{
|
|
vp8_idct_dequant_0_2x_vecops(q, dq, dst, stride);
|
|
}
|
|
}
|
|
if (((short *)(eobs))[1])
|
|
{
|
|
if (((short *)(eobs))[1] & 0xfefe)
|
|
{
|
|
vp8_idct_dequant_full_2x_vecops(q+32, dq, dst+8, stride);
|
|
}
|
|
else
|
|
{
|
|
vp8_idct_dequant_0_2x_vecops(q+32, dq, dst+8, stride);
|
|
}
|
|
}
|
|
q += 64;
|
|
dst += stride*4;
|
|
eobs += 4;
|
|
|
|
idct_CacheLines<4>(dst + 8*stride, stride);
|
|
|
|
if (((short *)(eobs))[0])
|
|
{
|
|
if (((short *)(eobs))[0] & 0xfefe)
|
|
{
|
|
vp8_idct_dequant_full_2x_vecops(q, dq, dst, stride);
|
|
}
|
|
else
|
|
{
|
|
vp8_idct_dequant_0_2x_vecops(q, dq, dst, stride);
|
|
}
|
|
}
|
|
if (((short *)(eobs))[1])
|
|
{
|
|
if (((short *)(eobs))[1] & 0xfefe)
|
|
{
|
|
vp8_idct_dequant_full_2x_vecops(q+32, dq, dst+8, stride);
|
|
}
|
|
else
|
|
{
|
|
vp8_idct_dequant_0_2x_vecops(q+32, dq, dst+8, stride);
|
|
}
|
|
}
|
|
q += 64;
|
|
dst += stride*4;
|
|
eobs += 4;
|
|
|
|
idct_CacheLines<4>(dst + 8*stride, stride);
|
|
|
|
if (((short *)(eobs))[0])
|
|
{
|
|
if (((short *)(eobs))[0] & 0xfefe)
|
|
{
|
|
vp8_idct_dequant_full_2x_vecops(q, dq, dst, stride);
|
|
}
|
|
else
|
|
{
|
|
vp8_idct_dequant_0_2x_vecops(q, dq, dst, stride);
|
|
}
|
|
}
|
|
if (((short *)(eobs))[1])
|
|
{
|
|
if (((short *)(eobs))[1] & 0xfefe)
|
|
{
|
|
vp8_idct_dequant_full_2x_vecops(q+32, dq, dst+8, stride);
|
|
}
|
|
else
|
|
{
|
|
vp8_idct_dequant_0_2x_vecops(q+32, dq, dst+8, stride);
|
|
}
|
|
}
|
|
q += 64;
|
|
dst += stride*4;
|
|
eobs += 4;
|
|
|
|
if (((short *)(eobs))[0])
|
|
{
|
|
if (((short *)(eobs))[0] & 0xfefe)
|
|
{
|
|
vp8_idct_dequant_full_2x_vecops(q, dq, dst, stride);
|
|
}
|
|
else
|
|
{
|
|
vp8_idct_dequant_0_2x_vecops(q, dq, dst, stride);
|
|
}
|
|
}
|
|
if (((short *)(eobs))[1])
|
|
{
|
|
if (((short *)(eobs))[1] & 0xfefe)
|
|
{
|
|
vp8_idct_dequant_full_2x_vecops(q+32, dq, dst+8, stride);
|
|
}
|
|
else
|
|
{
|
|
vp8_idct_dequant_0_2x_vecops(q+32, dq, dst+8, stride);
|
|
}
|
|
}
|
|
#else
|
|
// the intention behind this approach is to avoid the branches and LHSs that occur when dequantising 8x4 sections of the image
|
|
// instead doing the full idct/dequantising regardless of whether there is only 1 coefficient to perform idct on.
|
|
// this appears to be slower in most cases, hence we're sticking with the version above.
|
|
idct_CacheLines<8>(dst, stride);
|
|
vp8_idct_dequant_full_4x_vecops(q, dq, dst, stride);
|
|
idct_CacheLines<4>(dst + 8*stride, stride);
|
|
vp8_idct_dequant_full_4x_vecops(q+64, dq, dst + 4*stride, stride);
|
|
idct_CacheLines<4>(dst + 12*stride, stride);
|
|
vp8_idct_dequant_full_4x_vecops(q+128, dq, dst + 8*stride, stride);
|
|
vp8_idct_dequant_full_4x_vecops(q+192, dq, dst + 12*stride, stride);
|
|
#endif
|
|
// PIXEndNamedEvent();
|
|
#if DEBUG_DEQUANT_IDCT
|
|
{
|
|
int error_row = -1;
|
|
error_row = idctdebug_compare_block<4, 4>(temp_q, dq, temp_buffer, temp_pitch, orig_eobs, orig_dst, stride);
|
|
if (error_row != -1)
|
|
{
|
|
idctdebug_process_single_row<4, 4>(error_row, temp_q2, dq, orig_eobs, temp_buffer2, temp_pitch);
|
|
}
|
|
}
|
|
#endif
|
|
}
|
|
|
|
extern "C"
|
|
void vp8_dequant_idct_add_uv_block_vecops(
|
|
short* __restrict q
|
|
, short* __restrict dq
|
|
, unsigned char* __restrict dstu
|
|
, unsigned char* __restrict dstv
|
|
, int stride
|
|
, char* __restrict eobs)
|
|
{
|
|
#if DEBUG_DEQUANT_IDCT
|
|
unsigned char* orig_dstu = dstu;
|
|
unsigned char* orig_dstv = dstv;
|
|
int temp_pitch = 8;
|
|
|
|
unsigned char temp_bufferu[64];
|
|
unsigned char temp_bufferv[64];
|
|
short temp_q[128];
|
|
duplicate_buffer(8, 8, dstu, stride, temp_bufferu, temp_pitch);
|
|
duplicate_buffer(8, 8, dstv, stride, temp_bufferv, temp_pitch);
|
|
duplicate_buffer(sizeof(short)*128, 1, (unsigned char*)q, sizeof(short)*128, (unsigned char*)temp_q, sizeof(short)*128);
|
|
|
|
unsigned char temp_bufferu2[64];
|
|
unsigned char temp_bufferv2[64];
|
|
short temp_q2[128];
|
|
duplicate_buffer(8, 8, dstu, stride, temp_bufferu2, temp_pitch);
|
|
duplicate_buffer(8, 8, dstv, stride, temp_bufferv2, temp_pitch);
|
|
duplicate_buffer(sizeof(short)*128, 1, (unsigned char*)q, sizeof(short)*128, (unsigned char*)temp_q2, sizeof(short)*128);
|
|
#endif
|
|
|
|
// PIXBeginNamedEvent(0, "Dequantize IDCT add UV block");
|
|
idct_CacheLines<8>(dstu, stride);
|
|
if (((short *)(eobs))[0])
|
|
{
|
|
if (((short *)(eobs))[0] & 0xfefe)
|
|
vp8_idct_dequant_full_2x_vecops(q, dq, dstu, stride);
|
|
else
|
|
vp8_idct_dequant_0_2x_vecops(q, dq, dstu, stride);
|
|
}
|
|
q += 32;
|
|
dstu += stride*4;
|
|
|
|
idct_CacheLines<4>(dstv, stride);
|
|
if (((short *)(eobs))[1])
|
|
{
|
|
if (((short *)(eobs))[1] & 0xfefe)
|
|
vp8_idct_dequant_full_2x_vecops(q, dq, dstu, stride);
|
|
else
|
|
vp8_idct_dequant_0_2x_vecops(q, dq, dstu, stride);
|
|
}
|
|
q += 32;
|
|
|
|
idct_CacheLines<4>(dstv + 4*stride, stride);
|
|
if (((short *)(eobs))[2])
|
|
{
|
|
if (((short *)(eobs))[2] & 0xfefe)
|
|
vp8_idct_dequant_full_2x_vecops(q, dq, dstv, stride);
|
|
else
|
|
vp8_idct_dequant_0_2x_vecops(q, dq, dstv, stride);
|
|
}
|
|
q += 32;
|
|
dstv += stride*4;
|
|
|
|
if (((short *)(eobs))[3])
|
|
{
|
|
if (((short *)(eobs))[3] & 0xfefe)
|
|
vp8_idct_dequant_full_2x_vecops(q, dq, dstv, stride);
|
|
else
|
|
vp8_idct_dequant_0_2x_vecops(q, dq, dstv, stride);
|
|
}
|
|
|
|
// PIXEndNamedEvent();
|
|
#if DEBUG_DEQUANT_IDCT
|
|
|
|
int error_row = -1;
|
|
|
|
error_row = idctdebug_compare_block<2, 2>(temp_q, dq, temp_bufferu, temp_pitch, eobs, orig_dstu, stride);
|
|
if (error_row != -1)
|
|
{
|
|
idctdebug_process_single_row<2, 2>(error_row, temp_q2, dq, eobs, temp_bufferu2, temp_pitch);
|
|
}
|
|
error_row = idctdebug_compare_block<2, 2>(temp_q+64, dq, temp_bufferv, temp_pitch, eobs+4, orig_dstv, stride);
|
|
if (error_row != -1)
|
|
{
|
|
idctdebug_process_single_row<2, 2>(error_row, temp_q2+64, dq, eobs+4, temp_bufferv2, temp_pitch);
|
|
}
|
|
#endif
|
|
}
|
|
|
|
#endif //defined(VECOPS_ENABLED)
|