JD2022-TU1/main/extern/libvpx/custom/vp8/common/generic/dequantize_vecops.cpp

763 lines
24 KiB
C++

/*
* Copyright (c) 2010 The WebM project authors. All Rights Reserved.
*
* Use of this source code is governed by a BSD-style license
* that can be found in the LICENSE file in the root of the source
* tree. An additional intellectual property rights grant can be found
* in the file PATENTS. All contributing project authors may
* be found in the AUTHORS file in the root of the source tree.
*/
#include "vpx_config.h"
#include "vp8/common/optimisation_vecops.h"
#if defined(VECOPS_ENABLED)
#include <assert.h>
#include "vp8/common/optimisation_profiling.h"
#define DEBUG_DEQUANT_IDCT 0
#if DEBUG_DEQUANT_IDCT
#include "vp8/common/optimisation_debug.h"
#endif
DECLARE_ALIGNED(32, const float, k_sincosPi8Sqrt2[8]) =
{0.5411987f, 0.5411987f, 0.5411987f, 0.5411987f, //sin // use values equal to the integer approximations
1.3065643f, 1.3065643f, 1.3065643f, 1.3065643f}; //cos // rather than the exact ones.
template <int count>
VPX_FORCEINLINE static void idct_CacheLines(const void* __restrict pAddr, int iStride)
{
for (int i=0; i<count; ++i)
{
CacheTouch(pAddr, i*iStride);
}
}
VPX_FORCEINLINE static void idct_LoadAndUnpack16I16s(
short* __restrict pIn,
v128i_t* __restrict pvOut)
{
pvOut[1] = VecLoadAlignedI32(pIn, 0);
pvOut[3] = VecLoadAlignedI32(pIn, 16);
pvOut[0] = VecUnpackLoSignedHalf(pvOut[1]);
pvOut[1] = VecUnpackHiSignedHalf(pvOut[1]);
pvOut[2] = VecUnpackLoSignedHalf(pvOut[3]);
pvOut[3] = VecUnpackHiSignedHalf(pvOut[3]);
}
VPX_FORCEINLINE static void idct_LoadUnpackAndConvert16I16sToF32s(
short* __restrict pIn,
v128f_t* __restrict pvOut
)
{
v128i_t vI32[4];
idct_LoadAndUnpack16I16s(pIn, vI32);
pvOut[0] = VecConvertI32ToF32(vI32[0]);
pvOut[1] = VecConvertI32ToF32(vI32[1]);
pvOut[2] = VecConvertI32ToF32(vI32[2]);
pvOut[3] = VecConvertI32ToF32(vI32[3]);
}
VPX_FORCEINLINE static v128i_t idct_CalcQDQ(
v128i_t vQ,
v128f_t vDQ
)
{
v128f_t vfQ;
vfQ = VecConvertI32ToF32(vQ);
vfQ = VecMulFloat(vfQ, vDQ);
return VecConvertF32ToI32(vfQ);
}
VPX_FORCEINLINE static void idct_Calculate(v128i_t* __restrict vInOutQDQ)
{
v128i_t vP, vQ, vR, vS;
v128f_t vBs, vBc, vDs, vDc;
v128f_t vSinFactor = VecLoadAlignedF32(k_sincosPi8Sqrt2, 0);
v128f_t vCosFactor = VecLoadAlignedF32(k_sincosPi8Sqrt2, 16);
vBs = VecConvertI32ToF32(vInOutQDQ[1]);
vDs = VecConvertI32ToF32(vInOutQDQ[3]);
vBc = VecMulFloat(vBs, vCosFactor);
vDc = VecMulFloat(vDs, vCosFactor);
vBs = VecMulFloat(vBs, vSinFactor);
vDs = VecMulFloat(vDs, vSinFactor);
vBc = VecRoundFloatNegInfinity(vBc);
vDc = VecRoundFloatNegInfinity(vDc);
vBs = VecRoundFloatNegInfinity(vBs);
vDs = VecRoundFloatNegInfinity(vDs);
vP = VecAddSignedWordSaturate(vInOutQDQ[0], vInOutQDQ[2]);
vQ = VecSubSignedWordSaturate(vInOutQDQ[0], vInOutQDQ[2]);
vR = VecSubSignedWordSaturate(VecConvertF32ToI32(vBs), VecConvertF32ToI32(vDc));
vS = VecAddSignedWordSaturate(VecConvertF32ToI32(vBc), VecConvertF32ToI32(vDs));
vInOutQDQ[0] = VecAddSignedWordSaturate(vP, vS);
vInOutQDQ[1] = VecAddSignedWordSaturate(vQ, vR);
vInOutQDQ[2] = VecSubSignedWordSaturate(vQ, vR);
vInOutQDQ[3] = VecSubSignedWordSaturate(vP, vS);
}
VPX_FORCEINLINE static void idct_Transpose4x4(v128i_t* __restrict pvRows)
{
v128i_t vTemp[4];
vTemp[0] = VecInterleaveLoWord(pvRows[0], pvRows[2]);
vTemp[1] = VecInterleaveHiWord(pvRows[0], pvRows[2]);
vTemp[2] = VecInterleaveLoWord(pvRows[1], pvRows[3]);
vTemp[3] = VecInterleaveHiWord(pvRows[1], pvRows[3]);
pvRows[0] = VecInterleaveLoWord( vTemp[0], vTemp[2]);
pvRows[1] = VecInterleaveHiWord( vTemp[0], vTemp[2]);
pvRows[2] = VecInterleaveLoWord( vTemp[1], vTemp[3]);
pvRows[3] = VecInterleaveHiWord( vTemp[1], vTemp[3]);
}
VPX_FORCEINLINE static void idct_AddResultLo(unsigned char* __restrict pDst, v128i_t vVal)
{
v128i_t vCur, vLo;
vCur = VecLoadUnalignedI32(pDst, 0);
vLo = VecUnpackLoUnsignedByte(vCur);
vLo = VecAddSignedHalfSaturate(vLo, vVal);
vCur = VecPackSignedHalfToUnsignedByteSaturate(vLo, VecUnpackHiUnsignedByte(vCur));
VecStoreUnalignedI32(vCur, pDst, 0);
}
VPX_FORCEINLINE static void idct_AddResult(unsigned char* pDst, v128i_t vLoR, v128i_t vHiR)
{
v128i_t vCur, vLo, vHi;
vCur = VecLoadUnalignedI32(pDst, 0);
vLo = VecUnpackLoUnsignedByte(vCur);
vHi = VecUnpackHiUnsignedByte(vCur);
vLo = VecAddSignedHalfSaturate(vLo, vLoR);
vHi = VecAddSignedHalfSaturate(vHi, vHiR);
vCur = VecPackSignedHalfToUnsignedByteSaturate(vLo, vHi);
VecStoreUnalignedI32(vCur, pDst, 0);
}
VPX_FORCEINLINE void idct_Calc4x4(
short* __restrict pQ,
v128f_t* __restrict pvDQ,
v128i_t* __restrict pvOut)
{
v128i_t vQ[4], vQDQ[4];
idct_LoadAndUnpack16I16s(pQ, vQ);
vQDQ[0] = idct_CalcQDQ(vQ[0], pvDQ[0]);
vQDQ[1] = idct_CalcQDQ(vQ[1], pvDQ[1]);
vQDQ[2] = idct_CalcQDQ(vQ[2], pvDQ[2]);
vQDQ[3] = idct_CalcQDQ(vQ[3], pvDQ[3]);
idct_Calculate(vQDQ);
idct_Transpose4x4(vQDQ);
idct_Calculate(vQDQ);
idct_Transpose4x4(vQDQ);
pvOut[0] = VecShiftRightArithmeticWordImmediate<3>(VecAddSignedWordSaturate(vQDQ[0], VecSplatImmediateWord<4>()));
pvOut[1] = VecShiftRightArithmeticWordImmediate<3>(VecAddSignedWordSaturate(vQDQ[1], VecSplatImmediateWord<4>()));
pvOut[2] = VecShiftRightArithmeticWordImmediate<3>(VecAddSignedWordSaturate(vQDQ[2], VecSplatImmediateWord<4>()));
pvOut[3] = VecShiftRightArithmeticWordImmediate<3>(VecAddSignedWordSaturate(vQDQ[3], VecSplatImmediateWord<4>()));
}
void vp8_idct_dequant_0_1x_vecops(
short* __restrict q
, short* __restrict dq
, unsigned char* __restrict dst
, int dst_stride)
{
PRF_Scoped("vp8_idct_dequant_0_1x_vecops");
// the block will have ((q[0] * dq[0]) + 4) >> 3 added to it
v128i_t vRes;
{
v128i_t vQ, vDQ;
v128f_t vQf, vDQf;
vQ = VecLoadAlignedI32(q, 0);
vDQ = VecLoadUnalignedI32(dq, 0);
vQ = VecUnpackLoSignedHalf(vQ);
vDQ = VecUnpackLoSignedHalf(vDQ);
vQf = VecConvertI32ToF32(vQ);
vDQf = VecConvertI32ToF32(vDQ);
vQf = VecMulFloat(vQf, vDQf);
vQ = VecConvertF32ToI32(vQf);
vQ = VecAddSignedWordSaturate(vQ, VecSplatImmediateWord<4>());
vQ = VecShiftRightArithmeticWordImmediate<3>(vQ);
vQ = VecShuffleWord<0,0,0,0>(vQ);
vRes = VecPackSignedWordToSignedHalfSaturate(vQ, VecSplatImmediateWord<0>());
}
idct_AddResultLo(dst , vRes);
idct_AddResultLo(dst + dst_stride, vRes);
idct_AddResultLo(dst + 2*dst_stride, vRes);
idct_AddResultLo(dst + 3*dst_stride, vRes);
q[0] = 0;
}
void vp8_idct_dequant_0_2x_vecops(
short* __restrict q
, short* __restrict dq
, unsigned char* __restrict dst
, int dst_stride)
{
PRF_Scoped("vp8_idct_dequant_0_2x_vecops");
// calculate the values to add to the two 4x4 blocks at dst and dst+4.
// the first block will have ((q[0] * dq[0]) + 4) >> 3 added to it
// the second block will have ((q[16] * dq[0]) + 4) >> 3 added to it
v128i_t vRes;
{
v128i_t vQ[2], vDQ;
v128f_t vQf[2], vDQf;
vQ[0] = VecLoadAlignedI32(q, 0);
vQ[1] = VecLoadAlignedI32(q, 32);
vDQ = VecLoadUnalignedI32(dq, 0);
vQ[0] = VecUnpackLoSignedHalf(vQ[0]);
vQ[1] = VecUnpackLoSignedHalf(vQ[1]);
vDQ = VecUnpackLoSignedHalf(vDQ);
vQf[0] = VecConvertI32ToF32(vQ[0]);
vQf[1] = VecConvertI32ToF32(vQ[1]);
vDQf = VecConvertI32ToF32(vDQ);
vQf[0] = VecMulFloat(vQf[0], vDQf);
vQf[1] = VecMulFloat(vQf[1], vDQf);
vQ[0] = VecConvertF32ToI32(vQf[0]);
vQ[1] = VecConvertF32ToI32(vQf[1]);
vQ[0] = VecAddSignedWordSaturate(vQ[0], VecSplatImmediateWord<4>());
vQ[1] = VecAddSignedWordSaturate(vQ[1], VecSplatImmediateWord<4>());
vQ[0] = VecShiftRightArithmeticWordImmediate<3>(vQ[0]);
vQ[1] = VecShiftRightArithmeticWordImmediate<3>(vQ[1]);
vQ[0] = VecShuffleWord<0,0,0,0>(vQ[0]);
vQ[1] = VecShuffleWord<0,0,0,0>(vQ[1]);
vRes = VecPackSignedWordToSignedHalfSaturate(vQ[0], vQ[1]);
}
idct_AddResultLo(dst , vRes);
idct_AddResultLo(dst + dst_stride, vRes);
idct_AddResultLo(dst + 2*dst_stride, vRes);
idct_AddResultLo(dst + 3*dst_stride, vRes);
q[0] = 0;
q[16] = 0;
}
void vp8_idct_dequant_0_4x_vecops(
short* __restrict q
, short* __restrict dq
, unsigned char* __restrict dst
, int dst_stride)
{
// calculate the values to add to the two 4x4 blocks at dst and dst+4.
// the first block will have ((q[0] * dq[0]) + 4) >> 3 added to it
// the second block will have ((q[16] * dq[0]) + 4) >> 3 added to it
// the third block will have ((q[32] * dq[0]) + 4) >> 3 added to it
// the fourth block will have ((q[48] * dq[0]) + 4) >> 3 added to it
v128i_t vResLo, vResHi;
{
v128i_t vQ[4], vDQ;
v128f_t vQf[4], vDQf;
vQ[0] = VecLoadAlignedI32(q, 0);
vQ[1] = VecLoadAlignedI32(q, 32);
vQ[2] = VecLoadAlignedI32(q, 64);
vQ[3] = VecLoadAlignedI32(q, 96);
vDQ = VecLoadUnalignedI32(dq, 0);
vQ[0] = VecUnpackLoSignedHalf(vQ[0]);
vQ[1] = VecUnpackLoSignedHalf(vQ[1]);
vQ[2] = VecUnpackLoSignedHalf(vQ[2]);
vQ[3] = VecUnpackLoSignedHalf(vQ[3]);
vDQ = VecUnpackLoSignedHalf(vDQ);
vQf[0] = VecConvertI32ToF32(vQ[0]);
vQf[1] = VecConvertI32ToF32(vQ[1]);
vQf[2] = VecConvertI32ToF32(vQ[2]);
vQf[3] = VecConvertI32ToF32(vQ[3]);
vDQf = VecConvertI32ToF32(vDQ);
vQf[0] = VecMulFloat(vQf[0], vDQf);
vQf[1] = VecMulFloat(vQf[1], vDQf);
vQf[2] = VecMulFloat(vQf[2], vDQf);
vQf[3] = VecMulFloat(vQf[3], vDQf);
vQ[0] = VecConvertF32ToI32(vQf[0]);
vQ[1] = VecConvertF32ToI32(vQf[1]);
vQ[2] = VecConvertF32ToI32(vQf[2]);
vQ[3] = VecConvertF32ToI32(vQf[3]);
vQ[0] = VecAddSignedWordSaturate(vQ[0], VecSplatImmediateWord<4>());
vQ[1] = VecAddSignedWordSaturate(vQ[1], VecSplatImmediateWord<4>());
vQ[2] = VecAddSignedWordSaturate(vQ[2], VecSplatImmediateWord<4>());
vQ[3] = VecAddSignedWordSaturate(vQ[3], VecSplatImmediateWord<4>());
vQ[0] = VecShiftRightArithmeticWordImmediate<3>(vQ[0]);
vQ[1] = VecShiftRightArithmeticWordImmediate<3>(vQ[1]);
vQ[2] = VecShiftRightArithmeticWordImmediate<3>(vQ[2]);
vQ[3] = VecShiftRightArithmeticWordImmediate<3>(vQ[3]);
vQ[0] = VecShuffleWord<0,0,0,0>(vQ[0]);
vQ[1] = VecShuffleWord<0,0,0,0>(vQ[1]);
vQ[2] = VecShuffleWord<0,0,0,0>(vQ[2]);
vQ[3] = VecShuffleWord<0,0,0,0>(vQ[3]);
vResLo = VecPackSignedWordToSignedHalfSaturate(vQ[0], vQ[1]);
vResHi = VecPackSignedWordToSignedHalfSaturate(vQ[2], vQ[3]);
}
idct_AddResult(dst , vResLo, vResHi);
idct_AddResult(dst + dst_stride, vResLo, vResHi);
idct_AddResult(dst + 2*dst_stride, vResLo, vResHi);
idct_AddResult(dst + 3*dst_stride, vResLo, vResHi);
q[0] = 0;
q[16] = 0;
q[32] = 0;
q[48] = 0;
}
void vp8_idct_dequant_full_2x_vecops(
short* __restrict q
, short* __restrict dq
, unsigned char* __restrict dst
, int dst_stride)
{
PRF_Scoped("vp8_idct_dequant_full_2x_vecops");
//
// This function dequantises 32 q values (16 per 4x4 output block),
// performs the IDCT function and adds the result to the current values in dst.
// Each 16 values of Q are multiplied by the 16 values of DQ.
//
v128i_t vResults[8];
v128f_t vDQ[4];
idct_LoadUnpackAndConvert16I16sToF32s(dq, vDQ);
idct_Calc4x4(q , vDQ, &vResults[0]);
idct_Calc4x4(q+16, vDQ, &vResults[4]);
vResults[0] = VecPackSignedWordToSignedHalfSaturate(vResults[0], vResults[4]);
vResults[1] = VecPackSignedWordToSignedHalfSaturate(vResults[1], vResults[5]);
vResults[2] = VecPackSignedWordToSignedHalfSaturate(vResults[2], vResults[6]);
vResults[3] = VecPackSignedWordToSignedHalfSaturate(vResults[3], vResults[7]);
idct_AddResultLo(dst , vResults[0]);
idct_AddResultLo(dst + dst_stride, vResults[1]);
idct_AddResultLo(dst + 2*dst_stride, vResults[2]);
idct_AddResultLo(dst + 3*dst_stride, vResults[3]);
v128i_t vZero = VecSplatImmediateWord<0>();
VecStoreAlignedI32(vZero, q, 0);
VecStoreAlignedI32(vZero, q, 0x10);
VecStoreAlignedI32(vZero, q, 0x20);
VecStoreAlignedI32(vZero, q, 0x30);
}
void vp8_idct_dequant_full_4x_vecops(
short* __restrict q
, short* __restrict dq
, unsigned char* __restrict dst
, int dst_stride)
{
//
// This function dequantises/performs the idct on 64 q values (16 per 4x4 output block, i.e. a whole block high row of a macroblock)
// and adds the result to the current values in dst.
// Each 16 values of Q are multiplied by the 16 values of DQ.
// While it is more efficient than the 2x version (as it writes out the data more efficiently)
// it is situational - when the whole row of blocks have more than one dct coefficient.
// Also, it might be beneficial to provide a 4x version of the dequant_0.
//
v128i_t vResults[16];
v128f_t vDQ[4];
idct_LoadUnpackAndConvert16I16sToF32s(dq, vDQ);
idct_Calc4x4(q , vDQ, &vResults[ 0]);
idct_Calc4x4(q+16, vDQ, &vResults[ 4]);
idct_Calc4x4(q+32, vDQ, &vResults[ 8]);
idct_Calc4x4(q+48, vDQ, &vResults[12]);
vResults[0] = VecPackSignedWordToSignedHalfSaturate(vResults[ 0], vResults[ 4]);
vResults[1] = VecPackSignedWordToSignedHalfSaturate(vResults[ 1], vResults[ 5]);
vResults[2] = VecPackSignedWordToSignedHalfSaturate(vResults[ 2], vResults[ 6]);
vResults[3] = VecPackSignedWordToSignedHalfSaturate(vResults[ 3], vResults[ 7]);
vResults[4] = VecPackSignedWordToSignedHalfSaturate(vResults[ 8], vResults[12]);
vResults[5] = VecPackSignedWordToSignedHalfSaturate(vResults[ 9], vResults[13]);
vResults[6] = VecPackSignedWordToSignedHalfSaturate(vResults[10], vResults[14]);
vResults[7] = VecPackSignedWordToSignedHalfSaturate(vResults[11], vResults[15]);
idct_AddResult(dst , vResults[0], vResults[4]);
idct_AddResult(dst + dst_stride, vResults[1], vResults[5]);
idct_AddResult(dst + 2*dst_stride, vResults[2], vResults[6]);
idct_AddResult(dst + 3*dst_stride, vResults[3], vResults[7]);
v128i_t vZero = VecSplatImmediateWord<0>();
VecStoreAlignedI32(vZero, q, 0);
VecStoreAlignedI32(vZero, q, 0x10);
VecStoreAlignedI32(vZero, q, 0x20);
VecStoreAlignedI32(vZero, q, 0x30);
VecStoreAlignedI32(vZero, q, 0x40);
VecStoreAlignedI32(vZero, q, 0x50);
VecStoreAlignedI32(vZero, q, 0x60);
VecStoreAlignedI32(vZero, q, 0x70);
}
#if DEBUG_DEQUANT_IDCT
extern "C" void vp8_dequant_idct_add_c(short *input, short *dq, unsigned char *dest, int stride);
extern "C" void vp8_dc_only_idct_add_c(short input_dc, unsigned char * pred, int pred_stride, unsigned char *dst_ptr, int dst_stride);
template <int kCols, int kRows>
int idctdebug_compare_block(short* q, short* dq, unsigned char* dst, int stride, const char* eobs, const unsigned char* to_compare, int to_compare_stride)
{
int error_row = -1;
for (int i = 0; i < kRows; i++)
{
for (int j = 0; j < kCols; j++)
{
if (*eobs++ > 1)
{
vp8_dequant_idct_add_c (q, dq, dst, stride);
}
else
{
vp8_dc_only_idct_add_c (q[0]*dq[0], dst, stride, dst, stride);
((int *)q)[0] = 0;
}
if (!compare_b(dst, stride, to_compare, to_compare_stride))
{
error_row = i;
}
q += 16;
dst += 4;
to_compare += 4;
}
dst += 4*stride - (kCols*4);
to_compare += 4*to_compare_stride - (kCols*4);
}
return error_row;
}
template <int kCols, int kRows>
void idctdebug_process_single_row(int error_row, short* q, short* dq, char* eobs, unsigned char* dst, int dst_stride )
{
int qofst = error_row*16*kCols;
int dstofst = error_row*dst_stride*4;
eobs += error_row*kRows;
for (int i=0; i<kCols; i+=2)
{
if (((short *)(eobs))[i/2])
{
if (((short *)(eobs))[i/2] & 0xfefe)
{
vp8_idct_dequant_full_2x_vecops(q+qofst, dq, dst+dstofst, dst_stride);
}
else
{
vp8_idct_dequant_0_2x_vecops(q+qofst, dq, dst+dstofst, dst_stride);
}
}
qofst += 32;
dstofst += 8;
}
}
#endif
extern "C"
void vp8_dequant_idct_add_vecops(
short* __restrict q
, short* __restrict dq
, unsigned char* __restrict dst
, int dst_stride)
{
PRF_Scoped("vp8_dequant_idct_add_vecops");
v128i_t vResults[4];
v128f_t vDQ[4];
idct_LoadUnpackAndConvert16I16sToF32s(dq, vDQ);
idct_Calc4x4(q, vDQ, vResults);
vResults[0] = VecPackSignedWordToSignedHalfSaturate(vResults[0], VecSplatImmediateWord<0>());
vResults[1] = VecPackSignedWordToSignedHalfSaturate(vResults[1], VecSplatImmediateWord<0>());
vResults[2] = VecPackSignedWordToSignedHalfSaturate(vResults[2], VecSplatImmediateWord<0>());
vResults[3] = VecPackSignedWordToSignedHalfSaturate(vResults[3], VecSplatImmediateWord<0>());
idct_AddResultLo(dst , vResults[0]);
idct_AddResultLo(dst + dst_stride, vResults[1]);
idct_AddResultLo(dst + 2*dst_stride, vResults[2]);
idct_AddResultLo(dst + 3*dst_stride, vResults[3]);
v128i_t vZero = VecSplatImmediateWord<0>();
VecStoreAlignedI32(vZero, q, 0);
VecStoreAlignedI32(vZero, q, 0x10);
}
extern "C"
void vp8_dequant_idct_add_y_block_vecops(
short* __restrict q
, short* __restrict dq
, unsigned char* __restrict dst
, int stride
, char* eobs)
{
#if DEBUG_DEQUANT_IDCT
unsigned char* orig_dst = dst;
char* orig_eobs = eobs;
char* orig_eobs2 = eobs;
int temp_pitch = 16;
unsigned char temp_buffer[256];
short temp_q[256];
duplicate_buffer(16, 16, dst, stride, temp_buffer, temp_pitch);
duplicate_buffer(sizeof(short)*256, 1, (unsigned char*)q, sizeof(short)*256, (unsigned char*)temp_q, sizeof(short)*256);
unsigned char temp_buffer2[256];
short temp_q2[256];
duplicate_buffer(16, 16, dst, stride, temp_buffer2, temp_pitch);
duplicate_buffer(sizeof(short)*256, 1, (unsigned char*)q, sizeof(short)*256, (unsigned char*)temp_q2, sizeof(short)*256);
#endif
// PIXBeginNamedEvent(0, "Dequantize IDCT add Y block");
#if 0
// effectively what the section below does, but without the cache prefetching
for (int i = 0; i < 4; i++)
{
if (((short *)(eobs))[0])
{
if (((short *)(eobs))[0] & 0xfefe)
{
vp8_idct_dequant_full_2x_vecops(q, dq, dst, stride);
}
else
{
vp8_idct_dequant_0_2x_vecops(q, dq, dst, stride);
}
}
if (((short *)(eobs))[1])
{
if (((short *)(eobs))[1] & 0xfefe)
{
vp8_idct_dequant_full_2x_vecops(q+32, dq, dst+8, stride);
}
else
{
vp8_idct_dequant_0_2x_vecops(q+32, dq, dst+8, stride);
}
}
q += 64;
dst += stride*4;
eobs += 4;
}
#elif 1
idct_CacheLines<8>(dst, stride);
if (((short *)(eobs))[0])
{
if (((short *)(eobs))[0] & 0xfefe)
{
vp8_idct_dequant_full_2x_vecops(q, dq, dst, stride);
}
else
{
vp8_idct_dequant_0_2x_vecops(q, dq, dst, stride);
}
}
if (((short *)(eobs))[1])
{
if (((short *)(eobs))[1] & 0xfefe)
{
vp8_idct_dequant_full_2x_vecops(q+32, dq, dst+8, stride);
}
else
{
vp8_idct_dequant_0_2x_vecops(q+32, dq, dst+8, stride);
}
}
q += 64;
dst += stride*4;
eobs += 4;
idct_CacheLines<4>(dst + 8*stride, stride);
if (((short *)(eobs))[0])
{
if (((short *)(eobs))[0] & 0xfefe)
{
vp8_idct_dequant_full_2x_vecops(q, dq, dst, stride);
}
else
{
vp8_idct_dequant_0_2x_vecops(q, dq, dst, stride);
}
}
if (((short *)(eobs))[1])
{
if (((short *)(eobs))[1] & 0xfefe)
{
vp8_idct_dequant_full_2x_vecops(q+32, dq, dst+8, stride);
}
else
{
vp8_idct_dequant_0_2x_vecops(q+32, dq, dst+8, stride);
}
}
q += 64;
dst += stride*4;
eobs += 4;
idct_CacheLines<4>(dst + 8*stride, stride);
if (((short *)(eobs))[0])
{
if (((short *)(eobs))[0] & 0xfefe)
{
vp8_idct_dequant_full_2x_vecops(q, dq, dst, stride);
}
else
{
vp8_idct_dequant_0_2x_vecops(q, dq, dst, stride);
}
}
if (((short *)(eobs))[1])
{
if (((short *)(eobs))[1] & 0xfefe)
{
vp8_idct_dequant_full_2x_vecops(q+32, dq, dst+8, stride);
}
else
{
vp8_idct_dequant_0_2x_vecops(q+32, dq, dst+8, stride);
}
}
q += 64;
dst += stride*4;
eobs += 4;
if (((short *)(eobs))[0])
{
if (((short *)(eobs))[0] & 0xfefe)
{
vp8_idct_dequant_full_2x_vecops(q, dq, dst, stride);
}
else
{
vp8_idct_dequant_0_2x_vecops(q, dq, dst, stride);
}
}
if (((short *)(eobs))[1])
{
if (((short *)(eobs))[1] & 0xfefe)
{
vp8_idct_dequant_full_2x_vecops(q+32, dq, dst+8, stride);
}
else
{
vp8_idct_dequant_0_2x_vecops(q+32, dq, dst+8, stride);
}
}
#else
// the intention behind this approach is to avoid the branches and LHSs that occur when dequantising 8x4 sections of the image
// instead doing the full idct/dequantising regardless of whether there is only 1 coefficient to perform idct on.
// this appears to be slower in most cases, hence we're sticking with the version above.
idct_CacheLines<8>(dst, stride);
vp8_idct_dequant_full_4x_vecops(q, dq, dst, stride);
idct_CacheLines<4>(dst + 8*stride, stride);
vp8_idct_dequant_full_4x_vecops(q+64, dq, dst + 4*stride, stride);
idct_CacheLines<4>(dst + 12*stride, stride);
vp8_idct_dequant_full_4x_vecops(q+128, dq, dst + 8*stride, stride);
vp8_idct_dequant_full_4x_vecops(q+192, dq, dst + 12*stride, stride);
#endif
// PIXEndNamedEvent();
#if DEBUG_DEQUANT_IDCT
{
int error_row = -1;
error_row = idctdebug_compare_block<4, 4>(temp_q, dq, temp_buffer, temp_pitch, orig_eobs, orig_dst, stride);
if (error_row != -1)
{
idctdebug_process_single_row<4, 4>(error_row, temp_q2, dq, orig_eobs, temp_buffer2, temp_pitch);
}
}
#endif
}
extern "C"
void vp8_dequant_idct_add_uv_block_vecops(
short* __restrict q
, short* __restrict dq
, unsigned char* __restrict dstu
, unsigned char* __restrict dstv
, int stride
, char* __restrict eobs)
{
#if DEBUG_DEQUANT_IDCT
unsigned char* orig_dstu = dstu;
unsigned char* orig_dstv = dstv;
int temp_pitch = 8;
unsigned char temp_bufferu[64];
unsigned char temp_bufferv[64];
short temp_q[128];
duplicate_buffer(8, 8, dstu, stride, temp_bufferu, temp_pitch);
duplicate_buffer(8, 8, dstv, stride, temp_bufferv, temp_pitch);
duplicate_buffer(sizeof(short)*128, 1, (unsigned char*)q, sizeof(short)*128, (unsigned char*)temp_q, sizeof(short)*128);
unsigned char temp_bufferu2[64];
unsigned char temp_bufferv2[64];
short temp_q2[128];
duplicate_buffer(8, 8, dstu, stride, temp_bufferu2, temp_pitch);
duplicate_buffer(8, 8, dstv, stride, temp_bufferv2, temp_pitch);
duplicate_buffer(sizeof(short)*128, 1, (unsigned char*)q, sizeof(short)*128, (unsigned char*)temp_q2, sizeof(short)*128);
#endif
// PIXBeginNamedEvent(0, "Dequantize IDCT add UV block");
idct_CacheLines<8>(dstu, stride);
if (((short *)(eobs))[0])
{
if (((short *)(eobs))[0] & 0xfefe)
vp8_idct_dequant_full_2x_vecops(q, dq, dstu, stride);
else
vp8_idct_dequant_0_2x_vecops(q, dq, dstu, stride);
}
q += 32;
dstu += stride*4;
idct_CacheLines<4>(dstv, stride);
if (((short *)(eobs))[1])
{
if (((short *)(eobs))[1] & 0xfefe)
vp8_idct_dequant_full_2x_vecops(q, dq, dstu, stride);
else
vp8_idct_dequant_0_2x_vecops(q, dq, dstu, stride);
}
q += 32;
idct_CacheLines<4>(dstv + 4*stride, stride);
if (((short *)(eobs))[2])
{
if (((short *)(eobs))[2] & 0xfefe)
vp8_idct_dequant_full_2x_vecops(q, dq, dstv, stride);
else
vp8_idct_dequant_0_2x_vecops(q, dq, dstv, stride);
}
q += 32;
dstv += stride*4;
if (((short *)(eobs))[3])
{
if (((short *)(eobs))[3] & 0xfefe)
vp8_idct_dequant_full_2x_vecops(q, dq, dstv, stride);
else
vp8_idct_dequant_0_2x_vecops(q, dq, dstv, stride);
}
// PIXEndNamedEvent();
#if DEBUG_DEQUANT_IDCT
int error_row = -1;
error_row = idctdebug_compare_block<2, 2>(temp_q, dq, temp_bufferu, temp_pitch, eobs, orig_dstu, stride);
if (error_row != -1)
{
idctdebug_process_single_row<2, 2>(error_row, temp_q2, dq, eobs, temp_bufferu2, temp_pitch);
}
error_row = idctdebug_compare_block<2, 2>(temp_q+64, dq, temp_bufferv, temp_pitch, eobs+4, orig_dstv, stride);
if (error_row != -1)
{
idctdebug_process_single_row<2, 2>(error_row, temp_q2+64, dq, eobs+4, temp_bufferv2, temp_pitch);
}
#endif
}
#endif //defined(VECOPS_ENABLED)