/* * Copyright (c) 2010 The WebM project authors. All Rights Reserved. * * Use of this source code is governed by a BSD-style license * that can be found in the LICENSE file in the root of the source * tree. An additional intellectual property rights grant can be found * in the file PATENTS. All contributing project authors may * be found in the AUTHORS file in the root of the source tree. */ #include "vpx_rtcd.h" #ifdef RVL # define POWERPC_IDCT4X4_OPTIMS #endif /**************************************************************************** * Notes: * * This implementation makes use of 16 bit fixed point verio of two multiply * constants: * 1. sqrt(2) * cos (pi/8) * 2. sqrt(2) * sin (pi/8) * Becuase the first constant is bigger than 1, to maintain the same 16 bit * fixed point precision as the second one, we use a trick of * x * a = x + x*(a-1) * so * x * sqrt(2) * cos (pi/8) = x + x * (sqrt(2) *cos(pi/8)-1). **************************************************************************/ static const int cospi8sqrt2minus1 = 20091; static const int sinpi8sqrt2 = 35468; #ifdef POWERPC_IDCT4X4_OPTIMS /*#define IDCT4X4_LOAD_TWO_PACKED_SHORTS(index0, index1) \ asm \ { \ lwzu i##index0, 4(input); \ clrlwi i##index1, i##index0, 16; \ srawi i##index0, i##index0, 16; \ extsh i##index1, i##index1; \ }; \*/ #define IDCT4X4_PROCESS_LINE(index0, index1, index2, index3) \ a1 = i##index0 + i##index2; \ b1 = i##index0 - i##index2; \ \ temp1 = (i##index1 * sinpi8sqrt2) >> 16; \ temp2 = i##index3 + ((i##index3 * cospi8sqrt2minus1) >> 16); \ c1 = temp1 - temp2; \ \ temp1 = i##index1 + ((i##index1 * cospi8sqrt2minus1) >> 16); \ temp2 = (i##index3 * sinpi8sqrt2) >> 16; \ d1 = temp1 + temp2; \ \ i##index0 = a1 + d1; \ i##index3 = a1 - d1; \ \ i##index1 = b1 + c1; \ i##index2 = b1 - c1; \ #define IDCT4X4_ADD_PREDICTION_AND_CLAMP(index) \ addi i##index, i##index, 4; \ srawi i##index, i##index, 3; \ srwi prediction, rowPrediction, 24; \ add. i##index, i##index, prediction; \ slwi rowPrediction, rowPrediction, 8; \ bge positive##index; \ li i##index, 0; \ positive##index: \ cmpwi i##index, 0xff; \ blt under255##index; \ li i##index, 0xff; \ under255##index: \ slwi rowOutput, rowOutput, 8; \ or rowOutput, rowOutput, i##index; \ #define IDCT4X4_OUTPUT_LINE(index0, index1, index2, index3) \ asm \ { \ lwz rowPrediction, 0(pred_ptr); \ IDCT4X4_ADD_PREDICTION_AND_CLAMP(index0); \ IDCT4X4_ADD_PREDICTION_AND_CLAMP(index1); \ IDCT4X4_ADD_PREDICTION_AND_CLAMP(index2); \ IDCT4X4_ADD_PREDICTION_AND_CLAMP(index3); \ stw rowOutput, 0(dst_ptr); \ }; \ dst_ptr += dst_stride; \ pred_ptr += pred_stride; \ void vp8_short_idct4x4llm_c(register short *input, register unsigned char *pred_ptr, register int pred_stride, register unsigned char *dst_ptr, register int dst_stride) { register int a1, b1, c1, d1; register int temp1, temp2; register i0, i1, i2, i3; register i4, i5, i6, i7; register i8, i9, i10, i11; register i12, i13, i14, i15; register rowPrediction, prediction; register rowOutput; // this one is not faster, but may be improved to become so /*input -= 2; IDCT4X4_LOAD_TWO_PACKED_SHORTS(0, 1); IDCT4X4_LOAD_TWO_PACKED_SHORTS(2, 3); IDCT4X4_LOAD_TWO_PACKED_SHORTS(4, 5); IDCT4X4_LOAD_TWO_PACKED_SHORTS(6, 7); IDCT4X4_LOAD_TWO_PACKED_SHORTS(8, 9); IDCT4X4_LOAD_TWO_PACKED_SHORTS(10, 11); IDCT4X4_LOAD_TWO_PACKED_SHORTS(12, 13); IDCT4X4_LOAD_TWO_PACKED_SHORTS(14, 15);*/ i0 = input[0]; i1 = input[1]; i2 = input[2]; i3 = input[3]; i4 = input[4]; i5 = input[5]; i6 = input[6]; i7 = input[7]; i8 = input[8]; i9 = input[9]; i10 = input[10]; i11 = input[11]; i12 = input[12]; i13 = input[13]; i14 = input[14]; i15 = input[15]; // columns IDCT4X4_PROCESS_LINE(0, 4, 8, 12); IDCT4X4_PROCESS_LINE(1, 5, 9, 13); IDCT4X4_PROCESS_LINE(2, 6, 10, 14); IDCT4X4_PROCESS_LINE(3, 7, 11, 15); // rows IDCT4X4_PROCESS_LINE(0, 1, 2, 3); IDCT4X4_OUTPUT_LINE (0, 1, 2, 3); IDCT4X4_PROCESS_LINE(4, 5, 6, 7); IDCT4X4_OUTPUT_LINE (4, 5, 6, 7); IDCT4X4_PROCESS_LINE(8, 9, 10, 11); IDCT4X4_OUTPUT_LINE (8, 9, 10, 11); IDCT4X4_PROCESS_LINE(12, 13, 14, 15); IDCT4X4_OUTPUT_LINE (12, 13, 14, 15); } #else // POWERPC_IDCT4X4_OPTIMS void vp8_short_idct4x4llm_c(short *input, unsigned char *pred_ptr, int pred_stride, unsigned char *dst_ptr, int dst_stride) { int i; int r, c; int a1, b1, c1, d1; short output[16]; short *ip = input; short *op = output; int temp1, temp2; int shortpitch = 4; for (i = 0; i < 4; i++) { a1 = ip[0] + ip[8]; b1 = ip[0] - ip[8]; temp1 = (ip[4] * sinpi8sqrt2) >> 16; temp2 = ip[12] + ((ip[12] * cospi8sqrt2minus1) >> 16); c1 = temp1 - temp2; temp1 = ip[4] + ((ip[4] * cospi8sqrt2minus1) >> 16); temp2 = (ip[12] * sinpi8sqrt2) >> 16; d1 = temp1 + temp2; op[shortpitch*0] = a1 + d1; op[shortpitch*3] = a1 - d1; op[shortpitch*1] = b1 + c1; op[shortpitch*2] = b1 - c1; ip++; op++; } ip = output; op = output; for (i = 0; i < 4; i++) { a1 = ip[0] + ip[2]; b1 = ip[0] - ip[2]; temp1 = (ip[1] * sinpi8sqrt2) >> 16; temp2 = ip[3] + ((ip[3] * cospi8sqrt2minus1) >> 16); c1 = temp1 - temp2; temp1 = ip[1] + ((ip[1] * cospi8sqrt2minus1) >> 16); temp2 = (ip[3] * sinpi8sqrt2) >> 16; d1 = temp1 + temp2; op[0] = (a1 + d1 + 4) >> 3; op[3] = (a1 - d1 + 4) >> 3; op[1] = (b1 + c1 + 4) >> 3; op[2] = (b1 - c1 + 4) >> 3; ip += shortpitch; op += shortpitch; } ip = output; for (r = 0; r < 4; r++) { for (c = 0; c < 4; c++) { int a = ip[c] + pred_ptr[c] ; if (a < 0) a = 0; if (a > 255) a = 255; dst_ptr[c] = (unsigned char) a ; } ip += 4; dst_ptr += dst_stride; pred_ptr += pred_stride; } } #endif // POWERPC_IDCT4X4_OPTIMS void vp8_dc_only_idct_add_c(short input_dc, unsigned char *pred_ptr, int pred_stride, unsigned char *dst_ptr, int dst_stride) { int a1 = ((input_dc + 4) >> 3); int r, c; for (r = 0; r < 4; r++) { for (c = 0; c < 4; c++) { int a = a1 + pred_ptr[c] ; if (a < 0) a = 0; if (a > 255) a = 255; dst_ptr[c] = (unsigned char) a ; } dst_ptr += dst_stride; pred_ptr += pred_stride; } } void vp8_short_inv_walsh4x4_c(short *input, short *mb_dqcoeff) { short output[16]; int i; int a1, b1, c1, d1; int a2, b2, c2, d2; short *ip = input; short *op = output; for (i = 0; i < 4; i++) { a1 = ip[0] + ip[12]; b1 = ip[4] + ip[8]; c1 = ip[4] - ip[8]; d1 = ip[0] - ip[12]; op[0] = a1 + b1; op[4] = c1 + d1; op[8] = a1 - b1; op[12] = d1 - c1; ip++; op++; } ip = output; op = output; for (i = 0; i < 4; i++) { a1 = ip[0] + ip[3]; b1 = ip[1] + ip[2]; c1 = ip[1] - ip[2]; d1 = ip[0] - ip[3]; a2 = a1 + b1; b2 = c1 + d1; c2 = a1 - b1; d2 = d1 - c1; op[0] = (a2 + 3) >> 3; op[1] = (b2 + 3) >> 3; op[2] = (c2 + 3) >> 3; op[3] = (d2 + 3) >> 3; ip += 4; op += 4; } for(i = 0; i < 16; i++) { mb_dqcoeff[i * 16] = output[i]; } } void vp8_short_inv_walsh4x4_1_c(short *input, short *mb_dqcoeff) { int i; int a1; a1 = ((input[0] + 3) >> 3); for(i = 0; i < 16; i++) { mb_dqcoeff[i * 16] = a1; } } void vp8_idct_dequant_0_1x_c(short* q, short* dq, unsigned char* dst, int dst_stride) { vp8_dc_only_idct_add(q[0]*dq[0], dst, dst_stride, dst, dst_stride); q[0] = 0; }