From 9886db71d3d655183131fedfa4792ffcb54d4c00 Mon Sep 17 00:00:00 2001 From: Narayan Kalaburgi Date: Mon, 7 Sep 2026 17:42:50 +0530 Subject: [PATCH] libhevcdec: Unify itrans_res family and remove redundant HBD implementation Currently, libhevc maintains two separate implementations for inverse transform residual functions: ihevc_itrans_res_* and ihevc_hbd_itrans_res_*. Since both 8-bit and HBD inverse transform residual routines output signed 16-bit residuals (WORD16 *pi2_dst), the only algorithmic difference between 8-bit and HBD variants is the 2nd stage shift: Stage 1 shift: IT_SHIFT_STAGE_1 = 7 (fixed across all bit depths) Stage 2 shift: shift = 20 - bit_depth - 8-bit: 20 - 8 = 12 (IT_SHIFT_STAGE_2) - 10-bit: 20 - 10 = 10 This commit cleans up the entire itrans_res family (4x4_ttype1, 4x4, dc, 8x8, 16x16, 32x32) as follows: 1. common/ihevc_itrans_res.h & common/ihevc_itrans_res.c: - Add 'UWORD8 bit_depth' parameter to ihevc_itrans_res_* function signatures and typedefs. - Replace hardcoded 'shift = IT_SHIFT_STAGE_2;' with 'shift = 20 - bit_depth;'. - Remove obsolete ihevc_hbd_itrans_res_* prototypes and typedefs. 2. common/ihevc_hbd_itrans_res.c: - Delete file (~2,335 redundant lines removed). 3. Update tests/common/ihevc_itrans_res_test.cc to test bit depths 8 and 10 --- Android.bp | 1 - common/Android.bp | 1 - common/common.cmake | 1 - common/ihevc_hbd_itrans_res.c | 2335 -------------------- common/ihevc_itrans_res.c | 40 +- common/ihevc_itrans_res.h | 76 +- decoder/ihevcd_api.c | 8 - decoder/ihevcd_function_selector.h | 7 - decoder/ihevcd_function_selector_generic.c | 7 - decoder/ihevcd_iquant_itrans_recon_ctb.c | 31 +- decoder/ihevcd_structs.h | 23 +- tests/common/ihevc_itrans_res_test.cc | 30 +- 12 files changed, 62 insertions(+), 2498 deletions(-) delete mode 100644 common/ihevc_hbd_itrans_res.c diff --git a/Android.bp b/Android.bp index 9a9619b..e01221a 100644 --- a/Android.bp +++ b/Android.bp @@ -102,7 +102,6 @@ cc_library_static { "common/ihevc_hbd_inter_pred_filters.c", "common/ihevc_hbd_intra_pred_filters.c", "common/ihevc_hbd_iquant_recon.c", - "common/ihevc_hbd_itrans_res.c", "common/ihevc_hbd_itrans_recon.c", "common/ihevc_hbd_itrans_recon_16x16.c", "common/ihevc_hbd_itrans_recon_32x32.c", diff --git a/common/Android.bp b/common/Android.bp index 1caec0f..fb93524 100644 --- a/common/Android.bp +++ b/common/Android.bp @@ -64,7 +64,6 @@ cc_library_static { "ihevc_hbd_inter_pred_filters.c", "ihevc_hbd_intra_pred_filters.c", "ihevc_hbd_iquant_recon.c", - "ihevc_hbd_itrans_res.c", "ihevc_hbd_itrans_recon.c", "ihevc_hbd_itrans_recon_16x16.c", "ihevc_hbd_itrans_recon_32x32.c", diff --git a/common/common.cmake b/common/common.cmake index a62f669..a3bb40f 100644 --- a/common/common.cmake +++ b/common/common.cmake @@ -77,7 +77,6 @@ list( "${HEVC_ROOT}/common/ihevc_hbd_intra_pred_filters.c" "${HEVC_ROOT}/common/ihevc_hbd_iquant_recon.c" "${HEVC_ROOT}/common/ihevc_hbd_itrans_recon.c" - "${HEVC_ROOT}/common/ihevc_hbd_itrans_res.c" "${HEVC_ROOT}/common/ihevc_hbd_itrans_recon_16x16.c" "${HEVC_ROOT}/common/ihevc_hbd_itrans_recon_32x32.c" "${HEVC_ROOT}/common/ihevc_hbd_itrans_recon_8x8.c" diff --git a/common/ihevc_hbd_itrans_res.c b/common/ihevc_hbd_itrans_res.c deleted file mode 100644 index 5bef4fd..0000000 --- a/common/ihevc_hbd_itrans_res.c +++ /dev/null @@ -1,2335 +0,0 @@ -/****************************************************************************** -* -* Copyright (C) 2012 Ittiam Systems Pvt Ltd, Bangalore -* -* Licensed under the Apache License, Version 2.0 (the "License"); -* you may not use this file except in compliance with the License. -* You may obtain a copy of the License at: -* -* http://www.apache.org/licenses/LICENSE-2.0 -* -* Unless required by applicable law or agreed to in writing, software -* distributed under the License is distributed on an "AS IS" BASIS, -* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -* See the License for the specific language governing permissions and -* limitations under the License. -* -******************************************************************************/ -/** - ******************************************************************************* - * @file - * ihevc_hbd_itrans_res.c - * - * @brief - * Contains function definitions for inverse transform - * - * @author - * Ittiam - * - * @par List of Functions: - * - ihevc_hbd_itrans_res_4x4_ttype1() - * - ihevc_hbd_itrans_res_4x4() - * - ihevc_hbd_itrans_res_dc() - * - ihevc_hbd_itrans_res_8x8() - * - ihevc_hbd_itrans_res_16x16() - * - ihevc_hbd_itrans_res_32x32() - * - * @remarks - * None - * - ******************************************************************************* - */ - -#include -#include - -#include "ihevc_typedefs.h" -#include "ihevc_macros.h" -#include "ihevc_platform_macros.h" -#include "ihevc_defs.h" -#include "ihevc_trans_tables.h" -#include "ihevc_trans_macros.h" -#include "ihevc_itrans_res.h" - - -void ihevc_hbd_itrans_res_4x4_ttype1(WORD16 *pi2_src, - WORD16 *pi2_tmp, - WORD16 *pi2_dst, - WORD32 src_strd, - WORD32 dst_strd, - WORD32 zero_cols, - WORD32 zero_rows, - UWORD8 u1_bit_depth) -{ - WORD32 i, c[4]; - WORD32 add; - WORD32 shift; - WORD16 *pi2_tmp_orig; - WORD32 trans_size; - UNUSED(zero_rows); - trans_size = TRANS_SIZE_4; - - pi2_tmp_orig = pi2_tmp; - - /* Inverse Transform 1st stage */ - shift = IT_SHIFT_STAGE_1; - add = 1 << (shift - 1); - - for(i = 0; i < trans_size; i++) - { - /* Checking for Zero Cols */ - if((zero_cols & 1) == 1) - { - memset(pi2_tmp, 0, trans_size * sizeof(WORD16)); - } - else - { - // Intermediate Variables - c[0] = pi2_src[0] + pi2_src[2 * src_strd]; - c[1] = pi2_src[2 * src_strd] + pi2_src[3 * src_strd]; - c[2] = pi2_src[0] - pi2_src[3 * src_strd]; - c[3] = 74 * pi2_src[src_strd]; - - pi2_tmp[0] = - CLIP_S16((29 * c[0] + 55 * c[1] + c[3] + add) >> shift); - pi2_tmp[1] = - CLIP_S16((55 * c[2] - 29 * c[1] + c[3] + add) >> shift); - pi2_tmp[2] = - CLIP_S16((74 * (pi2_src[0] - pi2_src[2 * src_strd] + pi2_src[3 * src_strd]) + add) >> shift); - pi2_tmp[3] = - CLIP_S16((55 * c[0] + 29 * c[2] - c[3] + add) >> shift); - } - pi2_src++; - pi2_tmp += trans_size; - zero_cols = zero_cols >> 1; - } - - pi2_tmp = pi2_tmp_orig; - - /* Inverse Transform 2nd stage */ - shift = 20 - u1_bit_depth; - add = 1 << (shift - 1); - - for(i = 0; i < trans_size; i++) - { - WORD32 itrans_out; - // Intermediate Variables - c[0] = pi2_tmp[0] + pi2_tmp[2 * trans_size]; - c[1] = pi2_tmp[2 * trans_size] + pi2_tmp[3 * trans_size]; - c[2] = pi2_tmp[0] - pi2_tmp[3 * trans_size]; - c[3] = 74 * pi2_tmp[trans_size]; - - pi2_dst[0] = - CLIP_S16((29 * c[0] + 55 * c[1] + c[3] + add) >> shift); - - pi2_dst[1] = - CLIP_S16((55 * c[2] - 29 * c[1] + c[3] + add) >> shift); - - pi2_dst[2] = - CLIP_S16((74 * (pi2_tmp[0] - pi2_tmp[2 * trans_size] + pi2_tmp[3 * trans_size]) + add) >> shift); - - pi2_dst[3] = - CLIP_S16((55 * c[0] + 29 * c[2] - c[3] + add) >> shift); - - pi2_tmp++; - pi2_dst += dst_strd; - } -} - - -void ihevc_hbd_itrans_res_4x4(WORD16 *pi2_src, - WORD16 *pi2_tmp, - WORD16 *pi2_dst, - WORD32 src_strd, - WORD32 dst_strd, - WORD32 zero_cols, - WORD32 zero_rows, - UWORD8 u1_bit_depth) -{ - WORD32 j; - WORD32 e[2], o[2]; - WORD32 add; - WORD32 shift; - WORD16 *pi2_tmp_orig; - WORD32 trans_size; - UNUSED(zero_rows); - trans_size = TRANS_SIZE_4; - - pi2_tmp_orig = pi2_tmp; - - /* Inverse Transform 1st stage */ - shift = IT_SHIFT_STAGE_1; - add = 1 << (shift - 1); - - for(j = 0; j < trans_size; j++) - { - /* Checking for Zero Cols */ - if((zero_cols & 1) == 1) - { - memset(pi2_tmp, 0, trans_size * sizeof(WORD16)); - } - else - { - - /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */ - o[0] = g_ai2_ihevc_trans_4[1][0] * pi2_src[src_strd] - + g_ai2_ihevc_trans_4[3][0] * pi2_src[3 * src_strd]; - o[1] = g_ai2_ihevc_trans_4[1][1] * pi2_src[src_strd] - + g_ai2_ihevc_trans_4[3][1] * pi2_src[3 * src_strd]; - e[0] = g_ai2_ihevc_trans_4[0][0] * pi2_src[0] - + g_ai2_ihevc_trans_4[2][0] * pi2_src[2 * src_strd]; - e[1] = g_ai2_ihevc_trans_4[0][1] * pi2_src[0] - + g_ai2_ihevc_trans_4[2][1] * pi2_src[2 * src_strd]; - - pi2_tmp[0] = - CLIP_S16(((e[0] + o[0] + add) >> shift)); - pi2_tmp[1] = - CLIP_S16(((e[1] + o[1] + add) >> shift)); - pi2_tmp[2] = - CLIP_S16(((e[1] - o[1] + add) >> shift)); - pi2_tmp[3] = - CLIP_S16(((e[0] - o[0] + add) >> shift)); - - } - pi2_src++; - pi2_tmp += trans_size; - zero_cols = zero_cols >> 1; - } - - pi2_tmp = pi2_tmp_orig; - - /* Inverse Transform 2nd stage */ - shift = 20 - u1_bit_depth; - add = 1 << (shift - 1); - - for(j = 0; j < trans_size; j++) - { - WORD32 itrans_out; - /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */ - o[0] = g_ai2_ihevc_trans_4[1][0] * pi2_tmp[trans_size] - + g_ai2_ihevc_trans_4[3][0] * pi2_tmp[3 * trans_size]; - o[1] = g_ai2_ihevc_trans_4[1][1] * pi2_tmp[trans_size] - + g_ai2_ihevc_trans_4[3][1] * pi2_tmp[3 * trans_size]; - e[0] = g_ai2_ihevc_trans_4[0][0] * pi2_tmp[0] - + g_ai2_ihevc_trans_4[2][0] * pi2_tmp[2 * trans_size]; - e[1] = g_ai2_ihevc_trans_4[0][1] * pi2_tmp[0] - + g_ai2_ihevc_trans_4[2][1] * pi2_tmp[2 * trans_size]; - - pi2_dst[0] = - CLIP_S16(((e[0] + o[0] + add) >> shift)); - - pi2_dst[1] = - CLIP_S16(((e[1] + o[1] + add) >> shift)); - - pi2_dst[2] = - CLIP_S16(((e[1] - o[1] + add) >> shift)); - - pi2_dst[3] = - CLIP_S16(((e[0] - o[0] + add) >> shift)); - - pi2_tmp++; - pi2_dst += dst_strd; - - } -} - - -void ihevc_hbd_itrans_res_dc(WORD16 *pi2_dst, - WORD32 dst_strd, - WORD32 log2_trans_size, - WORD16 i2_coeff_value, - UWORD8 u1_bit_depth) -{ - WORD32 row, col; - WORD32 add, shift; - WORD32 dc_value, quant_out; - WORD32 trans_size; - - trans_size = (1 << log2_trans_size); - - quant_out = i2_coeff_value; - - shift = IT_SHIFT_STAGE_1; - add = 1 << (shift - 1); - dc_value = CLIP_S16((quant_out * 64 + add) >> shift); - shift = 20 - u1_bit_depth; - add = 1 << (shift - 1); - dc_value = CLIP_S16((dc_value * 64 + add) >> shift); - - for(row = 0; row < trans_size; row++) - for(col = 0; col < trans_size; col++) - pi2_dst[row * dst_strd + col] = dc_value; - -} - - -void ihevc_hbd_itrans_res_8x8(WORD16 *pi2_src, - WORD16 *pi2_tmp, - WORD16 *pi2_dst, - WORD32 src_strd, - WORD32 dst_strd, - WORD32 zero_cols, - WORD32 zero_rows, - UWORD8 u1_bit_depth) -{ - WORD32 j, k; - WORD32 e[4], o[4]; - WORD32 ee[2], eo[2]; - WORD32 add; - WORD32 shift; - WORD16 *pi2_tmp_orig; - WORD32 trans_size; - WORD32 zero_rows_2nd_stage = zero_cols; - WORD32 row_limit_2nd_stage; - - trans_size = TRANS_SIZE_8; - - pi2_tmp_orig = pi2_tmp; - - if((zero_cols & 0xF0) == 0xF0) - row_limit_2nd_stage = 4; - else - row_limit_2nd_stage = TRANS_SIZE_8; - - - if((zero_rows & 0xF0) == 0xF0) /* First 4 rows of input are non-zero */ - { - /************************************************************************************************/ - /**********************************START - IT_RECON_8x8******************************************/ - /************************************************************************************************/ - - /* Inverse Transform 1st stage */ - shift = IT_SHIFT_STAGE_1; - add = 1 << (shift - 1); - - for(j = 0; j < row_limit_2nd_stage; j++) - { - /* Checking for Zero Cols */ - if((zero_cols & 1) == 1) - { - memset(pi2_tmp, 0, trans_size * sizeof(WORD16)); - } - else - { - /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */ - for(k = 0; k < 4; k++) - { - o[k] = g_ai2_ihevc_trans_8[1][k] * pi2_src[src_strd] - + g_ai2_ihevc_trans_8[3][k] - * pi2_src[3 * src_strd]; - } - eo[0] = g_ai2_ihevc_trans_8[2][0] * pi2_src[2 * src_strd]; - eo[1] = g_ai2_ihevc_trans_8[2][1] * pi2_src[2 * src_strd]; - ee[0] = g_ai2_ihevc_trans_8[0][0] * pi2_src[0]; - ee[1] = g_ai2_ihevc_trans_8[0][1] * pi2_src[0]; - - /* Combining e and o terms at each hierarchy levels to calculate the final spatial domain vector */ - e[0] = ee[0] + eo[0]; - e[3] = ee[0] - eo[0]; - e[1] = ee[1] + eo[1]; - e[2] = ee[1] - eo[1]; - for(k = 0; k < 4; k++) - { - pi2_tmp[k] = - CLIP_S16(((e[k] + o[k] + add) >> shift)); - pi2_tmp[k + 4] = - CLIP_S16(((e[3 - k] - o[3 - k] + add) >> shift)); - } - } - pi2_src++; - pi2_tmp += trans_size; - zero_cols = zero_cols >> 1; - } - - pi2_tmp = pi2_tmp_orig; - - /* Inverse Transform 2nd stage */ - shift = 20 - u1_bit_depth; - add = 1 << (shift - 1); - if((zero_rows_2nd_stage & 0xF0) == 0xF0) /* First 4 rows of output of 1st stage are non-zero */ - { - for(j = 0; j < trans_size; j++) - { - /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */ - for(k = 0; k < 4; k++) - { - o[k] = g_ai2_ihevc_trans_8[1][k] * pi2_tmp[trans_size] - + g_ai2_ihevc_trans_8[3][k] * pi2_tmp[3 * trans_size]; - } - eo[0] = g_ai2_ihevc_trans_8[2][0] * pi2_tmp[2 * trans_size]; - eo[1] = g_ai2_ihevc_trans_8[2][1] * pi2_tmp[2 * trans_size]; - ee[0] = g_ai2_ihevc_trans_8[0][0] * pi2_tmp[0]; - ee[1] = g_ai2_ihevc_trans_8[0][1] * pi2_tmp[0]; - - /* Combining e and o terms at each hierarchy levels to calculate the final spatial domain vector */ - e[0] = ee[0] + eo[0]; - e[3] = ee[0] - eo[0]; - e[1] = ee[1] + eo[1]; - e[2] = ee[1] - eo[1]; - for(k = 0; k < 4; k++) - { - pi2_dst[k] = - CLIP_S16(((e[k] + o[k] + add) >> shift)); - - pi2_dst[k + 4] = - CLIP_S16(((e[3 - k] - o[3 - k] + add) >> shift)); - } - pi2_tmp++; - pi2_dst += dst_strd; - } - } - else /* All rows of output of 1st stage are non-zero */ - { - for(j = 0; j < trans_size; j++) - { - /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */ - for(k = 0; k < 4; k++) - { - o[k] = g_ai2_ihevc_trans_8[1][k] * pi2_tmp[trans_size] - + g_ai2_ihevc_trans_8[3][k] - * pi2_tmp[3 * trans_size] - + g_ai2_ihevc_trans_8[5][k] - * pi2_tmp[5 * trans_size] - + g_ai2_ihevc_trans_8[7][k] - * pi2_tmp[7 * trans_size]; - } - - eo[0] = g_ai2_ihevc_trans_8[2][0] * pi2_tmp[2 * trans_size] - + g_ai2_ihevc_trans_8[6][0] * pi2_tmp[6 * trans_size]; - eo[1] = g_ai2_ihevc_trans_8[2][1] * pi2_tmp[2 * trans_size] - + g_ai2_ihevc_trans_8[6][1] * pi2_tmp[6 * trans_size]; - ee[0] = g_ai2_ihevc_trans_8[0][0] * pi2_tmp[0] - + g_ai2_ihevc_trans_8[4][0] * pi2_tmp[4 * trans_size]; - ee[1] = g_ai2_ihevc_trans_8[0][1] * pi2_tmp[0] - + g_ai2_ihevc_trans_8[4][1] * pi2_tmp[4 * trans_size]; - - /* Combining e and o terms at each hierarchy levels to calculate the final spatial domain vector */ - e[0] = ee[0] + eo[0]; - e[3] = ee[0] - eo[0]; - e[1] = ee[1] + eo[1]; - e[2] = ee[1] - eo[1]; - for(k = 0; k < 4; k++) - { - pi2_dst[k] = - CLIP_S16(((e[k] + o[k] + add) >> shift)); - - pi2_dst[k + 4] = - CLIP_S16(((e[3 - k] - o[3 - k] + add) >> shift)); - } - pi2_tmp++; - pi2_dst += dst_strd; - } - } - /************************************************************************************************/ - /************************************END - IT_RECON_8x8******************************************/ - /************************************************************************************************/ - } - else /* All rows of input are non-zero */ - { - /************************************************************************************************/ - /**********************************START - IT_RECON_8x8******************************************/ - /************************************************************************************************/ - - /* Inverse Transform 1st stage */ - shift = IT_SHIFT_STAGE_1; - add = 1 << (shift - 1); - - for(j = 0; j < row_limit_2nd_stage; j++) - { - /* Checking for Zero Cols */ - if((zero_cols & 1) == 1) - { - memset(pi2_tmp, 0, trans_size * sizeof(WORD16)); - } - else - { - /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */ - for(k = 0; k < 4; k++) - { - o[k] = g_ai2_ihevc_trans_8[1][k] * pi2_src[src_strd] - + g_ai2_ihevc_trans_8[3][k] - * pi2_src[3 * src_strd] - + g_ai2_ihevc_trans_8[5][k] - * pi2_src[5 * src_strd] - + g_ai2_ihevc_trans_8[7][k] - * pi2_src[7 * src_strd]; - } - - eo[0] = g_ai2_ihevc_trans_8[2][0] * pi2_src[2 * src_strd] - + g_ai2_ihevc_trans_8[6][0] * pi2_src[6 * src_strd]; - eo[1] = g_ai2_ihevc_trans_8[2][1] * pi2_src[2 * src_strd] - + g_ai2_ihevc_trans_8[6][1] * pi2_src[6 * src_strd]; - ee[0] = g_ai2_ihevc_trans_8[0][0] * pi2_src[0] - + g_ai2_ihevc_trans_8[4][0] * pi2_src[4 * src_strd]; - ee[1] = g_ai2_ihevc_trans_8[0][1] * pi2_src[0] - + g_ai2_ihevc_trans_8[4][1] * pi2_src[4 * src_strd]; - - /* Combining e and o terms at each hierarchy levels to calculate the final spatial domain vector */ - e[0] = ee[0] + eo[0]; - e[3] = ee[0] - eo[0]; - e[1] = ee[1] + eo[1]; - e[2] = ee[1] - eo[1]; - for(k = 0; k < 4; k++) - { - pi2_tmp[k] = - CLIP_S16(((e[k] + o[k] + add) >> shift)); - pi2_tmp[k + 4] = - CLIP_S16(((e[3 - k] - o[3 - k] + add) >> shift)); - } - } - pi2_src++; - pi2_tmp += trans_size; - zero_cols = zero_cols >> 1; - } - - pi2_tmp = pi2_tmp_orig; - - /* Inverse Transform 2nd stage */ - shift = 20 - u1_bit_depth; - add = 1 << (shift - 1); - if((zero_rows_2nd_stage & 0xF0) == 0xF0) /* First 4 rows of output of 1st stage are non-zero */ - { - for(j = 0; j < trans_size; j++) - { - /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */ - for(k = 0; k < 4; k++) - { - o[k] = g_ai2_ihevc_trans_8[1][k] * pi2_tmp[trans_size] - + g_ai2_ihevc_trans_8[3][k] * pi2_tmp[3 * trans_size]; - } - eo[0] = g_ai2_ihevc_trans_8[2][0] * pi2_tmp[2 * trans_size]; - eo[1] = g_ai2_ihevc_trans_8[2][1] * pi2_tmp[2 * trans_size]; - ee[0] = g_ai2_ihevc_trans_8[0][0] * pi2_tmp[0]; - ee[1] = g_ai2_ihevc_trans_8[0][1] * pi2_tmp[0]; - - /* Combining e and o terms at each hierarchy levels to calculate the final spatial domain vector */ - e[0] = ee[0] + eo[0]; - e[3] = ee[0] - eo[0]; - e[1] = ee[1] + eo[1]; - e[2] = ee[1] - eo[1]; - for(k = 0; k < 4; k++) - { - pi2_dst[k] = - CLIP_S16(((e[k] + o[k] + add) >> shift)); - - pi2_dst[k + 4] = - CLIP_S16(((e[3 - k] - o[3 - k] + add) >> shift)); - } - pi2_tmp++; - pi2_dst += dst_strd; - } - } - else /* All rows of output of 1st stage are non-zero */ - { - for(j = 0; j < trans_size; j++) - { - /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */ - for(k = 0; k < 4; k++) - { - o[k] = g_ai2_ihevc_trans_8[1][k] * pi2_tmp[trans_size] - + g_ai2_ihevc_trans_8[3][k] - * pi2_tmp[3 * trans_size] - + g_ai2_ihevc_trans_8[5][k] - * pi2_tmp[5 * trans_size] - + g_ai2_ihevc_trans_8[7][k] - * pi2_tmp[7 * trans_size]; - } - - eo[0] = g_ai2_ihevc_trans_8[2][0] * pi2_tmp[2 * trans_size] - + g_ai2_ihevc_trans_8[6][0] * pi2_tmp[6 * trans_size]; - eo[1] = g_ai2_ihevc_trans_8[2][1] * pi2_tmp[2 * trans_size] - + g_ai2_ihevc_trans_8[6][1] * pi2_tmp[6 * trans_size]; - ee[0] = g_ai2_ihevc_trans_8[0][0] * pi2_tmp[0] - + g_ai2_ihevc_trans_8[4][0] * pi2_tmp[4 * trans_size]; - ee[1] = g_ai2_ihevc_trans_8[0][1] * pi2_tmp[0] - + g_ai2_ihevc_trans_8[4][1] * pi2_tmp[4 * trans_size]; - - /* Combining e and o terms at each hierarchy levels to calculate the final spatial domain vector */ - e[0] = ee[0] + eo[0]; - e[3] = ee[0] - eo[0]; - e[1] = ee[1] + eo[1]; - e[2] = ee[1] - eo[1]; - for(k = 0; k < 4; k++) - { - pi2_dst[k] = - CLIP_S16(((e[k] + o[k] + add) >> shift)); - - pi2_dst[k + 4] = - CLIP_S16(((e[3 - k] - o[3 - k] + add) >> shift)); - } - pi2_tmp++; - pi2_dst += dst_strd; - } - } - /************************************************************************************************/ - /************************************END - IT_RECON_8x8******************************************/ - /************************************************************************************************/ - } -} - - -void ihevc_hbd_itrans_res_16x16(WORD16 *pi2_src, - WORD16 *pi2_tmp, - WORD16 *pi2_dst, - WORD32 src_strd, - WORD32 dst_strd, - WORD32 zero_cols, - WORD32 zero_rows, - UWORD8 u1_bit_depth) -{ - WORD32 j, k; - WORD32 e[8], o[8]; - WORD32 ee[4], eo[4]; - WORD32 eee[2], eeo[2]; - WORD32 add; - WORD32 shift; - WORD16 *pi2_tmp_orig; - WORD32 trans_size; - WORD32 zero_rows_2nd_stage = zero_cols; - WORD32 row_limit_2nd_stage; - - if((zero_cols & 0xFFF0) == 0xFFF0) - row_limit_2nd_stage = 4; - else if((zero_cols & 0xFF00) == 0xFF00) - row_limit_2nd_stage = 8; - else - row_limit_2nd_stage = TRANS_SIZE_16; - - trans_size = TRANS_SIZE_16; - pi2_tmp_orig = pi2_tmp; - if((zero_rows & 0xFFF0) == 0xFFF0) /* First 4 rows of input are non-zero */ - { - /* Inverse Transform 1st stage */ - /************************************************************************************************/ - /**********************************START - IT_RECON_16x16****************************************/ - /************************************************************************************************/ - - shift = IT_SHIFT_STAGE_1; - add = 1 << (shift - 1); - - for(j = 0; j < row_limit_2nd_stage; j++) - { - /* Checking for Zero Cols */ - if((zero_cols & 1) == 1) - { - memset(pi2_tmp, 0, trans_size * sizeof(WORD16)); - } - else - { - /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */ - for(k = 0; k < 8; k++) - { - o[k] = g_ai2_ihevc_trans_16[1][k] * pi2_src[src_strd] - + g_ai2_ihevc_trans_16[3][k] - * pi2_src[3 * src_strd]; - } - for(k = 0; k < 4; k++) - { - eo[k] = g_ai2_ihevc_trans_16[2][k] * pi2_src[2 * src_strd]; - } - eeo[0] = 0; - eee[0] = g_ai2_ihevc_trans_16[0][0] * pi2_src[0]; - eeo[1] = 0; - eee[1] = g_ai2_ihevc_trans_16[0][1] * pi2_src[0]; - - /* Combining e and o terms at each hierarchy levels to calculate the final spatial domain vector */ - for(k = 0; k < 2; k++) - { - ee[k] = eee[k] + eeo[k]; - ee[k + 2] = eee[1 - k] - eeo[1 - k]; - } - for(k = 0; k < 4; k++) - { - e[k] = ee[k] + eo[k]; - e[k + 4] = ee[3 - k] - eo[3 - k]; - } - for(k = 0; k < 8; k++) - { - pi2_tmp[k] = - CLIP_S16(((e[k] + o[k] + add) >> shift)); - pi2_tmp[k + 8] = - CLIP_S16(((e[7 - k] - o[7 - k] + add) >> shift)); - } - } - pi2_src++; - pi2_tmp += trans_size; - zero_cols = zero_cols >> 1; - } - - pi2_tmp = pi2_tmp_orig; - - /* Inverse Transform 2nd stage */ - shift = 20 - u1_bit_depth; - add = 1 << (shift - 1); - - if((zero_rows_2nd_stage & 0xFFF0) == 0xFFF0) /* First 4 rows of output of 1st stage are non-zero */ - { - for(j = 0; j < trans_size; j++) - { - /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */ - for(k = 0; k < 8; k++) - { - o[k] = g_ai2_ihevc_trans_16[1][k] * pi2_tmp[trans_size] - + g_ai2_ihevc_trans_16[3][k] - * pi2_tmp[3 * trans_size]; - } - for(k = 0; k < 4; k++) - { - eo[k] = g_ai2_ihevc_trans_16[2][k] * pi2_tmp[2 * trans_size]; - } - eeo[0] = 0; - eee[0] = g_ai2_ihevc_trans_16[0][0] * pi2_tmp[0]; - eeo[1] = 0; - eee[1] = g_ai2_ihevc_trans_16[0][1] * pi2_tmp[0]; - - /* Combining e and o terms at each hierarchy levels to calculate the final spatial domain vector */ - for(k = 0; k < 2; k++) - { - ee[k] = eee[k] + eeo[k]; - ee[k + 2] = eee[1 - k] - eeo[1 - k]; - } - for(k = 0; k < 4; k++) - { - e[k] = ee[k] + eo[k]; - e[k + 4] = ee[3 - k] - eo[3 - k]; - } - for(k = 0; k < 8; k++) - { - pi2_dst[k] = - CLIP_S16(((e[k] + o[k] + add) >> shift)); - - pi2_dst[k + 8] = - CLIP_S16(((e[7 - k] - o[7 - k] + add) >> shift)); - } - pi2_tmp++; - pi2_dst += dst_strd; - } - } - else if((zero_rows_2nd_stage & 0xFF00) == 0xFF00) /* First 4 rows of output of 1st stage are non-zero */ - { - for(j = 0; j < trans_size; j++) - { - /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */ - for(k = 0; k < 8; k++) - { - o[k] = g_ai2_ihevc_trans_16[1][k] * pi2_tmp[trans_size] - + g_ai2_ihevc_trans_16[3][k] - * pi2_tmp[3 * trans_size] - + g_ai2_ihevc_trans_16[5][k] - * pi2_tmp[5 * trans_size] - + g_ai2_ihevc_trans_16[7][k] - * pi2_tmp[7 * trans_size]; - } - for(k = 0; k < 4; k++) - { - eo[k] = g_ai2_ihevc_trans_16[2][k] * pi2_tmp[2 * trans_size] - + g_ai2_ihevc_trans_16[6][k] - * pi2_tmp[6 * trans_size]; - } - eeo[0] = g_ai2_ihevc_trans_16[4][0] * pi2_tmp[4 * trans_size]; - eee[0] = g_ai2_ihevc_trans_16[0][0] * pi2_tmp[0]; - eeo[1] = g_ai2_ihevc_trans_16[4][1] * pi2_tmp[4 * trans_size]; - eee[1] = g_ai2_ihevc_trans_16[0][1] * pi2_tmp[0]; - - /* Combining e and o terms at each hierarchy levels to calculate the final spatial domain vector */ - for(k = 0; k < 2; k++) - { - ee[k] = eee[k] + eeo[k]; - ee[k + 2] = eee[1 - k] - eeo[1 - k]; - } - for(k = 0; k < 4; k++) - { - e[k] = ee[k] + eo[k]; - e[k + 4] = ee[3 - k] - eo[3 - k]; - } - for(k = 0; k < 8; k++) - { - pi2_dst[k] = - CLIP_S16(((e[k] + o[k] + add) >> shift)); - - pi2_dst[k + 8] = - CLIP_S16(((e[7 - k] - o[7 - k] + add) >> shift)); - } - pi2_tmp++; - pi2_dst += dst_strd; - } - } - else /* All rows of output of 1st stage are non-zero */ - { - for(j = 0; j < trans_size; j++) - { - /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */ - for(k = 0; k < 8; k++) - { - o[k] = g_ai2_ihevc_trans_16[1][k] * pi2_tmp[trans_size] - + g_ai2_ihevc_trans_16[3][k] - * pi2_tmp[3 * trans_size] - + g_ai2_ihevc_trans_16[5][k] - * pi2_tmp[5 * trans_size] - + g_ai2_ihevc_trans_16[7][k] - * pi2_tmp[7 * trans_size] - + g_ai2_ihevc_trans_16[9][k] - * pi2_tmp[9 * trans_size] - + g_ai2_ihevc_trans_16[11][k] - * pi2_tmp[11 * trans_size] - + g_ai2_ihevc_trans_16[13][k] - * pi2_tmp[13 * trans_size] - + g_ai2_ihevc_trans_16[15][k] - * pi2_tmp[15 * trans_size]; - } - for(k = 0; k < 4; k++) - { - eo[k] = g_ai2_ihevc_trans_16[2][k] * pi2_tmp[2 * trans_size] - + g_ai2_ihevc_trans_16[6][k] - * pi2_tmp[6 * trans_size] - + g_ai2_ihevc_trans_16[10][k] - * pi2_tmp[10 * trans_size] - + g_ai2_ihevc_trans_16[14][k] - * pi2_tmp[14 * trans_size]; - } - eeo[0] = - g_ai2_ihevc_trans_16[4][0] * pi2_tmp[4 * trans_size] - + g_ai2_ihevc_trans_16[12][0] - * pi2_tmp[12 - * trans_size]; - eee[0] = g_ai2_ihevc_trans_16[0][0] * pi2_tmp[0] - + g_ai2_ihevc_trans_16[8][0] * pi2_tmp[8 * trans_size]; - eeo[1] = - g_ai2_ihevc_trans_16[4][1] * pi2_tmp[4 * trans_size] - + g_ai2_ihevc_trans_16[12][1] - * pi2_tmp[12 - * trans_size]; - eee[1] = g_ai2_ihevc_trans_16[0][1] * pi2_tmp[0] - + g_ai2_ihevc_trans_16[8][1] * pi2_tmp[8 * trans_size]; - - /* Combining e and o terms at each hierarchy levels to calculate the final spatial domain vector */ - for(k = 0; k < 2; k++) - { - ee[k] = eee[k] + eeo[k]; - ee[k + 2] = eee[1 - k] - eeo[1 - k]; - } - for(k = 0; k < 4; k++) - { - e[k] = ee[k] + eo[k]; - e[k + 4] = ee[3 - k] - eo[3 - k]; - } - for(k = 0; k < 8; k++) - { - pi2_dst[k] = - CLIP_S16(((e[k] + o[k] + add) >> shift)); - - pi2_dst[k + 8] = - CLIP_S16(((e[7 - k] - o[7 - k] + add) >> shift)); - } - pi2_tmp++; - pi2_dst += dst_strd; - } - } - /************************************************************************************************/ - /************************************END - IT_RECON_16x16****************************************/ - /************************************************************************************************/ - } - else if((zero_rows & 0xFF00) == 0xFF00) /* First 8 rows of input are non-zero */ - { - /* Inverse Transform 1st stage */ - /************************************************************************************************/ - /**********************************START - IT_RECON_16x16****************************************/ - /************************************************************************************************/ - - shift = IT_SHIFT_STAGE_1; - add = 1 << (shift - 1); - - for(j = 0; j < row_limit_2nd_stage; j++) - { - /* Checking for Zero Cols */ - if((zero_cols & 1) == 1) - { - memset(pi2_tmp, 0, trans_size * sizeof(WORD16)); - } - else - { - /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */ - for(k = 0; k < 8; k++) - { - o[k] = g_ai2_ihevc_trans_16[1][k] * pi2_src[src_strd] - + g_ai2_ihevc_trans_16[3][k] - * pi2_src[3 * src_strd] - + g_ai2_ihevc_trans_16[5][k] - * pi2_src[5 * src_strd] - + g_ai2_ihevc_trans_16[7][k] - * pi2_src[7 * src_strd]; - } - for(k = 0; k < 4; k++) - { - eo[k] = g_ai2_ihevc_trans_16[2][k] * pi2_src[2 * src_strd] - + g_ai2_ihevc_trans_16[6][k] - * pi2_src[6 * src_strd]; - } - eeo[0] = g_ai2_ihevc_trans_16[4][0] * pi2_src[4 * src_strd]; - eee[0] = g_ai2_ihevc_trans_16[0][0] * pi2_src[0]; - eeo[1] = g_ai2_ihevc_trans_16[4][1] * pi2_src[4 * src_strd]; - eee[1] = g_ai2_ihevc_trans_16[0][1] * pi2_src[0]; - - /* Combining e and o terms at each hierarchy levels to calculate the final spatial domain vector */ - for(k = 0; k < 2; k++) - { - ee[k] = eee[k] + eeo[k]; - ee[k + 2] = eee[1 - k] - eeo[1 - k]; - } - for(k = 0; k < 4; k++) - { - e[k] = ee[k] + eo[k]; - e[k + 4] = ee[3 - k] - eo[3 - k]; - } - for(k = 0; k < 8; k++) - { - pi2_tmp[k] = - CLIP_S16(((e[k] + o[k] + add) >> shift)); - pi2_tmp[k + 8] = - CLIP_S16(((e[7 - k] - o[7 - k] + add) >> shift)); - } - } - pi2_src++; - pi2_tmp += trans_size; - zero_cols = zero_cols >> 1; - } - - pi2_tmp = pi2_tmp_orig; - - /* Inverse Transform 2nd stage */ - shift = 20 - u1_bit_depth; - add = 1 << (shift - 1); - - if((zero_rows_2nd_stage & 0xFFF0) == 0xFFF0) /* First 4 rows of output of 1st stage are non-zero */ - { - for(j = 0; j < trans_size; j++) - { - /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */ - for(k = 0; k < 8; k++) - { - o[k] = g_ai2_ihevc_trans_16[1][k] * pi2_tmp[trans_size] - + g_ai2_ihevc_trans_16[3][k] - * pi2_tmp[3 * trans_size]; - } - for(k = 0; k < 4; k++) - { - eo[k] = g_ai2_ihevc_trans_16[2][k] * pi2_tmp[2 * trans_size]; - } - eeo[0] = 0; - eee[0] = g_ai2_ihevc_trans_16[0][0] * pi2_tmp[0]; - eeo[1] = 0; - eee[1] = g_ai2_ihevc_trans_16[0][1] * pi2_tmp[0]; - - /* Combining e and o terms at each hierarchy levels to calculate the final spatial domain vector */ - for(k = 0; k < 2; k++) - { - ee[k] = eee[k] + eeo[k]; - ee[k + 2] = eee[1 - k] - eeo[1 - k]; - } - for(k = 0; k < 4; k++) - { - e[k] = ee[k] + eo[k]; - e[k + 4] = ee[3 - k] - eo[3 - k]; - } - for(k = 0; k < 8; k++) - { - pi2_dst[k] = - CLIP_S16(((e[k] + o[k] + add) >> shift)); - - pi2_dst[k + 8] = - CLIP_S16(((e[7 - k] - o[7 - k] + add) >> shift)); - } - pi2_tmp++; - pi2_dst += dst_strd; - } - } - else if((zero_rows_2nd_stage & 0xFF00) == 0xFF00) /* First 4 rows of output of 1st stage are non-zero */ - { - for(j = 0; j < trans_size; j++) - { - /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */ - for(k = 0; k < 8; k++) - { - o[k] = g_ai2_ihevc_trans_16[1][k] * pi2_tmp[trans_size] - + g_ai2_ihevc_trans_16[3][k] - * pi2_tmp[3 * trans_size] - + g_ai2_ihevc_trans_16[5][k] - * pi2_tmp[5 * trans_size] - + g_ai2_ihevc_trans_16[7][k] - * pi2_tmp[7 * trans_size]; - } - for(k = 0; k < 4; k++) - { - eo[k] = g_ai2_ihevc_trans_16[2][k] * pi2_tmp[2 * trans_size] - + g_ai2_ihevc_trans_16[6][k] - * pi2_tmp[6 * trans_size]; - } - eeo[0] = g_ai2_ihevc_trans_16[4][0] * pi2_tmp[4 * trans_size]; - eee[0] = g_ai2_ihevc_trans_16[0][0] * pi2_tmp[0]; - eeo[1] = g_ai2_ihevc_trans_16[4][1] * pi2_tmp[4 * trans_size]; - eee[1] = g_ai2_ihevc_trans_16[0][1] * pi2_tmp[0]; - - /* Combining e and o terms at each hierarchy levels to calculate the final spatial domain vector */ - for(k = 0; k < 2; k++) - { - ee[k] = eee[k] + eeo[k]; - ee[k + 2] = eee[1 - k] - eeo[1 - k]; - } - for(k = 0; k < 4; k++) - { - e[k] = ee[k] + eo[k]; - e[k + 4] = ee[3 - k] - eo[3 - k]; - } - for(k = 0; k < 8; k++) - { - pi2_dst[k] = - CLIP_S16(((e[k] + o[k] + add) >> shift)); - - pi2_dst[k + 8] = - CLIP_S16(((e[7 - k] - o[7 - k] + add) >> shift)); - } - pi2_tmp++; - pi2_dst += dst_strd; - } - } - else /* All rows of output of 1st stage are non-zero */ - { - for(j = 0; j < trans_size; j++) - { - /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */ - for(k = 0; k < 8; k++) - { - o[k] = g_ai2_ihevc_trans_16[1][k] * pi2_tmp[trans_size] - + g_ai2_ihevc_trans_16[3][k] - * pi2_tmp[3 * trans_size] - + g_ai2_ihevc_trans_16[5][k] - * pi2_tmp[5 * trans_size] - + g_ai2_ihevc_trans_16[7][k] - * pi2_tmp[7 * trans_size] - + g_ai2_ihevc_trans_16[9][k] - * pi2_tmp[9 * trans_size] - + g_ai2_ihevc_trans_16[11][k] - * pi2_tmp[11 * trans_size] - + g_ai2_ihevc_trans_16[13][k] - * pi2_tmp[13 * trans_size] - + g_ai2_ihevc_trans_16[15][k] - * pi2_tmp[15 * trans_size]; - } - for(k = 0; k < 4; k++) - { - eo[k] = g_ai2_ihevc_trans_16[2][k] * pi2_tmp[2 * trans_size] - + g_ai2_ihevc_trans_16[6][k] - * pi2_tmp[6 * trans_size] - + g_ai2_ihevc_trans_16[10][k] - * pi2_tmp[10 * trans_size] - + g_ai2_ihevc_trans_16[14][k] - * pi2_tmp[14 * trans_size]; - } - eeo[0] = - g_ai2_ihevc_trans_16[4][0] * pi2_tmp[4 * trans_size] - + g_ai2_ihevc_trans_16[12][0] - * pi2_tmp[12 - * trans_size]; - eee[0] = g_ai2_ihevc_trans_16[0][0] * pi2_tmp[0] - + g_ai2_ihevc_trans_16[8][0] * pi2_tmp[8 * trans_size]; - eeo[1] = - g_ai2_ihevc_trans_16[4][1] * pi2_tmp[4 * trans_size] - + g_ai2_ihevc_trans_16[12][1] - * pi2_tmp[12 - * trans_size]; - eee[1] = g_ai2_ihevc_trans_16[0][1] * pi2_tmp[0] - + g_ai2_ihevc_trans_16[8][1] * pi2_tmp[8 * trans_size]; - - /* Combining e and o terms at each hierarchy levels to calculate the final spatial domain vector */ - for(k = 0; k < 2; k++) - { - ee[k] = eee[k] + eeo[k]; - ee[k + 2] = eee[1 - k] - eeo[1 - k]; - } - for(k = 0; k < 4; k++) - { - e[k] = ee[k] + eo[k]; - e[k + 4] = ee[3 - k] - eo[3 - k]; - } - for(k = 0; k < 8; k++) - { - pi2_dst[k] = - CLIP_S16(((e[k] + o[k] + add) >> shift)); - - pi2_dst[k + 8] = - CLIP_S16(((e[7 - k] - o[7 - k] + add) >> shift)); - } - pi2_tmp++; - pi2_dst += dst_strd; - } - } - /************************************************************************************************/ - /************************************END - IT_RECON_16x16****************************************/ - /************************************************************************************************/ - } - else /* All rows of input are non-zero */ - { - /* Inverse Transform 1st stage */ - /************************************************************************************************/ - /**********************************START - IT_RECON_16x16****************************************/ - /************************************************************************************************/ - - shift = IT_SHIFT_STAGE_1; - add = 1 << (shift - 1); - - for(j = 0; j < row_limit_2nd_stage; j++) - { - /* Checking for Zero Cols */ - if((zero_cols & 1) == 1) - { - memset(pi2_tmp, 0, trans_size * sizeof(WORD16)); - } - else - { - /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */ - for(k = 0; k < 8; k++) - { - o[k] = g_ai2_ihevc_trans_16[1][k] * pi2_src[src_strd] - + g_ai2_ihevc_trans_16[3][k] - * pi2_src[3 * src_strd] - + g_ai2_ihevc_trans_16[5][k] - * pi2_src[5 * src_strd] - + g_ai2_ihevc_trans_16[7][k] - * pi2_src[7 * src_strd] - + g_ai2_ihevc_trans_16[9][k] - * pi2_src[9 * src_strd] - + g_ai2_ihevc_trans_16[11][k] - * pi2_src[11 * src_strd] - + g_ai2_ihevc_trans_16[13][k] - * pi2_src[13 * src_strd] - + g_ai2_ihevc_trans_16[15][k] - * pi2_src[15 * src_strd]; - } - for(k = 0; k < 4; k++) - { - eo[k] = g_ai2_ihevc_trans_16[2][k] * pi2_src[2 * src_strd] - + g_ai2_ihevc_trans_16[6][k] - * pi2_src[6 * src_strd] - + g_ai2_ihevc_trans_16[10][k] - * pi2_src[10 * src_strd] - + g_ai2_ihevc_trans_16[14][k] - * pi2_src[14 * src_strd]; - } - eeo[0] = g_ai2_ihevc_trans_16[4][0] * pi2_src[4 * src_strd] - + g_ai2_ihevc_trans_16[12][0] - * pi2_src[12 * src_strd]; - eee[0] = - g_ai2_ihevc_trans_16[0][0] * pi2_src[0] - + g_ai2_ihevc_trans_16[8][0] - * pi2_src[8 - * src_strd]; - eeo[1] = g_ai2_ihevc_trans_16[4][1] * pi2_src[4 * src_strd] - + g_ai2_ihevc_trans_16[12][1] - * pi2_src[12 * src_strd]; - eee[1] = - g_ai2_ihevc_trans_16[0][1] * pi2_src[0] - + g_ai2_ihevc_trans_16[8][1] - * pi2_src[8 - * src_strd]; - - /* Combining e and o terms at each hierarchy levels to calculate the final spatial domain vector */ - for(k = 0; k < 2; k++) - { - ee[k] = eee[k] + eeo[k]; - ee[k + 2] = eee[1 - k] - eeo[1 - k]; - } - for(k = 0; k < 4; k++) - { - e[k] = ee[k] + eo[k]; - e[k + 4] = ee[3 - k] - eo[3 - k]; - } - for(k = 0; k < 8; k++) - { - pi2_tmp[k] = - CLIP_S16(((e[k] + o[k] + add) >> shift)); - pi2_tmp[k + 8] = - CLIP_S16(((e[7 - k] - o[7 - k] + add) >> shift)); - } - } - pi2_src++; - pi2_tmp += trans_size; - zero_cols = zero_cols >> 1; - } - - pi2_tmp = pi2_tmp_orig; - - /* Inverse Transform 2nd stage */ - shift = 20 - u1_bit_depth; - add = 1 << (shift - 1); - - if((zero_rows_2nd_stage & 0xFFF0) == 0xFFF0) /* First 4 rows of output of 1st stage are non-zero */ - { - for(j = 0; j < trans_size; j++) - { - /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */ - for(k = 0; k < 8; k++) - { - o[k] = g_ai2_ihevc_trans_16[1][k] * pi2_tmp[trans_size] - + g_ai2_ihevc_trans_16[3][k] - * pi2_tmp[3 * trans_size]; - } - for(k = 0; k < 4; k++) - { - eo[k] = g_ai2_ihevc_trans_16[2][k] * pi2_tmp[2 * trans_size]; - } - eeo[0] = 0; - eee[0] = g_ai2_ihevc_trans_16[0][0] * pi2_tmp[0]; - eeo[1] = 0; - eee[1] = g_ai2_ihevc_trans_16[0][1] * pi2_tmp[0]; - - /* Combining e and o terms at each hierarchy levels to calculate the final spatial domain vector */ - for(k = 0; k < 2; k++) - { - ee[k] = eee[k] + eeo[k]; - ee[k + 2] = eee[1 - k] - eeo[1 - k]; - } - for(k = 0; k < 4; k++) - { - e[k] = ee[k] + eo[k]; - e[k + 4] = ee[3 - k] - eo[3 - k]; - } - for(k = 0; k < 8; k++) - { - pi2_dst[k] = - CLIP_S16(((e[k] + o[k] + add) >> shift)); - - pi2_dst[k + 8] = - CLIP_S16(((e[7 - k] - o[7 - k] + add) >> shift)); - } - pi2_tmp++; - pi2_dst += dst_strd; - } - } - else if((zero_rows_2nd_stage & 0xFF00) == 0xFF00) /* First 4 rows of output of 1st stage are non-zero */ - { - for(j = 0; j < trans_size; j++) - { - /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */ - for(k = 0; k < 8; k++) - { - o[k] = g_ai2_ihevc_trans_16[1][k] * pi2_tmp[trans_size] - + g_ai2_ihevc_trans_16[3][k] - * pi2_tmp[3 * trans_size] - + g_ai2_ihevc_trans_16[5][k] - * pi2_tmp[5 * trans_size] - + g_ai2_ihevc_trans_16[7][k] - * pi2_tmp[7 * trans_size]; - } - for(k = 0; k < 4; k++) - { - eo[k] = g_ai2_ihevc_trans_16[2][k] * pi2_tmp[2 * trans_size] - + g_ai2_ihevc_trans_16[6][k] - * pi2_tmp[6 * trans_size]; - } - eeo[0] = g_ai2_ihevc_trans_16[4][0] * pi2_tmp[4 * trans_size]; - eee[0] = g_ai2_ihevc_trans_16[0][0] * pi2_tmp[0]; - eeo[1] = g_ai2_ihevc_trans_16[4][1] * pi2_tmp[4 * trans_size]; - eee[1] = g_ai2_ihevc_trans_16[0][1] * pi2_tmp[0]; - - /* Combining e and o terms at each hierarchy levels to calculate the final spatial domain vector */ - for(k = 0; k < 2; k++) - { - ee[k] = eee[k] + eeo[k]; - ee[k + 2] = eee[1 - k] - eeo[1 - k]; - } - for(k = 0; k < 4; k++) - { - e[k] = ee[k] + eo[k]; - e[k + 4] = ee[3 - k] - eo[3 - k]; - } - for(k = 0; k < 8; k++) - { - pi2_dst[k] = - CLIP_S16(((e[k] + o[k] + add) >> shift)); - - pi2_dst[k + 8] = - CLIP_S16(((e[7 - k] - o[7 - k] + add) >> shift)); - } - pi2_tmp++; - pi2_dst += dst_strd; - } - } - else /* All rows of output of 1st stage are non-zero */ - { - for(j = 0; j < trans_size; j++) - { - /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */ - for(k = 0; k < 8; k++) - { - o[k] = g_ai2_ihevc_trans_16[1][k] * pi2_tmp[trans_size] - + g_ai2_ihevc_trans_16[3][k] - * pi2_tmp[3 * trans_size] - + g_ai2_ihevc_trans_16[5][k] - * pi2_tmp[5 * trans_size] - + g_ai2_ihevc_trans_16[7][k] - * pi2_tmp[7 * trans_size] - + g_ai2_ihevc_trans_16[9][k] - * pi2_tmp[9 * trans_size] - + g_ai2_ihevc_trans_16[11][k] - * pi2_tmp[11 * trans_size] - + g_ai2_ihevc_trans_16[13][k] - * pi2_tmp[13 * trans_size] - + g_ai2_ihevc_trans_16[15][k] - * pi2_tmp[15 * trans_size]; - } - for(k = 0; k < 4; k++) - { - eo[k] = g_ai2_ihevc_trans_16[2][k] * pi2_tmp[2 * trans_size] - + g_ai2_ihevc_trans_16[6][k] - * pi2_tmp[6 * trans_size] - + g_ai2_ihevc_trans_16[10][k] - * pi2_tmp[10 * trans_size] - + g_ai2_ihevc_trans_16[14][k] - * pi2_tmp[14 * trans_size]; - } - eeo[0] = - g_ai2_ihevc_trans_16[4][0] * pi2_tmp[4 * trans_size] - + g_ai2_ihevc_trans_16[12][0] - * pi2_tmp[12 - * trans_size]; - eee[0] = g_ai2_ihevc_trans_16[0][0] * pi2_tmp[0] - + g_ai2_ihevc_trans_16[8][0] * pi2_tmp[8 * trans_size]; - eeo[1] = - g_ai2_ihevc_trans_16[4][1] * pi2_tmp[4 * trans_size] - + g_ai2_ihevc_trans_16[12][1] - * pi2_tmp[12 - * trans_size]; - eee[1] = g_ai2_ihevc_trans_16[0][1] * pi2_tmp[0] - + g_ai2_ihevc_trans_16[8][1] * pi2_tmp[8 * trans_size]; - - /* Combining e and o terms at each hierarchy levels to calculate the final spatial domain vector */ - for(k = 0; k < 2; k++) - { - ee[k] = eee[k] + eeo[k]; - ee[k + 2] = eee[1 - k] - eeo[1 - k]; - } - for(k = 0; k < 4; k++) - { - e[k] = ee[k] + eo[k]; - e[k + 4] = ee[3 - k] - eo[3 - k]; - } - for(k = 0; k < 8; k++) - { - pi2_dst[k] = - CLIP_S16(((e[k] + o[k] + add) >> shift)); - - pi2_dst[k + 8] = - CLIP_S16(((e[7 - k] - o[7 - k] + add) >> shift)); - } - pi2_tmp++; - pi2_dst += dst_strd; - } - } - /************************************************************************************************/ - /************************************END - IT_RECON_16x16****************************************/ - /************************************************************************************************/ - } - -} - - -void ihevc_hbd_itrans_res_32x32(WORD16 *pi2_src, - WORD16 *pi2_tmp, - WORD16 *pi2_dst, - WORD32 src_strd, - WORD32 dst_strd, - WORD32 zero_cols, - WORD32 zero_rows, - UWORD8 u1_bit_depth) -{ - WORD32 j, k; - WORD32 e[16], o[16]; - WORD32 ee[8], eo[8]; - WORD32 eee[4], eeo[4]; - WORD32 eeee[2], eeeo[2]; - WORD32 add; - WORD32 shift; - WORD16 *pi2_tmp_orig; - WORD32 trans_size; - WORD32 zero_rows_2nd_stage = zero_cols; - WORD32 row_limit_2nd_stage; - - trans_size = TRANS_SIZE_32; - pi2_tmp_orig = pi2_tmp; - - if((zero_cols & 0xFFFFFFF0) == 0xFFFFFFF0) - row_limit_2nd_stage = 4; - else if((zero_cols & 0xFFFFFF00) == 0xFFFFFF00) - row_limit_2nd_stage = 8; - else - row_limit_2nd_stage = TRANS_SIZE_32; - - if((zero_rows & 0xFFFFFFF0) == 0xFFFFFFF0) /* First 4 rows of input are non-zero */ - { - /************************************************************************************************/ - /**********************************START - IT_RECON_32x32****************************************/ - /************************************************************************************************/ - /* Inverse Transform 1st stage */ - shift = IT_SHIFT_STAGE_1; - add = 1 << (shift - 1); - - for(j = 0; j < row_limit_2nd_stage; j++) - { - /* Checking for Zero Cols */ - if((zero_cols & 1) == 1) - { - memset(pi2_tmp, 0, trans_size * sizeof(WORD16)); - } - else - { - /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */ - for(k = 0; k < 16; k++) - { - o[k] = g_ai2_ihevc_trans_32[1][k] * pi2_src[src_strd] - + g_ai2_ihevc_trans_32[3][k] - * pi2_src[3 * src_strd]; - } - for(k = 0; k < 8; k++) - { - eo[k] = g_ai2_ihevc_trans_32[2][k] * pi2_src[2 * src_strd]; - } -// for(k = 0; k < 4; k++) - { - eeo[0] = 0; - eeo[1] = 0; - eeo[2] = 0; - eeo[3] = 0; - } - eeeo[0] = 0; - eeeo[1] = 0; - eeee[0] = g_ai2_ihevc_trans_32[0][0] * pi2_src[0]; - eeee[1] = g_ai2_ihevc_trans_32[0][1] * pi2_src[0]; - - /* Combining e and o terms at each hierarchy levels to calculate the final spatial domain vector */ - eee[0] = eeee[0] + eeeo[0]; - eee[3] = eeee[0] - eeeo[0]; - eee[1] = eeee[1] + eeeo[1]; - eee[2] = eeee[1] - eeeo[1]; - for(k = 0; k < 4; k++) - { - ee[k] = eee[k] + eeo[k]; - ee[k + 4] = eee[3 - k] - eeo[3 - k]; - } - for(k = 0; k < 8; k++) - { - e[k] = ee[k] + eo[k]; - e[k + 8] = ee[7 - k] - eo[7 - k]; - } - for(k = 0; k < 16; k++) - { - pi2_tmp[k] = - CLIP_S16(((e[k] + o[k] + add) >> shift)); - pi2_tmp[k + 16] = - CLIP_S16(((e[15 - k] - o[15 - k] + add) >> shift)); - } - } - pi2_src++; - pi2_tmp += trans_size; - zero_cols = zero_cols >> 1; - } - - pi2_tmp = pi2_tmp_orig; - - /* Inverse Transform 2nd stage */ - shift = 20 - u1_bit_depth; - add = 1 << (shift - 1); - if((zero_rows_2nd_stage & 0xFFFFFFF0) == 0xFFFFFFF0) /* First 4 rows of output of 1st stage are non-zero */ - { - for(j = 0; j < trans_size; j++) - { - /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */ - for(k = 0; k < 16; k++) - { - o[k] = g_ai2_ihevc_trans_32[1][k] * pi2_tmp[trans_size] - + g_ai2_ihevc_trans_32[3][k] - * pi2_tmp[3 * trans_size]; - } - for(k = 0; k < 8; k++) - { - eo[k] = g_ai2_ihevc_trans_32[2][k] * pi2_tmp[2 * trans_size]; - } -// for(k = 0; k < 4; k++) - { - eeo[0] = 0; - eeo[1] = 0; - eeo[2] = 0; - eeo[3] = 0; - } - eeeo[0] = 0; - eeeo[1] = 0; - eeee[0] = g_ai2_ihevc_trans_32[0][0] * pi2_tmp[0]; - eeee[1] = g_ai2_ihevc_trans_32[0][1] * pi2_tmp[0]; - - /* Combining e and o terms at each hierarchy levels to calculate the final spatial domain vector */ - eee[0] = eeee[0] + eeeo[0]; - eee[3] = eeee[0] - eeeo[0]; - eee[1] = eeee[1] + eeeo[1]; - eee[2] = eeee[1] - eeeo[1]; - for(k = 0; k < 4; k++) - { - ee[k] = eee[k] + eeo[k]; - ee[k + 4] = eee[3 - k] - eeo[3 - k]; - } - for(k = 0; k < 8; k++) - { - e[k] = ee[k] + eo[k]; - e[k + 8] = ee[7 - k] - eo[7 - k]; - } - for(k = 0; k < 16; k++) - { - pi2_dst[k] = - CLIP_S16(((e[k] + o[k] + add) >> shift)); - - pi2_dst[k + 16] = - CLIP_S16(((e[15 - k] - o[15 - k] + add) >> shift)); - } - pi2_tmp++; - pi2_dst += dst_strd; - } - } - else if((zero_rows_2nd_stage & 0xFFFFFF00) == 0xFFFFFF00) /* First 8 rows of output of 1st stage are non-zero */ - { - for(j = 0; j < trans_size; j++) - { - /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */ - for(k = 0; k < 16; k++) - { - o[k] = g_ai2_ihevc_trans_32[1][k] * pi2_tmp[trans_size] - + g_ai2_ihevc_trans_32[3][k] - * pi2_tmp[3 * trans_size] - + g_ai2_ihevc_trans_32[5][k] - * pi2_tmp[5 * trans_size] - + g_ai2_ihevc_trans_32[7][k] - * pi2_tmp[7 * trans_size]; - } - for(k = 0; k < 8; k++) - { - eo[k] = g_ai2_ihevc_trans_32[2][k] * pi2_tmp[2 * trans_size] - + g_ai2_ihevc_trans_32[6][k] - * pi2_tmp[6 * trans_size]; - } - for(k = 0; k < 4; k++) - { - eeo[k] = g_ai2_ihevc_trans_32[4][k] * pi2_tmp[4 * trans_size]; - } - eeeo[0] = 0; - eeeo[1] = 0; - eeee[0] = g_ai2_ihevc_trans_32[0][0] * pi2_tmp[0]; - eeee[1] = g_ai2_ihevc_trans_32[0][1] * pi2_tmp[0]; - - /* Combining e and o terms at each hierarchy levels to calculate the final spatial domain vector */ - eee[0] = eeee[0] + eeeo[0]; - eee[3] = eeee[0] - eeeo[0]; - eee[1] = eeee[1] + eeeo[1]; - eee[2] = eeee[1] - eeeo[1]; - for(k = 0; k < 4; k++) - { - ee[k] = eee[k] + eeo[k]; - ee[k + 4] = eee[3 - k] - eeo[3 - k]; - } - for(k = 0; k < 8; k++) - { - e[k] = ee[k] + eo[k]; - e[k + 8] = ee[7 - k] - eo[7 - k]; - } - for(k = 0; k < 16; k++) - { - pi2_dst[k] = - CLIP_S16(((e[k] + o[k] + add) >> shift)); - - pi2_dst[k + 16] = - CLIP_S16(((e[15 - k] - o[15 - k] + add) >> shift)); - } - pi2_tmp++; - pi2_dst += dst_strd; - } - } - else /* All rows of output of 1st stage are non-zero */ - { - for(j = 0; j < trans_size; j++) - { - /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */ - for(k = 0; k < 16; k++) - { - o[k] = g_ai2_ihevc_trans_32[1][k] * pi2_tmp[trans_size] - + g_ai2_ihevc_trans_32[3][k] - * pi2_tmp[3 * trans_size] - + g_ai2_ihevc_trans_32[5][k] - * pi2_tmp[5 * trans_size] - + g_ai2_ihevc_trans_32[7][k] - * pi2_tmp[7 * trans_size] - + g_ai2_ihevc_trans_32[9][k] - * pi2_tmp[9 * trans_size] - + g_ai2_ihevc_trans_32[11][k] - * pi2_tmp[11 * trans_size] - + g_ai2_ihevc_trans_32[13][k] - * pi2_tmp[13 * trans_size] - + g_ai2_ihevc_trans_32[15][k] - * pi2_tmp[15 * trans_size] - + g_ai2_ihevc_trans_32[17][k] - * pi2_tmp[17 * trans_size] - + g_ai2_ihevc_trans_32[19][k] - * pi2_tmp[19 * trans_size] - + g_ai2_ihevc_trans_32[21][k] - * pi2_tmp[21 * trans_size] - + g_ai2_ihevc_trans_32[23][k] - * pi2_tmp[23 * trans_size] - + g_ai2_ihevc_trans_32[25][k] - * pi2_tmp[25 * trans_size] - + g_ai2_ihevc_trans_32[27][k] - * pi2_tmp[27 * trans_size] - + g_ai2_ihevc_trans_32[29][k] - * pi2_tmp[29 * trans_size] - + g_ai2_ihevc_trans_32[31][k] - * pi2_tmp[31 * trans_size]; - } - for(k = 0; k < 8; k++) - { - eo[k] = g_ai2_ihevc_trans_32[2][k] * pi2_tmp[2 * trans_size] - + g_ai2_ihevc_trans_32[6][k] - * pi2_tmp[6 * trans_size] - + g_ai2_ihevc_trans_32[10][k] - * pi2_tmp[10 * trans_size] - + g_ai2_ihevc_trans_32[14][k] - * pi2_tmp[14 * trans_size] - + g_ai2_ihevc_trans_32[18][k] - * pi2_tmp[18 * trans_size] - + g_ai2_ihevc_trans_32[22][k] - * pi2_tmp[22 * trans_size] - + g_ai2_ihevc_trans_32[26][k] - * pi2_tmp[26 * trans_size] - + g_ai2_ihevc_trans_32[30][k] - * pi2_tmp[30 * trans_size]; - } - for(k = 0; k < 4; k++) - { - eeo[k] = g_ai2_ihevc_trans_32[4][k] * pi2_tmp[4 * trans_size] - + g_ai2_ihevc_trans_32[12][k] - * pi2_tmp[12 * trans_size] - + g_ai2_ihevc_trans_32[20][k] - * pi2_tmp[20 * trans_size] - + g_ai2_ihevc_trans_32[28][k] - * pi2_tmp[28 * trans_size]; - } - eeeo[0] = - g_ai2_ihevc_trans_32[8][0] * pi2_tmp[8 * trans_size] - + g_ai2_ihevc_trans_32[24][0] - * pi2_tmp[24 - * trans_size]; - eeeo[1] = - g_ai2_ihevc_trans_32[8][1] * pi2_tmp[8 * trans_size] - + g_ai2_ihevc_trans_32[24][1] - * pi2_tmp[24 - * trans_size]; - eeee[0] = - g_ai2_ihevc_trans_32[0][0] * pi2_tmp[0] - + g_ai2_ihevc_trans_32[16][0] - * pi2_tmp[16 - * trans_size]; - eeee[1] = - g_ai2_ihevc_trans_32[0][1] * pi2_tmp[0] - + g_ai2_ihevc_trans_32[16][1] - * pi2_tmp[16 - * trans_size]; - - /* Combining e and o terms at each hierarchy levels to calculate the final spatial domain vector */ - eee[0] = eeee[0] + eeeo[0]; - eee[3] = eeee[0] - eeeo[0]; - eee[1] = eeee[1] + eeeo[1]; - eee[2] = eeee[1] - eeeo[1]; - for(k = 0; k < 4; k++) - { - ee[k] = eee[k] + eeo[k]; - ee[k + 4] = eee[3 - k] - eeo[3 - k]; - } - for(k = 0; k < 8; k++) - { - e[k] = ee[k] + eo[k]; - e[k + 8] = ee[7 - k] - eo[7 - k]; - } - for(k = 0; k < 16; k++) - { - pi2_dst[k] = - CLIP_S16(((e[k] + o[k] + add) >> shift)); - - pi2_dst[k + 16] = - CLIP_S16(((e[15 - k] - o[15 - k] + add) >> shift)); - } - pi2_tmp++; - pi2_dst += dst_strd; - } - } - /************************************************************************************************/ - /************************************END - IT_RECON_32x32****************************************/ - /************************************************************************************************/ - } - else if((zero_rows & 0xFFFFFF00) == 0xFFFFFF00) /* First 8 rows of input are non-zero */ - { - /************************************************************************************************/ - /**********************************START - IT_RECON_32x32****************************************/ - /************************************************************************************************/ - /* Inverse Transform 1st stage */ - shift = IT_SHIFT_STAGE_1; - add = 1 << (shift - 1); - - for(j = 0; j < row_limit_2nd_stage; j++) - { - /* Checking for Zero Cols */ - if((zero_cols & 1) == 1) - { - memset(pi2_tmp, 0, trans_size * sizeof(WORD16)); - } - else - { - /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */ - for(k = 0; k < 16; k++) - { - o[k] = g_ai2_ihevc_trans_32[1][k] * pi2_src[src_strd] - + g_ai2_ihevc_trans_32[3][k] - * pi2_src[3 * src_strd] - + g_ai2_ihevc_trans_32[5][k] - * pi2_src[5 * src_strd] - + g_ai2_ihevc_trans_32[7][k] - * pi2_src[7 * src_strd]; - } - for(k = 0; k < 8; k++) - { - eo[k] = g_ai2_ihevc_trans_32[2][k] * pi2_src[2 * src_strd] - + g_ai2_ihevc_trans_32[6][k] - * pi2_src[6 * src_strd]; - } - for(k = 0; k < 4; k++) - { - eeo[k] = g_ai2_ihevc_trans_32[4][k] * pi2_src[4 * src_strd]; - } - eeeo[0] = 0; - eeeo[1] = 0; - eeee[0] = g_ai2_ihevc_trans_32[0][0] * pi2_src[0]; - eeee[1] = g_ai2_ihevc_trans_32[0][1] * pi2_src[0]; - - /* Combining e and o terms at each hierarchy levels to calculate the final spatial domain vector */ - eee[0] = eeee[0] + eeeo[0]; - eee[3] = eeee[0] - eeeo[0]; - eee[1] = eeee[1] + eeeo[1]; - eee[2] = eeee[1] - eeeo[1]; - for(k = 0; k < 4; k++) - { - ee[k] = eee[k] + eeo[k]; - ee[k + 4] = eee[3 - k] - eeo[3 - k]; - } - for(k = 0; k < 8; k++) - { - e[k] = ee[k] + eo[k]; - e[k + 8] = ee[7 - k] - eo[7 - k]; - } - for(k = 0; k < 16; k++) - { - pi2_tmp[k] = - CLIP_S16(((e[k] + o[k] + add) >> shift)); - pi2_tmp[k + 16] = - CLIP_S16(((e[15 - k] - o[15 - k] + add) >> shift)); - } - } - pi2_src++; - pi2_tmp += trans_size; - zero_cols = zero_cols >> 1; - } - - pi2_tmp = pi2_tmp_orig; - - /* Inverse Transform 2nd stage */ - shift = 20 - u1_bit_depth; - add = 1 << (shift - 1); - if((zero_rows_2nd_stage & 0xFFFFFFF0) == 0xFFFFFFF0) /* First 4 rows of output of 1st stage are non-zero */ - { - for(j = 0; j < trans_size; j++) - { - /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */ - for(k = 0; k < 16; k++) - { - o[k] = g_ai2_ihevc_trans_32[1][k] * pi2_tmp[trans_size] - + g_ai2_ihevc_trans_32[3][k] - * pi2_tmp[3 * trans_size]; - } - for(k = 0; k < 8; k++) - { - eo[k] = g_ai2_ihevc_trans_32[2][k] * pi2_tmp[2 * trans_size]; - } -// for(k = 0; k < 4; k++) - { - eeo[0] = 0; - eeo[1] = 0; - eeo[2] = 0; - eeo[3] = 0; - } - eeeo[0] = 0; - eeeo[1] = 0; - eeee[0] = g_ai2_ihevc_trans_32[0][0] * pi2_tmp[0]; - eeee[1] = g_ai2_ihevc_trans_32[0][1] * pi2_tmp[0]; - - /* Combining e and o terms at each hierarchy levels to calculate the final spatial domain vector */ - eee[0] = eeee[0] + eeeo[0]; - eee[3] = eeee[0] - eeeo[0]; - eee[1] = eeee[1] + eeeo[1]; - eee[2] = eeee[1] - eeeo[1]; - for(k = 0; k < 4; k++) - { - ee[k] = eee[k] + eeo[k]; - ee[k + 4] = eee[3 - k] - eeo[3 - k]; - } - for(k = 0; k < 8; k++) - { - e[k] = ee[k] + eo[k]; - e[k + 8] = ee[7 - k] - eo[7 - k]; - } - for(k = 0; k < 16; k++) - { - pi2_dst[k] = - CLIP_S16(((e[k] + o[k] + add) >> shift)); - - pi2_dst[k + 16] = - CLIP_S16(((e[15 - k] - o[15 - k] + add) >> shift)); - } - pi2_tmp++; - pi2_dst += dst_strd; - } - } - else if((zero_rows_2nd_stage & 0xFFFFFF00) == 0xFFFFFF00) /* First 8 rows of output of 1st stage are non-zero */ - { - for(j = 0; j < trans_size; j++) - { - /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */ - for(k = 0; k < 16; k++) - { - o[k] = g_ai2_ihevc_trans_32[1][k] * pi2_tmp[trans_size] - + g_ai2_ihevc_trans_32[3][k] - * pi2_tmp[3 * trans_size] - + g_ai2_ihevc_trans_32[5][k] - * pi2_tmp[5 * trans_size] - + g_ai2_ihevc_trans_32[7][k] - * pi2_tmp[7 * trans_size]; - } - for(k = 0; k < 8; k++) - { - eo[k] = g_ai2_ihevc_trans_32[2][k] * pi2_tmp[2 * trans_size] - + g_ai2_ihevc_trans_32[6][k] - * pi2_tmp[6 * trans_size]; - } - for(k = 0; k < 4; k++) - { - eeo[k] = g_ai2_ihevc_trans_32[4][k] * pi2_tmp[4 * trans_size]; - } - eeeo[0] = 0; - eeeo[1] = 0; - eeee[0] = g_ai2_ihevc_trans_32[0][0] * pi2_tmp[0]; - eeee[1] = g_ai2_ihevc_trans_32[0][1] * pi2_tmp[0]; - - /* Combining e and o terms at each hierarchy levels to calculate the final spatial domain vector */ - eee[0] = eeee[0] + eeeo[0]; - eee[3] = eeee[0] - eeeo[0]; - eee[1] = eeee[1] + eeeo[1]; - eee[2] = eeee[1] - eeeo[1]; - for(k = 0; k < 4; k++) - { - ee[k] = eee[k] + eeo[k]; - ee[k + 4] = eee[3 - k] - eeo[3 - k]; - } - for(k = 0; k < 8; k++) - { - e[k] = ee[k] + eo[k]; - e[k + 8] = ee[7 - k] - eo[7 - k]; - } - for(k = 0; k < 16; k++) - { - pi2_dst[k] = - CLIP_S16(((e[k] + o[k] + add) >> shift)); - - pi2_dst[k + 16] = - CLIP_S16(((e[15 - k] - o[15 - k] + add) >> shift)); - } - pi2_tmp++; - pi2_dst += dst_strd; - } - } - else /* All rows of output of 1st stage are non-zero */ - { - for(j = 0; j < trans_size; j++) - { - /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */ - for(k = 0; k < 16; k++) - { - o[k] = g_ai2_ihevc_trans_32[1][k] * pi2_tmp[trans_size] - + g_ai2_ihevc_trans_32[3][k] - * pi2_tmp[3 * trans_size] - + g_ai2_ihevc_trans_32[5][k] - * pi2_tmp[5 * trans_size] - + g_ai2_ihevc_trans_32[7][k] - * pi2_tmp[7 * trans_size] - + g_ai2_ihevc_trans_32[9][k] - * pi2_tmp[9 * trans_size] - + g_ai2_ihevc_trans_32[11][k] - * pi2_tmp[11 * trans_size] - + g_ai2_ihevc_trans_32[13][k] - * pi2_tmp[13 * trans_size] - + g_ai2_ihevc_trans_32[15][k] - * pi2_tmp[15 * trans_size] - + g_ai2_ihevc_trans_32[17][k] - * pi2_tmp[17 * trans_size] - + g_ai2_ihevc_trans_32[19][k] - * pi2_tmp[19 * trans_size] - + g_ai2_ihevc_trans_32[21][k] - * pi2_tmp[21 * trans_size] - + g_ai2_ihevc_trans_32[23][k] - * pi2_tmp[23 * trans_size] - + g_ai2_ihevc_trans_32[25][k] - * pi2_tmp[25 * trans_size] - + g_ai2_ihevc_trans_32[27][k] - * pi2_tmp[27 * trans_size] - + g_ai2_ihevc_trans_32[29][k] - * pi2_tmp[29 * trans_size] - + g_ai2_ihevc_trans_32[31][k] - * pi2_tmp[31 * trans_size]; - } - for(k = 0; k < 8; k++) - { - eo[k] = g_ai2_ihevc_trans_32[2][k] * pi2_tmp[2 * trans_size] - + g_ai2_ihevc_trans_32[6][k] - * pi2_tmp[6 * trans_size] - + g_ai2_ihevc_trans_32[10][k] - * pi2_tmp[10 * trans_size] - + g_ai2_ihevc_trans_32[14][k] - * pi2_tmp[14 * trans_size] - + g_ai2_ihevc_trans_32[18][k] - * pi2_tmp[18 * trans_size] - + g_ai2_ihevc_trans_32[22][k] - * pi2_tmp[22 * trans_size] - + g_ai2_ihevc_trans_32[26][k] - * pi2_tmp[26 * trans_size] - + g_ai2_ihevc_trans_32[30][k] - * pi2_tmp[30 * trans_size]; - } - for(k = 0; k < 4; k++) - { - eeo[k] = g_ai2_ihevc_trans_32[4][k] * pi2_tmp[4 * trans_size] - + g_ai2_ihevc_trans_32[12][k] - * pi2_tmp[12 * trans_size] - + g_ai2_ihevc_trans_32[20][k] - * pi2_tmp[20 * trans_size] - + g_ai2_ihevc_trans_32[28][k] - * pi2_tmp[28 * trans_size]; - } - eeeo[0] = - g_ai2_ihevc_trans_32[8][0] * pi2_tmp[8 * trans_size] - + g_ai2_ihevc_trans_32[24][0] - * pi2_tmp[24 - * trans_size]; - eeeo[1] = - g_ai2_ihevc_trans_32[8][1] * pi2_tmp[8 * trans_size] - + g_ai2_ihevc_trans_32[24][1] - * pi2_tmp[24 - * trans_size]; - eeee[0] = - g_ai2_ihevc_trans_32[0][0] * pi2_tmp[0] - + g_ai2_ihevc_trans_32[16][0] - * pi2_tmp[16 - * trans_size]; - eeee[1] = - g_ai2_ihevc_trans_32[0][1] * pi2_tmp[0] - + g_ai2_ihevc_trans_32[16][1] - * pi2_tmp[16 - * trans_size]; - - /* Combining e and o terms at each hierarchy levels to calculate the final spatial domain vector */ - eee[0] = eeee[0] + eeeo[0]; - eee[3] = eeee[0] - eeeo[0]; - eee[1] = eeee[1] + eeeo[1]; - eee[2] = eeee[1] - eeeo[1]; - for(k = 0; k < 4; k++) - { - ee[k] = eee[k] + eeo[k]; - ee[k + 4] = eee[3 - k] - eeo[3 - k]; - } - for(k = 0; k < 8; k++) - { - e[k] = ee[k] + eo[k]; - e[k + 8] = ee[7 - k] - eo[7 - k]; - } - for(k = 0; k < 16; k++) - { - pi2_dst[k] = - CLIP_S16(((e[k] + o[k] + add) >> shift)); - - pi2_dst[k + 16] = - CLIP_S16(((e[15 - k] - o[15 - k] + add) >> shift)); - } - pi2_tmp++; - pi2_dst += dst_strd; - } - } - /************************************************************************************************/ - /************************************END - IT_RECON_32x32****************************************/ - /************************************************************************************************/ - } - else /* All rows of input are non-zero */ - { - /************************************************************************************************/ - /**********************************START - IT_RECON_32x32****************************************/ - /************************************************************************************************/ - /* Inverse Transform 1st stage */ - shift = IT_SHIFT_STAGE_1; - add = 1 << (shift - 1); - - for(j = 0; j < row_limit_2nd_stage; j++) - { - /* Checking for Zero Cols */ - if((zero_cols & 1) == 1) - { - memset(pi2_tmp, 0, trans_size * sizeof(WORD16)); - } - else - { - /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */ - for(k = 0; k < 16; k++) - { - o[k] = g_ai2_ihevc_trans_32[1][k] * pi2_src[src_strd] - + g_ai2_ihevc_trans_32[3][k] - * pi2_src[3 * src_strd] - + g_ai2_ihevc_trans_32[5][k] - * pi2_src[5 * src_strd] - + g_ai2_ihevc_trans_32[7][k] - * pi2_src[7 * src_strd] - + g_ai2_ihevc_trans_32[9][k] - * pi2_src[9 * src_strd] - + g_ai2_ihevc_trans_32[11][k] - * pi2_src[11 * src_strd] - + g_ai2_ihevc_trans_32[13][k] - * pi2_src[13 * src_strd] - + g_ai2_ihevc_trans_32[15][k] - * pi2_src[15 * src_strd] - + g_ai2_ihevc_trans_32[17][k] - * pi2_src[17 * src_strd] - + g_ai2_ihevc_trans_32[19][k] - * pi2_src[19 * src_strd] - + g_ai2_ihevc_trans_32[21][k] - * pi2_src[21 * src_strd] - + g_ai2_ihevc_trans_32[23][k] - * pi2_src[23 * src_strd] - + g_ai2_ihevc_trans_32[25][k] - * pi2_src[25 * src_strd] - + g_ai2_ihevc_trans_32[27][k] - * pi2_src[27 * src_strd] - + g_ai2_ihevc_trans_32[29][k] - * pi2_src[29 * src_strd] - + g_ai2_ihevc_trans_32[31][k] - * pi2_src[31 * src_strd]; - } - for(k = 0; k < 8; k++) - { - eo[k] = g_ai2_ihevc_trans_32[2][k] * pi2_src[2 * src_strd] - + g_ai2_ihevc_trans_32[6][k] - * pi2_src[6 * src_strd] - + g_ai2_ihevc_trans_32[10][k] - * pi2_src[10 * src_strd] - + g_ai2_ihevc_trans_32[14][k] - * pi2_src[14 * src_strd] - + g_ai2_ihevc_trans_32[18][k] - * pi2_src[18 * src_strd] - + g_ai2_ihevc_trans_32[22][k] - * pi2_src[22 * src_strd] - + g_ai2_ihevc_trans_32[26][k] - * pi2_src[26 * src_strd] - + g_ai2_ihevc_trans_32[30][k] - * pi2_src[30 * src_strd]; - } - for(k = 0; k < 4; k++) - { - eeo[k] = g_ai2_ihevc_trans_32[4][k] * pi2_src[4 * src_strd] - + g_ai2_ihevc_trans_32[12][k] - * pi2_src[12 * src_strd] - + g_ai2_ihevc_trans_32[20][k] - * pi2_src[20 * src_strd] - + g_ai2_ihevc_trans_32[28][k] - * pi2_src[28 * src_strd]; - } - eeeo[0] = g_ai2_ihevc_trans_32[8][0] * pi2_src[8 * src_strd] - + g_ai2_ihevc_trans_32[24][0] - * pi2_src[24 * src_strd]; - eeeo[1] = g_ai2_ihevc_trans_32[8][1] * pi2_src[8 * src_strd] - + g_ai2_ihevc_trans_32[24][1] - * pi2_src[24 * src_strd]; - eeee[0] = g_ai2_ihevc_trans_32[0][0] * pi2_src[0] - + g_ai2_ihevc_trans_32[16][0] - * pi2_src[16 * src_strd]; - eeee[1] = g_ai2_ihevc_trans_32[0][1] * pi2_src[0] - + g_ai2_ihevc_trans_32[16][1] - * pi2_src[16 * src_strd]; - - /* Combining e and o terms at each hierarchy levels to calculate the final spatial domain vector */ - eee[0] = eeee[0] + eeeo[0]; - eee[3] = eeee[0] - eeeo[0]; - eee[1] = eeee[1] + eeeo[1]; - eee[2] = eeee[1] - eeeo[1]; - for(k = 0; k < 4; k++) - { - ee[k] = eee[k] + eeo[k]; - ee[k + 4] = eee[3 - k] - eeo[3 - k]; - } - for(k = 0; k < 8; k++) - { - e[k] = ee[k] + eo[k]; - e[k + 8] = ee[7 - k] - eo[7 - k]; - } - for(k = 0; k < 16; k++) - { - pi2_tmp[k] = - CLIP_S16(((e[k] + o[k] + add) >> shift)); - pi2_tmp[k + 16] = - CLIP_S16(((e[15 - k] - o[15 - k] + add) >> shift)); - } - } - pi2_src++; - pi2_tmp += trans_size; - zero_cols = zero_cols >> 1; - } - - pi2_tmp = pi2_tmp_orig; - - /* Inverse Transform 2nd stage */ - shift = 20 - u1_bit_depth; - add = 1 << (shift - 1); - if((zero_rows_2nd_stage & 0xFFFFFFF0) == 0xFFFFFFF0) /* First 4 rows of output of 1st stage are non-zero */ - { - for(j = 0; j < trans_size; j++) - { - /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */ - for(k = 0; k < 16; k++) - { - o[k] = g_ai2_ihevc_trans_32[1][k] * pi2_tmp[trans_size] - + g_ai2_ihevc_trans_32[3][k] - * pi2_tmp[3 * trans_size]; - } - for(k = 0; k < 8; k++) - { - eo[k] = g_ai2_ihevc_trans_32[2][k] * pi2_tmp[2 * trans_size]; - } -// for(k = 0; k < 4; k++) - { - eeo[0] = 0; - eeo[1] = 0; - eeo[2] = 0; - eeo[3] = 0; - } - eeeo[0] = 0; - eeeo[1] = 0; - eeee[0] = g_ai2_ihevc_trans_32[0][0] * pi2_tmp[0]; - eeee[1] = g_ai2_ihevc_trans_32[0][1] * pi2_tmp[0]; - - /* Combining e and o terms at each hierarchy levels to calculate the final spatial domain vector */ - eee[0] = eeee[0] + eeeo[0]; - eee[3] = eeee[0] - eeeo[0]; - eee[1] = eeee[1] + eeeo[1]; - eee[2] = eeee[1] - eeeo[1]; - for(k = 0; k < 4; k++) - { - ee[k] = eee[k] + eeo[k]; - ee[k + 4] = eee[3 - k] - eeo[3 - k]; - } - for(k = 0; k < 8; k++) - { - e[k] = ee[k] + eo[k]; - e[k + 8] = ee[7 - k] - eo[7 - k]; - } - for(k = 0; k < 16; k++) - { - pi2_dst[k] = - CLIP_S16(((e[k] + o[k] + add) >> shift)); - - pi2_dst[k + 16] = - CLIP_S16(((e[15 - k] - o[15 - k] + add) >> shift)); - } - pi2_tmp++; - pi2_dst += dst_strd; - } - } - else if((zero_rows_2nd_stage & 0xFFFFFF00) == 0xFFFFFF00) /* First 8 rows of output of 1st stage are non-zero */ - { - for(j = 0; j < trans_size; j++) - { - /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */ - for(k = 0; k < 16; k++) - { - o[k] = g_ai2_ihevc_trans_32[1][k] * pi2_tmp[trans_size] - + g_ai2_ihevc_trans_32[3][k] - * pi2_tmp[3 * trans_size] - + g_ai2_ihevc_trans_32[5][k] - * pi2_tmp[5 * trans_size] - + g_ai2_ihevc_trans_32[7][k] - * pi2_tmp[7 * trans_size]; - } - for(k = 0; k < 8; k++) - { - eo[k] = g_ai2_ihevc_trans_32[2][k] * pi2_tmp[2 * trans_size] - + g_ai2_ihevc_trans_32[6][k] - * pi2_tmp[6 * trans_size]; - } - for(k = 0; k < 4; k++) - { - eeo[k] = g_ai2_ihevc_trans_32[4][k] * pi2_tmp[4 * trans_size]; - } - eeeo[0] = 0; - eeeo[1] = 0; - eeee[0] = g_ai2_ihevc_trans_32[0][0] * pi2_tmp[0]; - eeee[1] = g_ai2_ihevc_trans_32[0][1] * pi2_tmp[0]; - - /* Combining e and o terms at each hierarchy levels to calculate the final spatial domain vector */ - eee[0] = eeee[0] + eeeo[0]; - eee[3] = eeee[0] - eeeo[0]; - eee[1] = eeee[1] + eeeo[1]; - eee[2] = eeee[1] - eeeo[1]; - for(k = 0; k < 4; k++) - { - ee[k] = eee[k] + eeo[k]; - ee[k + 4] = eee[3 - k] - eeo[3 - k]; - } - for(k = 0; k < 8; k++) - { - e[k] = ee[k] + eo[k]; - e[k + 8] = ee[7 - k] - eo[7 - k]; - } - for(k = 0; k < 16; k++) - { - pi2_dst[k] = - CLIP_S16(((e[k] + o[k] + add) >> shift)); - - pi2_dst[k + 16] = - CLIP_S16(((e[15 - k] - o[15 - k] + add) >> shift)); - } - pi2_tmp++; - pi2_dst += dst_strd; - } - } - else /* All rows of output of 1st stage are non-zero */ - { - for(j = 0; j < trans_size; j++) - { - /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */ - for(k = 0; k < 16; k++) - { - o[k] = g_ai2_ihevc_trans_32[1][k] * pi2_tmp[trans_size] - + g_ai2_ihevc_trans_32[3][k] - * pi2_tmp[3 * trans_size] - + g_ai2_ihevc_trans_32[5][k] - * pi2_tmp[5 * trans_size] - + g_ai2_ihevc_trans_32[7][k] - * pi2_tmp[7 * trans_size] - + g_ai2_ihevc_trans_32[9][k] - * pi2_tmp[9 * trans_size] - + g_ai2_ihevc_trans_32[11][k] - * pi2_tmp[11 * trans_size] - + g_ai2_ihevc_trans_32[13][k] - * pi2_tmp[13 * trans_size] - + g_ai2_ihevc_trans_32[15][k] - * pi2_tmp[15 * trans_size] - + g_ai2_ihevc_trans_32[17][k] - * pi2_tmp[17 * trans_size] - + g_ai2_ihevc_trans_32[19][k] - * pi2_tmp[19 * trans_size] - + g_ai2_ihevc_trans_32[21][k] - * pi2_tmp[21 * trans_size] - + g_ai2_ihevc_trans_32[23][k] - * pi2_tmp[23 * trans_size] - + g_ai2_ihevc_trans_32[25][k] - * pi2_tmp[25 * trans_size] - + g_ai2_ihevc_trans_32[27][k] - * pi2_tmp[27 * trans_size] - + g_ai2_ihevc_trans_32[29][k] - * pi2_tmp[29 * trans_size] - + g_ai2_ihevc_trans_32[31][k] - * pi2_tmp[31 * trans_size]; - } - for(k = 0; k < 8; k++) - { - eo[k] = g_ai2_ihevc_trans_32[2][k] * pi2_tmp[2 * trans_size] - + g_ai2_ihevc_trans_32[6][k] - * pi2_tmp[6 * trans_size] - + g_ai2_ihevc_trans_32[10][k] - * pi2_tmp[10 * trans_size] - + g_ai2_ihevc_trans_32[14][k] - * pi2_tmp[14 * trans_size] - + g_ai2_ihevc_trans_32[18][k] - * pi2_tmp[18 * trans_size] - + g_ai2_ihevc_trans_32[22][k] - * pi2_tmp[22 * trans_size] - + g_ai2_ihevc_trans_32[26][k] - * pi2_tmp[26 * trans_size] - + g_ai2_ihevc_trans_32[30][k] - * pi2_tmp[30 * trans_size]; - } - for(k = 0; k < 4; k++) - { - eeo[k] = g_ai2_ihevc_trans_32[4][k] * pi2_tmp[4 * trans_size] - + g_ai2_ihevc_trans_32[12][k] - * pi2_tmp[12 * trans_size] - + g_ai2_ihevc_trans_32[20][k] - * pi2_tmp[20 * trans_size] - + g_ai2_ihevc_trans_32[28][k] - * pi2_tmp[28 * trans_size]; - } - eeeo[0] = - g_ai2_ihevc_trans_32[8][0] * pi2_tmp[8 * trans_size] - + g_ai2_ihevc_trans_32[24][0] - * pi2_tmp[24 - * trans_size]; - eeeo[1] = - g_ai2_ihevc_trans_32[8][1] * pi2_tmp[8 * trans_size] - + g_ai2_ihevc_trans_32[24][1] - * pi2_tmp[24 - * trans_size]; - eeee[0] = - g_ai2_ihevc_trans_32[0][0] * pi2_tmp[0] - + g_ai2_ihevc_trans_32[16][0] - * pi2_tmp[16 - * trans_size]; - eeee[1] = - g_ai2_ihevc_trans_32[0][1] * pi2_tmp[0] - + g_ai2_ihevc_trans_32[16][1] - * pi2_tmp[16 - * trans_size]; - - /* Combining e and o terms at each hierarchy levels to calculate the final spatial domain vector */ - eee[0] = eeee[0] + eeeo[0]; - eee[3] = eeee[0] - eeeo[0]; - eee[1] = eeee[1] + eeeo[1]; - eee[2] = eeee[1] - eeeo[1]; - for(k = 0; k < 4; k++) - { - ee[k] = eee[k] + eeo[k]; - ee[k + 4] = eee[3 - k] - eeo[3 - k]; - } - for(k = 0; k < 8; k++) - { - e[k] = ee[k] + eo[k]; - e[k + 8] = ee[7 - k] - eo[7 - k]; - } - for(k = 0; k < 16; k++) - { - pi2_dst[k] = - CLIP_S16(((e[k] + o[k] + add) >> shift)); - - pi2_dst[k + 16] = - CLIP_S16(((e[15 - k] - o[15 - k] + add) >> shift)); - } - pi2_tmp++; - pi2_dst += dst_strd; - } - } - /************************************************************************************************/ - /************************************END - IT_RECON_32x32****************************************/ - /************************************************************************************************/ - } -} diff --git a/common/ihevc_itrans_res.c b/common/ihevc_itrans_res.c index 5758685..7bc12eb 100644 --- a/common/ihevc_itrans_res.c +++ b/common/ihevc_itrans_res.c @@ -62,7 +62,8 @@ void ihevc_itrans_res_4x4_ttype1(WORD16 *pi2_src, WORD32 src_strd, WORD32 dst_strd, WORD32 zero_cols, - WORD32 zero_rows) + WORD32 zero_rows, + UWORD8 bit_depth) { WORD32 i, c[4]; WORD32 add; @@ -110,7 +111,7 @@ void ihevc_itrans_res_4x4_ttype1(WORD16 *pi2_src, pi2_tmp = pi2_tmp_orig; /* Inverse Transform 2nd stage */ - shift = IT_SHIFT_STAGE_2; + shift = 20 - bit_depth; add = 1 << (shift - 1); for(i = 0; i < trans_size; i++) @@ -146,7 +147,8 @@ void ihevc_itrans_res_4x4(WORD16 *pi2_src, WORD32 src_strd, WORD32 dst_strd, WORD32 zero_cols, - WORD32 zero_rows) + WORD32 zero_rows, + UWORD8 bit_depth) { WORD32 j; @@ -202,7 +204,7 @@ void ihevc_itrans_res_4x4(WORD16 *pi2_src, pi2_tmp = pi2_tmp_orig; /* Inverse Transform 2nd stage */ - shift = IT_SHIFT_STAGE_2; + shift = 20 - bit_depth; add = 1 << (shift - 1); for(j = 0; j < trans_size; j++) @@ -240,7 +242,8 @@ void ihevc_itrans_res_4x4(WORD16 *pi2_src, void ihevc_itrans_res_dc(WORD16 *pi2_dst, WORD32 dst_strd, WORD32 log2_trans_size, - WORD16 i2_coeff_value) + WORD16 i2_coeff_value, + UWORD8 bit_depth) { WORD32 row, col; WORD32 add, shift; @@ -254,7 +257,7 @@ void ihevc_itrans_res_dc(WORD16 *pi2_dst, shift = IT_SHIFT_STAGE_1; add = 1 << (shift - 1); dc_value = CLIP_S16((quant_out * 64 + add) >> shift); - shift = IT_SHIFT_STAGE_2; + shift = 20 - bit_depth; add = 1 << (shift - 1); dc_value = CLIP_S16((dc_value * 64 + add) >> shift); @@ -271,7 +274,8 @@ void ihevc_itrans_res_8x8(WORD16 *pi2_src, WORD32 src_strd, WORD32 dst_strd, WORD32 zero_cols, - WORD32 zero_rows) + WORD32 zero_rows, + UWORD8 bit_depth) { WORD32 j, k; WORD32 e[4], o[4]; @@ -345,7 +349,7 @@ void ihevc_itrans_res_8x8(WORD16 *pi2_src, pi2_tmp = pi2_tmp_orig; /* Inverse Transform 2nd stage */ - shift = IT_SHIFT_STAGE_2; + shift = 20 - bit_depth; add = 1 << (shift - 1); if((zero_rows_2nd_stage & 0xF0) == 0xF0) /* First 4 rows of output of 1st stage are non-zero */ { @@ -486,7 +490,7 @@ void ihevc_itrans_res_8x8(WORD16 *pi2_src, pi2_tmp = pi2_tmp_orig; /* Inverse Transform 2nd stage */ - shift = IT_SHIFT_STAGE_2; + shift = 20 - bit_depth; add = 1 << (shift - 1); if((zero_rows_2nd_stage & 0xF0) == 0xF0) /* First 4 rows of output of 1st stage are non-zero */ { @@ -575,7 +579,8 @@ void ihevc_itrans_res_16x16(WORD16 *pi2_src, WORD32 src_strd, WORD32 dst_strd, WORD32 zero_cols, - WORD32 zero_rows) + WORD32 zero_rows, + UWORD8 bit_depth) { WORD32 j, k; WORD32 e[8], o[8]; @@ -659,7 +664,7 @@ void ihevc_itrans_res_16x16(WORD16 *pi2_src, pi2_tmp = pi2_tmp_orig; /* Inverse Transform 2nd stage */ - shift = IT_SHIFT_STAGE_2; + shift = 20 - bit_depth; add = 1 << (shift - 1); if((zero_rows_2nd_stage & 0xFFF0) == 0xFFF0) /* First 4 rows of output of 1st stage are non-zero */ @@ -897,7 +902,7 @@ void ihevc_itrans_res_16x16(WORD16 *pi2_src, pi2_tmp = pi2_tmp_orig; /* Inverse Transform 2nd stage */ - shift = IT_SHIFT_STAGE_2; + shift = 20 - bit_depth; add = 1 << (shift - 1); if((zero_rows_2nd_stage & 0xFFF0) == 0xFFF0) /* First 4 rows of output of 1st stage are non-zero */ @@ -1159,7 +1164,7 @@ void ihevc_itrans_res_16x16(WORD16 *pi2_src, pi2_tmp = pi2_tmp_orig; /* Inverse Transform 2nd stage */ - shift = IT_SHIFT_STAGE_2; + shift = 20 - bit_depth; add = 1 << (shift - 1); if((zero_rows_2nd_stage & 0xFFF0) == 0xFFF0) /* First 4 rows of output of 1st stage are non-zero */ @@ -1339,7 +1344,8 @@ void ihevc_itrans_res_32x32(WORD16 *pi2_src, WORD32 src_strd, WORD32 dst_strd, WORD32 zero_cols, - WORD32 zero_rows) + WORD32 zero_rows, + UWORD8 bit_depth) { WORD32 j, k; WORD32 e[16], o[16]; @@ -1435,7 +1441,7 @@ void ihevc_itrans_res_32x32(WORD16 *pi2_src, pi2_tmp = pi2_tmp_orig; /* Inverse Transform 2nd stage */ - shift = IT_SHIFT_STAGE_2; + shift = 20 - bit_depth; add = 1 << (shift - 1); if((zero_rows_2nd_stage & 0xFFFFFFF0) == 0xFFFFFFF0) /* First 4 rows of output of 1st stage are non-zero */ { @@ -1742,7 +1748,7 @@ void ihevc_itrans_res_32x32(WORD16 *pi2_src, pi2_tmp = pi2_tmp_orig; /* Inverse Transform 2nd stage */ - shift = IT_SHIFT_STAGE_2; + shift = 20 - bit_depth; add = 1 << (shift - 1); if((zero_rows_2nd_stage & 0xFFFFFFF0) == 0xFFFFFFF0) /* First 4 rows of output of 1st stage are non-zero */ { @@ -2099,7 +2105,7 @@ void ihevc_itrans_res_32x32(WORD16 *pi2_src, pi2_tmp = pi2_tmp_orig; /* Inverse Transform 2nd stage */ - shift = IT_SHIFT_STAGE_2; + shift = 20 - bit_depth; add = 1 << (shift - 1); if((zero_rows_2nd_stage & 0xFFFFFFF0) == 0xFFFFFFF0) /* First 4 rows of output of 1st stage are non-zero */ { diff --git a/common/ihevc_itrans_res.h b/common/ihevc_itrans_res.h index 3e3d7cd..1966356 100644 --- a/common/ihevc_itrans_res.h +++ b/common/ihevc_itrans_res.h @@ -40,7 +40,8 @@ typedef void ihevc_itrans_res_4x4_ttype1_ft(WORD16 *pi2_src, WORD32 src_strd, WORD32 dst_strd, WORD32 zero_cols, - WORD32 zero_rows); + WORD32 zero_rows, + UWORD8 bit_depth); typedef void ihevc_itrans_res_4x4_ft(WORD16 *pi2_src, WORD16 *pi2_tmp, @@ -48,12 +49,14 @@ typedef void ihevc_itrans_res_4x4_ft(WORD16 *pi2_src, WORD32 src_strd, WORD32 dst_strd, WORD32 zero_cols, - WORD32 zero_rows); + WORD32 zero_rows, + UWORD8 bit_depth); typedef void ihevc_itrans_res_dc_ft(WORD16 *pi2_dst, WORD32 dst_strd, WORD32 log2_trans_size, - WORD16 i2_coeff_value); + WORD16 i2_coeff_value, + UWORD8 bit_depth); typedef void ihevc_itrans_res_8x8_ft(WORD16 *pi2_src, WORD16 *pi2_tmp, @@ -61,7 +64,8 @@ typedef void ihevc_itrans_res_8x8_ft(WORD16 *pi2_src, WORD32 src_strd, WORD32 dst_strd, WORD32 zero_cols, - WORD32 zero_rows); + WORD32 zero_rows, + UWORD8 bit_depth); typedef void ihevc_itrans_res_16x16_ft(WORD16 *pi2_src, WORD16 *pi2_tmp, @@ -69,7 +73,8 @@ typedef void ihevc_itrans_res_16x16_ft(WORD16 *pi2_src, WORD32 src_strd, WORD32 dst_strd, WORD32 zero_cols, - WORD32 zero_rows); + WORD32 zero_rows, + UWORD8 bit_depth); typedef void ihevc_itrans_res_32x32_ft(WORD16 *pi2_src, WORD16 *pi2_tmp, @@ -77,58 +82,8 @@ typedef void ihevc_itrans_res_32x32_ft(WORD16 *pi2_src, WORD32 src_strd, WORD32 dst_strd, WORD32 zero_cols, - WORD32 zero_rows); - -typedef void ihevc_hbd_itrans_res_4x4_ttype1_ft(WORD16 *pi2_src, - WORD16 *pi2_tmp, - WORD16 *pi2_dst, - WORD32 src_strd, - WORD32 dst_strd, - WORD32 zero_cols, - WORD32 zero_rows, - UWORD8 bit_depth); - -typedef void ihevc_hbd_itrans_res_4x4_ft(WORD16 *pi2_src, - WORD16 *pi2_tmp, - WORD16 *pi2_dst, - WORD32 src_strd, - WORD32 dst_strd, - WORD32 zero_cols, - WORD32 zero_rows, - UWORD8 bit_depth); - -typedef void ihevc_hbd_itrans_res_dc_ft(WORD16 *pi2_dst, - WORD32 dst_strd, - WORD32 log2_trans_size, - WORD16 i2_coeff_value, - UWORD8 bit_depth); - -typedef void ihevc_hbd_itrans_res_8x8_ft(WORD16 *pi2_src, - WORD16 *pi2_tmp, - WORD16 *pi2_dst, - WORD32 src_strd, - WORD32 dst_strd, - WORD32 zero_cols, - WORD32 zero_rows, - UWORD8 bit_depth); - -typedef void ihevc_hbd_itrans_res_16x16_ft(WORD16 *pi2_src, - WORD16 *pi2_tmp, - WORD16 *pi2_dst, - WORD32 src_strd, - WORD32 dst_strd, - WORD32 zero_cols, - WORD32 zero_rows, - UWORD8 bit_depth); - -typedef void ihevc_hbd_itrans_res_32x32_ft(WORD16 *pi2_src, - WORD16 *pi2_tmp, - WORD16 *pi2_dst, - WORD32 src_strd, - WORD32 dst_strd, - WORD32 zero_cols, - WORD32 zero_rows, - UWORD8 bit_depth); + WORD32 zero_rows, + UWORD8 bit_depth); typedef void ihevc_res_4x4_transform(WORD16 *pi2_src, WORD16 *pi2_dst, @@ -151,13 +106,6 @@ ihevc_itrans_res_8x8_ft ihevc_itrans_res_8x8; ihevc_itrans_res_16x16_ft ihevc_itrans_res_16x16; ihevc_itrans_res_32x32_ft ihevc_itrans_res_32x32; -ihevc_hbd_itrans_res_4x4_ttype1_ft ihevc_hbd_itrans_res_4x4_ttype1; -ihevc_hbd_itrans_res_4x4_ft ihevc_hbd_itrans_res_4x4; -ihevc_hbd_itrans_res_dc_ft ihevc_hbd_itrans_res_dc; -ihevc_hbd_itrans_res_8x8_ft ihevc_hbd_itrans_res_8x8; -ihevc_hbd_itrans_res_16x16_ft ihevc_hbd_itrans_res_16x16; -ihevc_hbd_itrans_res_32x32_ft ihevc_hbd_itrans_res_32x32; - ihevc_res_4x4_transform ihevc_res_4x4_rotate; ihevc_res_nxn_transform ihevc_res_nxn_copy; ihevc_res_nxn_transform ihevc_res_nxn_rdpcm_horz; diff --git a/decoder/ihevcd_api.c b/decoder/ihevcd_api.c index f08f6ec..f33dd5f 100644 --- a/decoder/ihevcd_api.c +++ b/decoder/ihevcd_api.c @@ -1060,14 +1060,6 @@ void ihevcd_update_function_ptr(codec_t *ps_codec) ps_codec->apf_hbd_intra_pred_chroma[9] = (pf_hbd_intra_pred_chroma)ps_codec->s_func_selector.ihevc_hbd_intra_pred_chroma_ver_fptr; ps_codec->apf_hbd_intra_pred_chroma[10] = (pf_hbd_intra_pred_chroma)ps_codec->s_func_selector.ihevc_hbd_intra_pred_chroma_mode_27_to_33_fptr; - ps_codec->apf_hbd_itrans_res[0] = (pf_hbd_itrans_res)ps_codec->s_func_selector.ihevc_hbd_itrans_res_4x4_ttype1_fptr; - ps_codec->apf_hbd_itrans_res[1] = (pf_hbd_itrans_res)ps_codec->s_func_selector.ihevc_hbd_itrans_res_4x4_fptr; - ps_codec->apf_hbd_itrans_res[2] = (pf_hbd_itrans_res)ps_codec->s_func_selector.ihevc_hbd_itrans_res_8x8_fptr; - ps_codec->apf_hbd_itrans_res[3] = (pf_hbd_itrans_res)ps_codec->s_func_selector.ihevc_hbd_itrans_res_16x16_fptr; - ps_codec->apf_hbd_itrans_res[4] = (pf_hbd_itrans_res)ps_codec->s_func_selector.ihevc_hbd_itrans_res_32x32_fptr; - - ps_codec->apf_hbd_itrans_res_dc = (pf_hbd_itrans_res_dc)ps_codec->s_func_selector.ihevc_hbd_itrans_res_dc_fptr; - ps_codec->apf_hbd_itrans_recon[0] = (pf_hbd_itrans_recon)ps_codec->s_func_selector.ihevc_hbd_itrans_recon_4x4_ttype1_fptr; ps_codec->apf_hbd_itrans_recon[1] = (pf_hbd_itrans_recon)ps_codec->s_func_selector.ihevc_hbd_itrans_recon_4x4_fptr; ps_codec->apf_hbd_itrans_recon[2] = (pf_hbd_itrans_recon)ps_codec->s_func_selector.ihevc_hbd_itrans_recon_8x8_fptr; diff --git a/decoder/ihevcd_function_selector.h b/decoder/ihevcd_function_selector.h index 799b474..3ad3bcf 100644 --- a/decoder/ihevcd_function_selector.h +++ b/decoder/ihevcd_function_selector.h @@ -194,13 +194,6 @@ typedef struct ihevc_hbd_chroma_itrans_recon_16x16_ft *ihevc_hbd_chroma_itrans_recon_16x16_fptr; ihevc_hbd_chroma_itrans_recon_32x32_ft *ihevc_hbd_chroma_itrans_recon_32x32_fptr; - ihevc_hbd_itrans_res_4x4_ttype1_ft *ihevc_hbd_itrans_res_4x4_ttype1_fptr; - ihevc_hbd_itrans_res_4x4_ft *ihevc_hbd_itrans_res_4x4_fptr; - ihevc_hbd_itrans_res_8x8_ft *ihevc_hbd_itrans_res_8x8_fptr; - ihevc_hbd_itrans_res_16x16_ft *ihevc_hbd_itrans_res_16x16_fptr; - ihevc_hbd_itrans_res_32x32_ft *ihevc_hbd_itrans_res_32x32_fptr; - ihevc_hbd_itrans_res_dc_ft *ihevc_hbd_itrans_res_dc_fptr; - ihevc_hbd_recon_4x4_ttype1_ft *ihevc_hbd_recon_4x4_ttype1_fptr; ihevc_hbd_recon_4x4_ft *ihevc_hbd_recon_4x4_fptr; ihevc_hbd_recon_8x8_ft *ihevc_hbd_recon_8x8_fptr; diff --git a/decoder/ihevcd_function_selector_generic.c b/decoder/ihevcd_function_selector_generic.c index c10dcc3..8172d14 100644 --- a/decoder/ihevcd_function_selector_generic.c +++ b/decoder/ihevcd_function_selector_generic.c @@ -212,13 +212,6 @@ void ihevcd_init_function_ptr_generic(func_selector_t *ps_func_selector) ps_func_selector->ihevcd_hbd_itrans_recon_dc_luma_fptr = &ihevcd_hbd_itrans_recon_dc_luma; ps_func_selector->ihevcd_hbd_itrans_recon_dc_chroma_fptr = &ihevcd_hbd_itrans_recon_dc_chroma; - ps_func_selector->ihevc_hbd_itrans_res_4x4_ttype1_fptr = &ihevc_hbd_itrans_res_4x4_ttype1; - ps_func_selector->ihevc_hbd_itrans_res_4x4_fptr = &ihevc_hbd_itrans_res_4x4; - ps_func_selector->ihevc_hbd_itrans_res_8x8_fptr = &ihevc_hbd_itrans_res_8x8; - ps_func_selector->ihevc_hbd_itrans_res_16x16_fptr = &ihevc_hbd_itrans_res_16x16; - ps_func_selector->ihevc_hbd_itrans_res_32x32_fptr = &ihevc_hbd_itrans_res_32x32; - ps_func_selector->ihevc_hbd_itrans_res_dc_fptr = &ihevc_hbd_itrans_res_dc; - ps_func_selector->ihevc_hbd_deblk_luma_vert_fptr = &ihevc_hbd_deblk_luma_vert; ps_func_selector->ihevc_hbd_deblk_luma_horz_fptr = &ihevc_hbd_deblk_luma_horz; ps_func_selector->ihevc_hbd_deblk_chroma_vert_fptr = &ihevc_hbd_deblk_chroma_vert; diff --git a/decoder/ihevcd_iquant_itrans_recon_ctb.c b/decoder/ihevcd_iquant_itrans_recon_ctb.c index 35e364b..2b4f81d 100644 --- a/decoder/ihevcd_iquant_itrans_recon_ctb.c +++ b/decoder/ihevcd_iquant_itrans_recon_ctb.c @@ -809,33 +809,16 @@ static void ihevcd_iquant_itrans_resi_recon_tu_plane(process_ctxt_t *ps_proc, if(0 == ps_pl_tu_ctxt->coeff_type) { WORD32 func_tmp_idx = chroma_plane != NULL_PLANE ? func_idx - 4 : func_idx; - if(ps_codec->i4_pixel_size_y > 1) { - ps_codec->apf_hbd_itrans_res[func_tmp_idx](ps_pl_tu_ctxt->pi2_tu_coeff, - ps_proc->pi2_itrans_intrmd_buf, residue_out, - ps_pl_tu_ctxt->tu_coeff_stride, trans_size, - ps_pl_tu_ctxt->zero_cols, - ps_pl_tu_ctxt->zero_rows, (UWORD8)bit_depth); - } - else - { - ps_codec->apf_itrans_res[func_tmp_idx](ps_pl_tu_ctxt->pi2_tu_coeff, - ps_proc->pi2_itrans_intrmd_buf, residue_out, - ps_pl_tu_ctxt->tu_coeff_stride, trans_size, - ps_pl_tu_ctxt->zero_cols, - ps_pl_tu_ctxt->zero_rows); - } + ps_codec->apf_itrans_res[func_tmp_idx](ps_pl_tu_ctxt->pi2_tu_coeff, + ps_proc->pi2_itrans_intrmd_buf, residue_out, + ps_pl_tu_ctxt->tu_coeff_stride, trans_size, + ps_pl_tu_ctxt->zero_cols, + ps_pl_tu_ctxt->zero_rows, (UWORD8)bit_depth); } else /* DC only */ { - if(ps_codec->i4_pixel_size_y > 1) { - ps_codec->apf_hbd_itrans_res_dc(residue_out, trans_size, log2_trans_size, - ps_pl_tu_ctxt->coeff_value, (UWORD8)bit_depth); - } - else - { - ps_codec->apf_itrans_res_dc(residue_out, trans_size, log2_trans_size, - ps_pl_tu_ctxt->coeff_value); - } + ps_codec->apf_itrans_res_dc(residue_out, trans_size, log2_trans_size, + ps_pl_tu_ctxt->coeff_value, (UWORD8)bit_depth); } ps_pl_tu_ctxt->zero_cols = 0; } diff --git a/decoder/ihevcd_structs.h b/decoder/ihevcd_structs.h index d944399..9f136d6 100644 --- a/decoder/ihevcd_structs.h +++ b/decoder/ihevcd_structs.h @@ -1639,7 +1639,8 @@ typedef void (*pf_itrans_res)(WORD16 *pi2_src, WORD32 i4_src_strd, WORD32 i4_dst_strd, WORD32 zero_cols, - WORD32 zero_rows); + WORD32 zero_rows, + UWORD8 bit_depth); typedef void (*pf_itrans_recon)(WORD16 *pi2_src, WORD16 *pi2_tmp, @@ -1669,7 +1670,8 @@ typedef void (*pf_itrans_recon_dc)(UWORD8 *pu1_pred, typedef void (*pf_itrans_res_dc)(WORD16 *pi2_dst, WORD32 dst_strd, WORD32 log2_trans_size, - WORD16 i2_coeff_value); + WORD16 i2_coeff_value, + UWORD8 bit_depth); typedef void (*pf_sao_luma)(UWORD8 *, @@ -1716,21 +1718,6 @@ typedef void (*pf_hbd_itrans_recon)(WORD16 *pi2_src, WORD32 i4_zero_rows, UWORD8 u1_bit_depth); -typedef void (*pf_hbd_itrans_res)(WORD16 *pi2_src, - WORD16 *pi2_tmp, - WORD16 *pi2_dst, - WORD32 src_strd, - WORD32 dst_strd, - WORD32 zero_cols, - WORD32 zero_rows, - UWORD8 bit_depth); - -typedef void (*pf_hbd_itrans_res_dc)(WORD16 *pi2_dst, - WORD32 dst_strd, - WORD32 log2_trans_size, - WORD16 i2_coeff_value, - UWORD8 bit_depth); - typedef void (*pf_hbd_recon)(WORD16 *pi2_src, UWORD16 *pu2_pred, UWORD16 *pu2_dst, @@ -2442,8 +2429,6 @@ struct _codec_t pf_hbd_sao_luma apf_hbd_sao_luma[4]; pf_hbd_sao_chroma apf_hbd_sao_chroma[4]; pf_hbd_inter_pred apf_hbd_inter_pred[22]; - pf_hbd_itrans_res apf_hbd_itrans_res[5]; - pf_hbd_itrans_res_dc apf_hbd_itrans_res_dc; /** Funtion pointers for all the leaf level functions */ func_selector_t s_func_selector; diff --git a/tests/common/ihevc_itrans_res_test.cc b/tests/common/ihevc_itrans_res_test.cc index 9d707d0..a664df0 100644 --- a/tests/common/ihevc_itrans_res_test.cc +++ b/tests/common/ihevc_itrans_res_test.cc @@ -72,11 +72,11 @@ protected: ref = get_ref_func_ptr(); } - template void RunTest(FuncPtr func_ptr) { + template void RunTest(FuncPtr func_ptr, UWORD8 bit_depth) { (ref->*func_ptr)(src_buf.data(), tmp_buf.data(), dst_buf_ref.data(), - src_strd, dst_strd, 0, 0); + src_strd, dst_strd, 0, 0, bit_depth); (tst->*func_ptr)(src_buf.data(), tmp_buf.data(), dst_buf_tst.data(), - src_strd, dst_strd, 0, 0); + src_strd, dst_strd, 0, 0, bit_depth); ASSERT_NO_FATAL_FAILURE(compare_output( dst_buf_ref, dst_buf_tst, trans_size, trans_size, dst_strd)); @@ -95,18 +95,20 @@ protected: }; TEST_P(ITransResTest, Run) { - if (trans_size == 4) { - if (ttype == 1) { - RunTest(&ihevc_func_selector_t::ihevc_itrans_res_4x4_ttype1_fptr); - } else { - RunTest(&ihevc_func_selector_t::ihevc_itrans_res_4x4_fptr); + for (UWORD8 bit_depth : {8, 10}) { + if (trans_size == 4) { + if (ttype == 1) { + RunTest(&ihevc_func_selector_t::ihevc_itrans_res_4x4_ttype1_fptr, bit_depth); + } else { + RunTest(&ihevc_func_selector_t::ihevc_itrans_res_4x4_fptr, bit_depth); + } + } else if (trans_size == 8) { + RunTest(&ihevc_func_selector_t::ihevc_itrans_res_8x8_fptr, bit_depth); + } else if (trans_size == 16) { + RunTest(&ihevc_func_selector_t::ihevc_itrans_res_16x16_fptr, bit_depth); + } else if (trans_size == 32) { + RunTest(&ihevc_func_selector_t::ihevc_itrans_res_32x32_fptr, bit_depth); } - } else if (trans_size == 8) { - RunTest(&ihevc_func_selector_t::ihevc_itrans_res_8x8_fptr); - } else if (trans_size == 16) { - RunTest(&ihevc_func_selector_t::ihevc_itrans_res_16x16_fptr); - } else if (trans_size == 32) { - RunTest(&ihevc_func_selector_t::ihevc_itrans_res_32x32_fptr); } }