mirror of
https://repo.dactyloidae.xyz/Dactyloidae/UXP.git
synced 2026-10-09 00:37:32 +09:00
Update libaom to rev b25610052a1398032320008d69b51d2da94f5928
This commit is contained in:
parent
8493557d20
commit
3d109dd4fc
240 changed files with 16977 additions and 6139 deletions
844
third_party/aom/av1/common/arm/av1_inv_txfm_neon.c
vendored
Normal file
844
third_party/aom/av1/common/arm/av1_inv_txfm_neon.c
vendored
Normal file
|
|
@ -0,0 +1,844 @@
|
|||
/*
|
||||
* Copyright (c) 2018, Alliance for Open Media. All rights reserved
|
||||
*
|
||||
* This source code is subject to the terms of the BSD 2 Clause License and
|
||||
* the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
|
||||
* was not distributed with this source code in the LICENSE file, you can
|
||||
* obtain it at www.aomedia.org/license/software. If the Alliance for Open
|
||||
* Media Patent License 1.0 was not distributed with this source code in the
|
||||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#include "config/aom_config.h"
|
||||
#include "config/aom_dsp_rtcd.h"
|
||||
#include "config/av1_rtcd.h"
|
||||
|
||||
#include "av1/common/av1_inv_txfm1d.h"
|
||||
#include "av1/common/av1_inv_txfm1d_cfg.h"
|
||||
#include "av1/common/av1_txfm.h"
|
||||
#include "av1/common/enums.h"
|
||||
#include "av1/common/idct.h"
|
||||
#include "av1/common/arm/av1_inv_txfm_neon.h"
|
||||
|
||||
static INLINE TxSetType find_TxSetType(TX_SIZE tx_size) {
|
||||
const TX_SIZE tx_size_sqr_up = txsize_sqr_up_map[tx_size];
|
||||
TxSetType tx_set_type;
|
||||
if (tx_size_sqr_up > TX_32X32) {
|
||||
tx_set_type = EXT_TX_SET_DCTONLY;
|
||||
} else if (tx_size_sqr_up == TX_32X32) {
|
||||
tx_set_type = EXT_TX_SET_DCT_IDTX;
|
||||
} else {
|
||||
tx_set_type = EXT_TX_SET_ALL16;
|
||||
}
|
||||
return tx_set_type;
|
||||
}
|
||||
|
||||
// 1D itx types
|
||||
typedef enum ATTRIBUTE_PACKED {
|
||||
IDCT_1D,
|
||||
IADST_1D,
|
||||
IFLIPADST_1D = IADST_1D,
|
||||
IIDENTITY_1D,
|
||||
ITX_TYPES_1D,
|
||||
} ITX_TYPE_1D;
|
||||
|
||||
static const ITX_TYPE_1D vitx_1d_tab[TX_TYPES] = {
|
||||
IDCT_1D, IADST_1D, IDCT_1D, IADST_1D,
|
||||
IFLIPADST_1D, IDCT_1D, IFLIPADST_1D, IADST_1D,
|
||||
IFLIPADST_1D, IIDENTITY_1D, IDCT_1D, IIDENTITY_1D,
|
||||
IADST_1D, IIDENTITY_1D, IFLIPADST_1D, IIDENTITY_1D,
|
||||
};
|
||||
|
||||
static const ITX_TYPE_1D hitx_1d_tab[TX_TYPES] = {
|
||||
IDCT_1D, IDCT_1D, IADST_1D, IADST_1D,
|
||||
IDCT_1D, IFLIPADST_1D, IFLIPADST_1D, IFLIPADST_1D,
|
||||
IADST_1D, IIDENTITY_1D, IIDENTITY_1D, IDCT_1D,
|
||||
IIDENTITY_1D, IADST_1D, IIDENTITY_1D, IFLIPADST_1D,
|
||||
};
|
||||
|
||||
// 1D functions
|
||||
static const transform_1d_neon lowbd_txfm_all_1d_arr[TX_SIZES][ITX_TYPES_1D] = {
|
||||
{ av1_idct4_new, av1_iadst4_new, av1_iidentity4_c },
|
||||
{ av1_idct8_new, av1_iadst8_new, av1_iidentity8_c },
|
||||
{ av1_idct16_new, av1_iadst16_new, av1_iidentity16_c },
|
||||
{ av1_idct32_new, NULL, NULL },
|
||||
{ av1_idct64_new, NULL, NULL },
|
||||
};
|
||||
|
||||
// Functions for blocks with eob at DC and within
|
||||
// topleft 8x8, 16x16, 32x32 corner
|
||||
static const transform_1d_neon
|
||||
lowbd_txfm_all_1d_zeros_w8_arr[TX_SIZES][ITX_TYPES_1D][4] = {
|
||||
{
|
||||
{ av1_idct4_new, av1_idct4_new, NULL, NULL },
|
||||
{ av1_iadst4_new, av1_iadst4_new, NULL, NULL },
|
||||
{ av1_iidentity4_c, av1_iidentity4_c, NULL, NULL },
|
||||
},
|
||||
{ { av1_idct8_new, av1_idct8_new, NULL, NULL },
|
||||
{ av1_iadst8_new, av1_iadst8_new, NULL, NULL },
|
||||
{ av1_iidentity8_c, av1_iidentity8_c, NULL, NULL } },
|
||||
{
|
||||
{ av1_idct16_new, av1_idct16_new, av1_idct16_new, NULL },
|
||||
{ av1_iadst16_new, av1_iadst16_new, av1_iadst16_new, NULL },
|
||||
{ av1_iidentity16_c, av1_iidentity16_c, av1_iidentity16_c, NULL },
|
||||
},
|
||||
{ { av1_idct32_new, av1_idct32_new, av1_idct32_new, av1_idct32_new },
|
||||
{ NULL, NULL, NULL, NULL },
|
||||
{ av1_iidentity32_c, av1_iidentity32_c, av1_iidentity32_c,
|
||||
av1_iidentity32_c } },
|
||||
{ { av1_idct64_new, av1_idct64_new, av1_idct64_new, av1_idct64_new },
|
||||
{ NULL, NULL, NULL, NULL },
|
||||
{ NULL, NULL, NULL, NULL } }
|
||||
};
|
||||
static INLINE void lowbd_inv_txfm2d_add_idtx_neon(const int32_t *input,
|
||||
uint8_t *output, int stride,
|
||||
TX_TYPE tx_type,
|
||||
TX_SIZE tx_size, int eob) {
|
||||
DECLARE_ALIGNED(32, int, txfm_buf[32 * 32 + 32 + 32]);
|
||||
int32_t *temp_in = txfm_buf;
|
||||
|
||||
int eobx, eoby;
|
||||
get_eobx_eoby_scan_default(&eobx, &eoby, tx_size, eob);
|
||||
const int8_t *shift = inv_txfm_shift_ls[tx_size];
|
||||
const int txw_idx = get_txw_idx(tx_size);
|
||||
const int txh_idx = get_txh_idx(tx_size);
|
||||
const int cos_bit_col = inv_cos_bit_col[txw_idx][txh_idx];
|
||||
const int cos_bit_row = inv_cos_bit_row[txw_idx][txh_idx];
|
||||
const int txfm_size_col = tx_size_wide[tx_size];
|
||||
const int txfm_size_row = tx_size_high[tx_size];
|
||||
const int buf_size_nonzero_h_div8 = (eoby + 8) >> 3;
|
||||
|
||||
const int rect_type = get_rect_tx_log_ratio(txfm_size_col, txfm_size_row);
|
||||
const int buf_offset = AOMMAX(txfm_size_row, txfm_size_col);
|
||||
|
||||
int32_t *temp_out = temp_in + buf_offset;
|
||||
int32_t *buf = temp_out + buf_offset;
|
||||
int32_t *buf_ptr = buf;
|
||||
const int8_t stage_range[MAX_TXFM_STAGE_NUM] = { 16 };
|
||||
int r, bd = 8;
|
||||
|
||||
const int fun_idx_x = lowbd_txfm_all_1d_zeros_idx[eobx];
|
||||
const int fun_idx_y = lowbd_txfm_all_1d_zeros_idx[eoby];
|
||||
const transform_1d_neon row_txfm =
|
||||
lowbd_txfm_all_1d_zeros_w8_arr[txw_idx][hitx_1d_tab[tx_type]][fun_idx_x];
|
||||
const transform_1d_neon col_txfm =
|
||||
lowbd_txfm_all_1d_zeros_w8_arr[txh_idx][vitx_1d_tab[tx_type]][fun_idx_y];
|
||||
|
||||
assert(col_txfm != NULL);
|
||||
assert(row_txfm != NULL);
|
||||
|
||||
// row tx
|
||||
int row_start = (buf_size_nonzero_h_div8 * 8);
|
||||
for (int i = 0; i < row_start; i++) {
|
||||
if (abs(rect_type) == 1) {
|
||||
for (int j = 0; j < txfm_size_col; j++)
|
||||
temp_in[j] = round_shift((int64_t)input[j] * NewInvSqrt2, NewSqrt2Bits);
|
||||
row_txfm(temp_in, buf_ptr, cos_bit_row, stage_range);
|
||||
} else {
|
||||
row_txfm(input, buf_ptr, cos_bit_row, stage_range);
|
||||
}
|
||||
av1_round_shift_array(buf_ptr, txfm_size_col, -shift[0]);
|
||||
input += txfm_size_col;
|
||||
buf_ptr += txfm_size_col;
|
||||
}
|
||||
|
||||
// Doing memset for the rows which are not processed in row transform.
|
||||
memset(buf_ptr, 0,
|
||||
sizeof(int32_t) * txfm_size_col * (txfm_size_row - row_start));
|
||||
|
||||
// col tx
|
||||
for (int c = 0; c < txfm_size_col; c++) {
|
||||
for (r = 0; r < txfm_size_row; ++r) temp_in[r] = buf[r * txfm_size_col + c];
|
||||
|
||||
col_txfm(temp_in, temp_out, cos_bit_col, stage_range);
|
||||
av1_round_shift_array(temp_out, txfm_size_row, -shift[1]);
|
||||
|
||||
for (r = 0; r < txfm_size_row; ++r) {
|
||||
output[r * stride + c] =
|
||||
highbd_clip_pixel_add(output[r * stride + c], temp_out[r], bd);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static INLINE void lowbd_inv_txfm2d_add_v_identity_neon(
|
||||
const int32_t *input, uint8_t *output, int stride, TX_TYPE tx_type,
|
||||
TX_SIZE tx_size, int eob) {
|
||||
DECLARE_ALIGNED(32, int, txfm_buf[32 * 32 + 32 + 32]);
|
||||
int32_t *temp_in = txfm_buf;
|
||||
|
||||
int eobx, eoby;
|
||||
get_eobx_eoby_scan_v_identity(&eobx, &eoby, tx_size, eob);
|
||||
const int8_t *shift = inv_txfm_shift_ls[tx_size];
|
||||
const int txw_idx = get_txw_idx(tx_size);
|
||||
const int txh_idx = get_txh_idx(tx_size);
|
||||
const int cos_bit_col = inv_cos_bit_col[txw_idx][txh_idx];
|
||||
const int cos_bit_row = inv_cos_bit_row[txw_idx][txh_idx];
|
||||
const int txfm_size_col = tx_size_wide[tx_size];
|
||||
const int txfm_size_row = tx_size_high[tx_size];
|
||||
const int buf_size_nonzero_h_div8 = (eoby + 8) >> 3;
|
||||
|
||||
const int rect_type = get_rect_tx_log_ratio(txfm_size_col, txfm_size_row);
|
||||
const int buf_offset = AOMMAX(txfm_size_row, txfm_size_col);
|
||||
|
||||
int32_t *temp_out = temp_in + buf_offset;
|
||||
int32_t *buf = temp_out + buf_offset;
|
||||
int32_t *buf_ptr = buf;
|
||||
const int8_t stage_range[MAX_TXFM_STAGE_NUM] = { 16 };
|
||||
int r, bd = 8;
|
||||
|
||||
const int fun_idx_x = lowbd_txfm_all_1d_zeros_idx[eobx];
|
||||
const int fun_idx_y = lowbd_txfm_all_1d_zeros_idx[eoby];
|
||||
const transform_1d_neon row_txfm =
|
||||
lowbd_txfm_all_1d_zeros_w8_arr[txw_idx][hitx_1d_tab[tx_type]][fun_idx_x];
|
||||
const transform_1d_neon col_txfm =
|
||||
lowbd_txfm_all_1d_zeros_w8_arr[txh_idx][vitx_1d_tab[tx_type]][fun_idx_y];
|
||||
|
||||
assert(col_txfm != NULL);
|
||||
assert(row_txfm != NULL);
|
||||
int ud_flip, lr_flip;
|
||||
get_flip_cfg(tx_type, &ud_flip, &lr_flip);
|
||||
|
||||
// row tx
|
||||
int row_start = (buf_size_nonzero_h_div8 * 8);
|
||||
for (int i = 0; i < row_start; i++) {
|
||||
if (abs(rect_type) == 1) {
|
||||
for (int j = 0; j < txfm_size_col; j++)
|
||||
temp_in[j] = round_shift((int64_t)input[j] * NewInvSqrt2, NewSqrt2Bits);
|
||||
row_txfm(temp_in, buf_ptr, cos_bit_row, stage_range);
|
||||
} else {
|
||||
row_txfm(input, buf_ptr, cos_bit_row, stage_range);
|
||||
}
|
||||
av1_round_shift_array(buf_ptr, txfm_size_col, -shift[0]);
|
||||
input += txfm_size_col;
|
||||
buf_ptr += txfm_size_col;
|
||||
}
|
||||
// Doing memset for the rows which are not processed in row transform.
|
||||
memset(buf_ptr, 0,
|
||||
sizeof(int32_t) * txfm_size_col * (txfm_size_row - row_start));
|
||||
|
||||
// col tx
|
||||
for (int c = 0; c < txfm_size_col; c++) {
|
||||
if (lr_flip == 0) {
|
||||
for (r = 0; r < txfm_size_row; ++r)
|
||||
temp_in[r] = buf[r * txfm_size_col + c];
|
||||
} else {
|
||||
// flip left right
|
||||
for (r = 0; r < txfm_size_row; ++r)
|
||||
temp_in[r] = buf[r * txfm_size_col + (txfm_size_col - c - 1)];
|
||||
}
|
||||
col_txfm(temp_in, temp_out, cos_bit_col, stage_range);
|
||||
av1_round_shift_array(temp_out, txfm_size_row, -shift[1]);
|
||||
|
||||
if (ud_flip == 0) {
|
||||
for (r = 0; r < txfm_size_row; ++r) {
|
||||
output[r * stride + c] =
|
||||
highbd_clip_pixel_add(output[r * stride + c], temp_out[r], bd);
|
||||
}
|
||||
} else {
|
||||
// flip upside down
|
||||
for (r = 0; r < txfm_size_row; ++r) {
|
||||
output[r * stride + c] = highbd_clip_pixel_add(
|
||||
output[r * stride + c], temp_out[txfm_size_row - r - 1], bd);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static INLINE void lowbd_inv_txfm2d_add_h_identity_neon(
|
||||
const int32_t *input, uint8_t *output, int stride, TX_TYPE tx_type,
|
||||
TX_SIZE tx_size, int eob) {
|
||||
DECLARE_ALIGNED(32, int, txfm_buf[32 * 32 + 32 + 32]);
|
||||
int32_t *temp_in = txfm_buf;
|
||||
|
||||
int eobx, eoby;
|
||||
get_eobx_eoby_scan_h_identity(&eobx, &eoby, tx_size, eob);
|
||||
const int8_t *shift = inv_txfm_shift_ls[tx_size];
|
||||
const int txw_idx = get_txw_idx(tx_size);
|
||||
const int txh_idx = get_txh_idx(tx_size);
|
||||
const int cos_bit_col = inv_cos_bit_col[txw_idx][txh_idx];
|
||||
const int cos_bit_row = inv_cos_bit_row[txw_idx][txh_idx];
|
||||
const int txfm_size_col = tx_size_wide[tx_size];
|
||||
const int txfm_size_row = tx_size_high[tx_size];
|
||||
const int buf_size_nonzero_h_div8 = (eoby + 8) >> 3;
|
||||
|
||||
const int rect_type = get_rect_tx_log_ratio(txfm_size_col, txfm_size_row);
|
||||
const int buf_offset = AOMMAX(txfm_size_row, txfm_size_col);
|
||||
|
||||
int32_t *temp_out = temp_in + buf_offset;
|
||||
int32_t *buf = temp_out + buf_offset;
|
||||
int32_t *buf_ptr = buf;
|
||||
const int8_t stage_range[MAX_TXFM_STAGE_NUM] = { 16 };
|
||||
int r, bd = 8;
|
||||
|
||||
const int fun_idx_x = lowbd_txfm_all_1d_zeros_idx[eobx];
|
||||
const int fun_idx_y = lowbd_txfm_all_1d_zeros_idx[eoby];
|
||||
const transform_1d_neon row_txfm =
|
||||
lowbd_txfm_all_1d_zeros_w8_arr[txw_idx][hitx_1d_tab[tx_type]][fun_idx_x];
|
||||
const transform_1d_neon col_txfm =
|
||||
lowbd_txfm_all_1d_zeros_w8_arr[txh_idx][vitx_1d_tab[tx_type]][fun_idx_y];
|
||||
|
||||
assert(col_txfm != NULL);
|
||||
assert(row_txfm != NULL);
|
||||
int ud_flip, lr_flip;
|
||||
get_flip_cfg(tx_type, &ud_flip, &lr_flip);
|
||||
|
||||
// row tx
|
||||
int row_start = (buf_size_nonzero_h_div8 * 8);
|
||||
for (int i = 0; i < row_start; i++) {
|
||||
if (abs(rect_type) == 1) {
|
||||
for (int j = 0; j < txfm_size_col; j++)
|
||||
temp_in[j] = round_shift((int64_t)input[j] * NewInvSqrt2, NewSqrt2Bits);
|
||||
row_txfm(temp_in, buf_ptr, cos_bit_row, stage_range);
|
||||
} else {
|
||||
row_txfm(input, buf_ptr, cos_bit_row, stage_range);
|
||||
}
|
||||
av1_round_shift_array(buf_ptr, txfm_size_col, -shift[0]);
|
||||
input += txfm_size_col;
|
||||
buf_ptr += txfm_size_col;
|
||||
}
|
||||
// Doing memset for the rows which are not processed in row transform.
|
||||
memset(buf_ptr, 0,
|
||||
sizeof(int32_t) * txfm_size_col * (txfm_size_row - row_start));
|
||||
|
||||
// col tx
|
||||
for (int c = 0; c < txfm_size_col; c++) {
|
||||
if (lr_flip == 0) {
|
||||
for (r = 0; r < txfm_size_row; ++r)
|
||||
temp_in[r] = buf[r * txfm_size_col + c];
|
||||
} else {
|
||||
// flip left right
|
||||
for (r = 0; r < txfm_size_row; ++r)
|
||||
temp_in[r] = buf[r * txfm_size_col + (txfm_size_col - c - 1)];
|
||||
}
|
||||
col_txfm(temp_in, temp_out, cos_bit_col, stage_range);
|
||||
av1_round_shift_array(temp_out, txfm_size_row, -shift[1]);
|
||||
|
||||
if (ud_flip == 0) {
|
||||
for (r = 0; r < txfm_size_row; ++r) {
|
||||
output[r * stride + c] =
|
||||
highbd_clip_pixel_add(output[r * stride + c], temp_out[r], bd);
|
||||
}
|
||||
} else {
|
||||
// flip upside down
|
||||
for (r = 0; r < txfm_size_row; ++r) {
|
||||
output[r * stride + c] = highbd_clip_pixel_add(
|
||||
output[r * stride + c], temp_out[txfm_size_row - r - 1], bd);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static INLINE void lowbd_inv_txfm2d_add_4x4_neon(const int32_t *input,
|
||||
uint8_t *output, int stride,
|
||||
TX_TYPE tx_type,
|
||||
TX_SIZE tx_size, int eob) {
|
||||
(void)eob;
|
||||
DECLARE_ALIGNED(32, int, txfm_buf[4 * 4 + 8 + 8]);
|
||||
int32_t *temp_in = txfm_buf;
|
||||
|
||||
const int8_t *shift = inv_txfm_shift_ls[tx_size];
|
||||
const int txw_idx = get_txw_idx(tx_size);
|
||||
const int txh_idx = get_txh_idx(tx_size);
|
||||
const int cos_bit_row = inv_cos_bit_row[txw_idx][txh_idx];
|
||||
const int cos_bit_col = inv_cos_bit_col[txw_idx][txh_idx];
|
||||
const int txfm_size_col = tx_size_wide[tx_size];
|
||||
const int txfm_size_row = tx_size_high[tx_size];
|
||||
const int buf_offset = AOMMAX(txfm_size_row, txfm_size_col);
|
||||
int32_t *temp_out = temp_in + buf_offset;
|
||||
int32_t *buf = temp_out + buf_offset;
|
||||
int32_t *buf_ptr = buf;
|
||||
const int8_t stage_range[MAX_TXFM_STAGE_NUM] = { 16 };
|
||||
int r, bd = 8;
|
||||
const transform_1d_neon row_txfm =
|
||||
lowbd_txfm_all_1d_arr[txw_idx][hitx_1d_tab[tx_type]];
|
||||
const transform_1d_neon col_txfm =
|
||||
lowbd_txfm_all_1d_arr[txh_idx][vitx_1d_tab[tx_type]];
|
||||
|
||||
int ud_flip, lr_flip;
|
||||
get_flip_cfg(tx_type, &ud_flip, &lr_flip);
|
||||
|
||||
for (int i = 0; i < txfm_size_row; i++) {
|
||||
row_txfm(input, buf_ptr, cos_bit_row, stage_range);
|
||||
|
||||
input += txfm_size_col;
|
||||
buf_ptr += txfm_size_col;
|
||||
}
|
||||
|
||||
for (int c = 0; c < txfm_size_col; ++c) {
|
||||
if (lr_flip == 0) {
|
||||
for (r = 0; r < txfm_size_row; ++r)
|
||||
temp_in[r] = buf[r * txfm_size_col + c];
|
||||
} else {
|
||||
// flip left right
|
||||
for (r = 0; r < txfm_size_row; ++r)
|
||||
temp_in[r] = buf[r * txfm_size_col + (txfm_size_col - c - 1)];
|
||||
}
|
||||
col_txfm(temp_in, temp_out, cos_bit_col, stage_range);
|
||||
av1_round_shift_array(temp_out, txfm_size_row, -shift[1]);
|
||||
|
||||
if (ud_flip == 0) {
|
||||
for (r = 0; r < txfm_size_row; ++r) {
|
||||
output[r * stride + c] =
|
||||
highbd_clip_pixel_add(output[r * stride + c], temp_out[r], bd);
|
||||
}
|
||||
} else {
|
||||
// flip upside down
|
||||
for (r = 0; r < txfm_size_row; ++r) {
|
||||
output[r * stride + c] = highbd_clip_pixel_add(
|
||||
output[r * stride + c], temp_out[txfm_size_row - r - 1], bd);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void lowbd_inv_txfm2d_add_4x8_neon(const int32_t *input, uint8_t *output,
|
||||
int stride, TX_TYPE tx_type, TX_SIZE tx_size,
|
||||
int eob) {
|
||||
(void)eob;
|
||||
DECLARE_ALIGNED(32, int, txfm_buf[4 * 8 + 8 + 8]);
|
||||
int32_t *temp_in = txfm_buf;
|
||||
|
||||
const int8_t *shift = inv_txfm_shift_ls[tx_size];
|
||||
const int txw_idx = get_txw_idx(tx_size);
|
||||
const int txh_idx = get_txh_idx(tx_size);
|
||||
const int cos_bit_row = inv_cos_bit_row[txw_idx][txh_idx];
|
||||
const int cos_bit_col = inv_cos_bit_col[txw_idx][txh_idx];
|
||||
const int txfm_size_col = tx_size_wide[tx_size];
|
||||
const int txfm_size_row = tx_size_high[tx_size];
|
||||
const int buf_offset = AOMMAX(txfm_size_row, txfm_size_col);
|
||||
int32_t *temp_out = temp_in + buf_offset;
|
||||
int32_t *buf = temp_out + buf_offset;
|
||||
int32_t *buf_ptr = buf;
|
||||
const int8_t stage_range[MAX_TXFM_STAGE_NUM] = { 16 };
|
||||
int r, bd = 8;
|
||||
const transform_1d_neon row_txfm =
|
||||
lowbd_txfm_all_1d_arr[txw_idx][hitx_1d_tab[tx_type]];
|
||||
const transform_1d_neon col_txfm =
|
||||
lowbd_txfm_all_1d_arr[txh_idx][vitx_1d_tab[tx_type]];
|
||||
|
||||
int ud_flip, lr_flip;
|
||||
get_flip_cfg(tx_type, &ud_flip, &lr_flip);
|
||||
|
||||
for (int i = 0; i < txfm_size_row; i++) {
|
||||
for (int j = 0; j < txfm_size_col; j++)
|
||||
temp_in[j] = round_shift((int64_t)input[j] * NewInvSqrt2, NewSqrt2Bits);
|
||||
|
||||
row_txfm(temp_in, buf_ptr, cos_bit_row, stage_range);
|
||||
input += txfm_size_col;
|
||||
buf_ptr += txfm_size_col;
|
||||
}
|
||||
|
||||
for (int c = 0; c < txfm_size_col; ++c) {
|
||||
if (lr_flip == 0) {
|
||||
for (r = 0; r < txfm_size_row; ++r)
|
||||
temp_in[r] = buf[r * txfm_size_col + c];
|
||||
} else {
|
||||
// flip left right
|
||||
for (r = 0; r < txfm_size_row; ++r)
|
||||
temp_in[r] = buf[r * txfm_size_col + (txfm_size_col - c - 1)];
|
||||
}
|
||||
col_txfm(temp_in, temp_out, cos_bit_col, stage_range);
|
||||
av1_round_shift_array(temp_out, txfm_size_row, -shift[1]);
|
||||
|
||||
if (ud_flip == 0) {
|
||||
for (r = 0; r < txfm_size_row; ++r) {
|
||||
output[r * stride + c] =
|
||||
highbd_clip_pixel_add(output[r * stride + c], temp_out[r], bd);
|
||||
}
|
||||
} else {
|
||||
// flip upside down
|
||||
for (r = 0; r < txfm_size_row; ++r) {
|
||||
output[r * stride + c] = highbd_clip_pixel_add(
|
||||
output[r * stride + c], temp_out[txfm_size_row - r - 1], bd);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void lowbd_inv_txfm2d_add_8x4_neon(const int32_t *input, uint8_t *output,
|
||||
int stride, TX_TYPE tx_type, TX_SIZE tx_size,
|
||||
int eob) {
|
||||
(void)eob;
|
||||
DECLARE_ALIGNED(32, int, txfm_buf[8 * 4 + 8 + 8]);
|
||||
int32_t *temp_in = txfm_buf;
|
||||
|
||||
const int8_t *shift = inv_txfm_shift_ls[tx_size];
|
||||
const int txw_idx = get_txw_idx(tx_size);
|
||||
const int txh_idx = get_txh_idx(tx_size);
|
||||
const int cos_bit_row = inv_cos_bit_row[txw_idx][txh_idx];
|
||||
const int cos_bit_col = inv_cos_bit_col[txw_idx][txh_idx];
|
||||
const int txfm_size_col = tx_size_wide[tx_size];
|
||||
const int txfm_size_row = tx_size_high[tx_size];
|
||||
const int buf_offset = AOMMAX(txfm_size_row, txfm_size_col);
|
||||
int32_t *temp_out = temp_in + buf_offset;
|
||||
int32_t *buf = temp_out + buf_offset;
|
||||
int32_t *buf_ptr = buf;
|
||||
const int8_t stage_range[MAX_TXFM_STAGE_NUM] = { 16 };
|
||||
int r, bd = 8;
|
||||
const transform_1d_neon row_txfm =
|
||||
lowbd_txfm_all_1d_arr[txw_idx][hitx_1d_tab[tx_type]];
|
||||
const transform_1d_neon col_txfm =
|
||||
lowbd_txfm_all_1d_arr[txh_idx][vitx_1d_tab[tx_type]];
|
||||
|
||||
int ud_flip, lr_flip;
|
||||
get_flip_cfg(tx_type, &ud_flip, &lr_flip);
|
||||
|
||||
for (int i = 0; i < txfm_size_row; i++) {
|
||||
for (int j = 0; j < txfm_size_col; j++)
|
||||
temp_in[j] = round_shift((int64_t)input[j] * NewInvSqrt2, NewSqrt2Bits);
|
||||
|
||||
row_txfm(temp_in, buf_ptr, cos_bit_row, stage_range);
|
||||
input += txfm_size_col;
|
||||
buf_ptr += txfm_size_col;
|
||||
}
|
||||
|
||||
for (int c = 0; c < txfm_size_col; ++c) {
|
||||
if (lr_flip == 0) {
|
||||
for (r = 0; r < txfm_size_row; ++r)
|
||||
temp_in[r] = buf[r * txfm_size_col + c];
|
||||
} else {
|
||||
// flip left right
|
||||
for (r = 0; r < txfm_size_row; ++r)
|
||||
temp_in[r] = buf[r * txfm_size_col + (txfm_size_col - c - 1)];
|
||||
}
|
||||
col_txfm(temp_in, temp_out, cos_bit_col, stage_range);
|
||||
av1_round_shift_array(temp_out, txfm_size_row, -shift[1]);
|
||||
|
||||
if (ud_flip == 0) {
|
||||
for (r = 0; r < txfm_size_row; ++r) {
|
||||
output[r * stride + c] =
|
||||
highbd_clip_pixel_add(output[r * stride + c], temp_out[r], bd);
|
||||
}
|
||||
} else {
|
||||
// flip upside down
|
||||
for (r = 0; r < txfm_size_row; ++r) {
|
||||
output[r * stride + c] = highbd_clip_pixel_add(
|
||||
output[r * stride + c], temp_out[txfm_size_row - r - 1], bd);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void lowbd_inv_txfm2d_add_4x16_neon(const int32_t *input, uint8_t *output,
|
||||
int stride, TX_TYPE tx_type,
|
||||
TX_SIZE tx_size, int eob) {
|
||||
(void)eob;
|
||||
DECLARE_ALIGNED(32, int, txfm_buf[4 * 16 + 16 + 16]);
|
||||
int32_t *temp_in = txfm_buf;
|
||||
|
||||
const int8_t *shift = inv_txfm_shift_ls[tx_size];
|
||||
const int txw_idx = get_txw_idx(tx_size);
|
||||
const int txh_idx = get_txh_idx(tx_size);
|
||||
const int cos_bit_row = inv_cos_bit_row[txw_idx][txh_idx];
|
||||
const int cos_bit_col = inv_cos_bit_col[txw_idx][txh_idx];
|
||||
const int txfm_size_col = tx_size_wide[tx_size];
|
||||
const int txfm_size_row = tx_size_high[tx_size];
|
||||
const int buf_offset = AOMMAX(txfm_size_row, txfm_size_col);
|
||||
int32_t *temp_out = temp_in + buf_offset;
|
||||
int32_t *buf = temp_out + buf_offset;
|
||||
int32_t *buf_ptr = buf;
|
||||
const int8_t stage_range[MAX_TXFM_STAGE_NUM] = { 16 };
|
||||
int r, bd = 8;
|
||||
const transform_1d_neon row_txfm =
|
||||
lowbd_txfm_all_1d_arr[txw_idx][hitx_1d_tab[tx_type]];
|
||||
const transform_1d_neon col_txfm =
|
||||
lowbd_txfm_all_1d_arr[txh_idx][vitx_1d_tab[tx_type]];
|
||||
|
||||
int ud_flip, lr_flip;
|
||||
get_flip_cfg(tx_type, &ud_flip, &lr_flip);
|
||||
|
||||
for (int i = 0; i < txfm_size_row; i++) {
|
||||
row_txfm(input, buf_ptr, cos_bit_row, stage_range);
|
||||
av1_round_shift_array(buf_ptr, txfm_size_col, -shift[0]);
|
||||
input += txfm_size_col;
|
||||
buf_ptr += txfm_size_col;
|
||||
}
|
||||
|
||||
for (int c = 0; c < txfm_size_col; ++c) {
|
||||
if (lr_flip == 0) {
|
||||
for (r = 0; r < txfm_size_row; ++r)
|
||||
temp_in[r] = buf[r * txfm_size_col + c];
|
||||
} else {
|
||||
// flip left right
|
||||
for (r = 0; r < txfm_size_row; ++r)
|
||||
temp_in[r] = buf[r * txfm_size_col + (txfm_size_col - c - 1)];
|
||||
}
|
||||
col_txfm(temp_in, temp_out, cos_bit_col, stage_range);
|
||||
av1_round_shift_array(temp_out, txfm_size_row, -shift[1]);
|
||||
|
||||
if (ud_flip == 0) {
|
||||
for (r = 0; r < txfm_size_row; ++r) {
|
||||
output[r * stride + c] =
|
||||
highbd_clip_pixel_add(output[r * stride + c], temp_out[r], bd);
|
||||
}
|
||||
} else {
|
||||
// flip upside down
|
||||
for (r = 0; r < txfm_size_row; ++r) {
|
||||
output[r * stride + c] = highbd_clip_pixel_add(
|
||||
output[r * stride + c], temp_out[txfm_size_row - r - 1], bd);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void lowbd_inv_txfm2d_add_16x4_neon(const int32_t *input, uint8_t *output,
|
||||
int stride, TX_TYPE tx_type,
|
||||
TX_SIZE tx_size, int eob) {
|
||||
(void)eob;
|
||||
|
||||
DECLARE_ALIGNED(32, int, txfm_buf[16 * 4 + 16 + 16]);
|
||||
int32_t *temp_in = txfm_buf;
|
||||
|
||||
const int8_t *shift = inv_txfm_shift_ls[tx_size];
|
||||
const int txw_idx = get_txw_idx(tx_size);
|
||||
const int txh_idx = get_txh_idx(tx_size);
|
||||
const int cos_bit_row = inv_cos_bit_row[txw_idx][txh_idx];
|
||||
const int cos_bit_col = inv_cos_bit_col[txw_idx][txh_idx];
|
||||
const int txfm_size_col = tx_size_wide[tx_size];
|
||||
const int txfm_size_row = tx_size_high[tx_size];
|
||||
const int buf_offset = AOMMAX(txfm_size_row, txfm_size_col);
|
||||
int32_t *temp_out = temp_in + buf_offset;
|
||||
int32_t *buf = temp_out + buf_offset;
|
||||
int32_t *buf_ptr = buf;
|
||||
const int8_t stage_range[MAX_TXFM_STAGE_NUM] = { 16 };
|
||||
int r, bd = 8;
|
||||
const transform_1d_neon row_txfm =
|
||||
lowbd_txfm_all_1d_arr[txw_idx][hitx_1d_tab[tx_type]];
|
||||
const transform_1d_neon col_txfm =
|
||||
lowbd_txfm_all_1d_arr[txh_idx][vitx_1d_tab[tx_type]];
|
||||
|
||||
int ud_flip, lr_flip;
|
||||
get_flip_cfg(tx_type, &ud_flip, &lr_flip);
|
||||
|
||||
for (int i = 0; i < txfm_size_row; i++) {
|
||||
row_txfm(input, buf_ptr, cos_bit_row, stage_range);
|
||||
av1_round_shift_array(buf_ptr, txfm_size_col, -shift[0]);
|
||||
input += txfm_size_col;
|
||||
buf_ptr += txfm_size_col;
|
||||
}
|
||||
|
||||
for (int c = 0; c < txfm_size_col; ++c) {
|
||||
if (lr_flip == 0) {
|
||||
for (r = 0; r < txfm_size_row; ++r)
|
||||
temp_in[r] = buf[r * txfm_size_col + c];
|
||||
} else {
|
||||
// flip left right
|
||||
for (r = 0; r < txfm_size_row; ++r)
|
||||
temp_in[r] = buf[r * txfm_size_col + (txfm_size_col - c - 1)];
|
||||
}
|
||||
col_txfm(temp_in, temp_out, cos_bit_col, stage_range);
|
||||
av1_round_shift_array(temp_out, txfm_size_row, -shift[1]);
|
||||
|
||||
if (ud_flip == 0) {
|
||||
for (r = 0; r < txfm_size_row; ++r) {
|
||||
output[r * stride + c] =
|
||||
highbd_clip_pixel_add(output[r * stride + c], temp_out[r], bd);
|
||||
}
|
||||
} else {
|
||||
// flip upside down
|
||||
for (r = 0; r < txfm_size_row; ++r) {
|
||||
output[r * stride + c] = highbd_clip_pixel_add(
|
||||
output[r * stride + c], temp_out[txfm_size_row - r - 1], bd);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static INLINE void lowbd_inv_txfm2d_add_no_identity_neon(
|
||||
const int32_t *input, uint8_t *output, int stride, TX_TYPE tx_type,
|
||||
TX_SIZE tx_size, int eob) {
|
||||
DECLARE_ALIGNED(32, int, txfm_buf[64 * 64 + 64 + 64]);
|
||||
int32_t *temp_in = txfm_buf;
|
||||
|
||||
int eobx, eoby, ud_flip, lr_flip, row_start;
|
||||
get_eobx_eoby_scan_default(&eobx, &eoby, tx_size, eob);
|
||||
const int8_t *shift = inv_txfm_shift_ls[tx_size];
|
||||
const int txw_idx = get_txw_idx(tx_size);
|
||||
const int txh_idx = get_txh_idx(tx_size);
|
||||
const int cos_bit_col = inv_cos_bit_col[txw_idx][txh_idx];
|
||||
const int cos_bit_row = inv_cos_bit_row[txw_idx][txh_idx];
|
||||
const int txfm_size_col = tx_size_wide[tx_size];
|
||||
const int txfm_size_row = tx_size_high[tx_size];
|
||||
const int buf_size_nonzero_h_div8 = (eoby + 8) >> 3;
|
||||
const int rect_type = get_rect_tx_log_ratio(txfm_size_col, txfm_size_row);
|
||||
const int buf_offset = AOMMAX(txfm_size_row, txfm_size_col);
|
||||
|
||||
int32_t *temp_out = temp_in + buf_offset;
|
||||
int32_t *buf = temp_out + buf_offset;
|
||||
int32_t *buf_ptr = buf;
|
||||
const int8_t stage_range[MAX_TXFM_STAGE_NUM] = { 16 };
|
||||
const int bd = 8;
|
||||
int r;
|
||||
|
||||
const int fun_idx_x = lowbd_txfm_all_1d_zeros_idx[eobx];
|
||||
const int fun_idx_y = lowbd_txfm_all_1d_zeros_idx[eoby];
|
||||
const transform_1d_neon row_txfm =
|
||||
lowbd_txfm_all_1d_zeros_w8_arr[txw_idx][hitx_1d_tab[tx_type]][fun_idx_x];
|
||||
const transform_1d_neon col_txfm =
|
||||
lowbd_txfm_all_1d_zeros_w8_arr[txh_idx][vitx_1d_tab[tx_type]][fun_idx_y];
|
||||
|
||||
assert(col_txfm != NULL);
|
||||
assert(row_txfm != NULL);
|
||||
|
||||
get_flip_cfg(tx_type, &ud_flip, &lr_flip);
|
||||
row_start = (buf_size_nonzero_h_div8 << 3);
|
||||
|
||||
for (int i = 0; i < row_start; i++) {
|
||||
if (abs(rect_type) == 1) {
|
||||
for (int j = 0; j < txfm_size_col; j++)
|
||||
temp_in[j] = round_shift((int64_t)input[j] * NewInvSqrt2, NewSqrt2Bits);
|
||||
row_txfm(temp_in, buf_ptr, cos_bit_row, stage_range);
|
||||
} else {
|
||||
row_txfm(input, buf_ptr, cos_bit_row, stage_range);
|
||||
}
|
||||
av1_round_shift_array(buf_ptr, txfm_size_col, -shift[0]);
|
||||
input += txfm_size_col;
|
||||
buf_ptr += txfm_size_col;
|
||||
}
|
||||
|
||||
// Doing memset for the rows which are not processed in row transform.
|
||||
memset(buf_ptr, 0,
|
||||
sizeof(int32_t) * txfm_size_col * (txfm_size_row - row_start));
|
||||
|
||||
for (int c = 0; c < txfm_size_col; c++) {
|
||||
if (lr_flip == 0) {
|
||||
for (r = 0; r < txfm_size_row; ++r)
|
||||
temp_in[r] = buf[r * txfm_size_col + c];
|
||||
} else {
|
||||
// flip left right
|
||||
for (r = 0; r < txfm_size_row; ++r)
|
||||
temp_in[r] = buf[r * txfm_size_col + (txfm_size_col - c - 1)];
|
||||
}
|
||||
col_txfm(temp_in, temp_out, cos_bit_col, stage_range);
|
||||
av1_round_shift_array(temp_out, txfm_size_row, -shift[1]);
|
||||
|
||||
if (ud_flip == 0) {
|
||||
for (r = 0; r < txfm_size_row; ++r) {
|
||||
output[r * stride + c] =
|
||||
highbd_clip_pixel_add(output[r * stride + c], temp_out[r], bd);
|
||||
}
|
||||
} else {
|
||||
// flip upside down
|
||||
for (r = 0; r < txfm_size_row; ++r) {
|
||||
output[r * stride + c] = highbd_clip_pixel_add(
|
||||
output[r * stride + c], temp_out[txfm_size_row - r - 1], bd);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static INLINE void lowbd_inv_txfm2d_add_universe_neon(
|
||||
const int32_t *input, uint8_t *output, int stride, TX_TYPE tx_type,
|
||||
TX_SIZE tx_size, int eob) {
|
||||
switch (tx_type) {
|
||||
case IDTX:
|
||||
lowbd_inv_txfm2d_add_idtx_neon(input, output, stride, tx_type, tx_size,
|
||||
eob);
|
||||
break;
|
||||
|
||||
case H_DCT:
|
||||
case H_ADST:
|
||||
case H_FLIPADST:
|
||||
lowbd_inv_txfm2d_add_v_identity_neon(input, output, stride, tx_type,
|
||||
tx_size, eob);
|
||||
break;
|
||||
|
||||
case V_DCT:
|
||||
case V_ADST:
|
||||
case V_FLIPADST:
|
||||
lowbd_inv_txfm2d_add_h_identity_neon(input, output, stride, tx_type,
|
||||
tx_size, eob);
|
||||
break;
|
||||
|
||||
default:
|
||||
lowbd_inv_txfm2d_add_no_identity_neon(input, output, stride, tx_type,
|
||||
tx_size, eob);
|
||||
break;
|
||||
}
|
||||
}
|
||||
void av1_lowbd_inv_txfm2d_add_neon(const int32_t *input, uint8_t *output,
|
||||
int stride, TX_TYPE tx_type, TX_SIZE tx_size,
|
||||
int eob) {
|
||||
int row;
|
||||
switch (tx_size) {
|
||||
case TX_4X4:
|
||||
lowbd_inv_txfm2d_add_4x4_neon(input, output, stride, tx_type, tx_size,
|
||||
eob);
|
||||
break;
|
||||
|
||||
case TX_4X8:
|
||||
lowbd_inv_txfm2d_add_4x8_neon(input, output, stride, tx_type, tx_size,
|
||||
eob);
|
||||
break;
|
||||
|
||||
case TX_8X4:
|
||||
lowbd_inv_txfm2d_add_8x4_neon(input, output, stride, tx_type, tx_size,
|
||||
eob);
|
||||
break;
|
||||
|
||||
case TX_4X16:
|
||||
lowbd_inv_txfm2d_add_4x16_neon(input, output, stride, tx_type, tx_size,
|
||||
eob);
|
||||
break;
|
||||
|
||||
case TX_16X4:
|
||||
lowbd_inv_txfm2d_add_16x4_neon(input, output, stride, tx_type, tx_size,
|
||||
eob);
|
||||
break;
|
||||
|
||||
case TX_16X64: {
|
||||
lowbd_inv_txfm2d_add_universe_neon(input, output, stride, tx_type,
|
||||
tx_size, eob);
|
||||
} break;
|
||||
|
||||
case TX_64X16: {
|
||||
int32_t mod_input[64 * 16];
|
||||
for (row = 0; row < 16; ++row) {
|
||||
memcpy(mod_input + row * 64, input + row * 32, 32 * sizeof(*mod_input));
|
||||
memset(mod_input + row * 64 + 32, 0, 32 * sizeof(*mod_input));
|
||||
}
|
||||
lowbd_inv_txfm2d_add_universe_neon(mod_input, output, stride, tx_type,
|
||||
tx_size, eob);
|
||||
} break;
|
||||
|
||||
case TX_32X64: {
|
||||
lowbd_inv_txfm2d_add_universe_neon(input, output, stride, tx_type,
|
||||
tx_size, eob);
|
||||
} break;
|
||||
|
||||
case TX_64X32: {
|
||||
int32_t mod_input[64 * 32];
|
||||
for (row = 0; row < 32; ++row) {
|
||||
memcpy(mod_input + row * 64, input + row * 32, 32 * sizeof(*mod_input));
|
||||
memset(mod_input + row * 64 + 32, 0, 32 * sizeof(*mod_input));
|
||||
}
|
||||
lowbd_inv_txfm2d_add_universe_neon(mod_input, output, stride, tx_type,
|
||||
tx_size, eob);
|
||||
} break;
|
||||
|
||||
case TX_64X64: {
|
||||
int32_t mod_input[64 * 64];
|
||||
for (row = 0; row < 32; ++row) {
|
||||
memcpy(mod_input + row * 64, input + row * 32, 32 * sizeof(*mod_input));
|
||||
memset(mod_input + row * 64 + 32, 0, 32 * sizeof(*mod_input));
|
||||
}
|
||||
lowbd_inv_txfm2d_add_universe_neon(mod_input, output, stride, tx_type,
|
||||
tx_size, eob);
|
||||
} break;
|
||||
|
||||
default:
|
||||
lowbd_inv_txfm2d_add_universe_neon(input, output, stride, tx_type,
|
||||
tx_size, eob);
|
||||
break;
|
||||
}
|
||||
}
|
||||
void av1_inv_txfm_add_neon(const tran_low_t *dqcoeff, uint8_t *dst, int stride,
|
||||
const TxfmParam *txfm_param) {
|
||||
const TX_TYPE tx_type = txfm_param->tx_type;
|
||||
if (!txfm_param->lossless) {
|
||||
av1_lowbd_inv_txfm2d_add_neon(dqcoeff, dst, stride, tx_type,
|
||||
txfm_param->tx_size, txfm_param->eob);
|
||||
} else {
|
||||
av1_inv_txfm_add_c(dqcoeff, dst, stride, txfm_param);
|
||||
}
|
||||
}
|
||||
152
third_party/aom/av1/common/arm/av1_inv_txfm_neon.h
vendored
Normal file
152
third_party/aom/av1/common/arm/av1_inv_txfm_neon.h
vendored
Normal file
|
|
@ -0,0 +1,152 @@
|
|||
/*
|
||||
* Copyright (c) 2018, Alliance for Open Media. All rights reserved
|
||||
*
|
||||
* This source code is subject to the terms of the BSD 2 Clause License and
|
||||
* the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
|
||||
* was not distributed with this source code in the LICENSE file, you can
|
||||
* obtain it at www.aomedia.org/license/software. If the Alliance for Open
|
||||
* Media Patent License 1.0 was not distributed with this source code in the
|
||||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
#ifndef AV1_COMMON_ARM_AV1_INV_TXFM_NEON_H_
|
||||
#define AV1_COMMON_ARM_AV1_INV_TXFM_NEON_H_
|
||||
|
||||
#include "config/aom_config.h"
|
||||
#include "config/av1_rtcd.h"
|
||||
|
||||
#include "aom/aom_integer.h"
|
||||
#include "av1/common/enums.h"
|
||||
#include "av1/common/av1_inv_txfm1d.h"
|
||||
#include "av1/common/av1_inv_txfm1d_cfg.h"
|
||||
#include "av1/common/av1_txfm.h"
|
||||
|
||||
typedef void (*transform_1d_neon)(const int32_t *input, int32_t *output,
|
||||
const int8_t cos_bit,
|
||||
const int8_t *stage_ptr);
|
||||
|
||||
DECLARE_ALIGNED(16, static const int16_t, av1_eob_to_eobxy_8x8_default[8]) = {
|
||||
0x0707, 0x0707, 0x0707, 0x0707, 0x0707, 0x0707, 0x0707, 0x0707,
|
||||
};
|
||||
|
||||
DECLARE_ALIGNED(16, static const int16_t,
|
||||
av1_eob_to_eobxy_16x16_default[16]) = {
|
||||
0x0707, 0x0707, 0x0f0f, 0x0f0f, 0x0f0f, 0x0f0f, 0x0f0f, 0x0f0f,
|
||||
0x0f0f, 0x0f0f, 0x0f0f, 0x0f0f, 0x0f0f, 0x0f0f, 0x0f0f, 0x0f0f,
|
||||
};
|
||||
|
||||
DECLARE_ALIGNED(16, static const int16_t,
|
||||
av1_eob_to_eobxy_32x32_default[32]) = {
|
||||
0x0707, 0x0f0f, 0x0f0f, 0x0f0f, 0x1f1f, 0x1f1f, 0x1f1f, 0x1f1f,
|
||||
0x1f1f, 0x1f1f, 0x1f1f, 0x1f1f, 0x1f1f, 0x1f1f, 0x1f1f, 0x1f1f,
|
||||
0x1f1f, 0x1f1f, 0x1f1f, 0x1f1f, 0x1f1f, 0x1f1f, 0x1f1f, 0x1f1f,
|
||||
0x1f1f, 0x1f1f, 0x1f1f, 0x1f1f, 0x1f1f, 0x1f1f, 0x1f1f, 0x1f1f,
|
||||
};
|
||||
|
||||
DECLARE_ALIGNED(16, static const int16_t, av1_eob_to_eobxy_8x16_default[16]) = {
|
||||
0x0707, 0x0707, 0x0707, 0x0707, 0x0707, 0x0f07, 0x0f07, 0x0f07,
|
||||
0x0f07, 0x0f07, 0x0f07, 0x0f07, 0x0f07, 0x0f07, 0x0f07, 0x0f07,
|
||||
};
|
||||
|
||||
DECLARE_ALIGNED(16, static const int16_t, av1_eob_to_eobxy_16x8_default[8]) = {
|
||||
0x0707, 0x0707, 0x070f, 0x070f, 0x070f, 0x070f, 0x070f, 0x070f,
|
||||
};
|
||||
|
||||
DECLARE_ALIGNED(16, static const int16_t,
|
||||
av1_eob_to_eobxy_16x32_default[32]) = {
|
||||
0x0707, 0x0707, 0x0f0f, 0x0f0f, 0x0f0f, 0x0f0f, 0x0f0f, 0x0f0f,
|
||||
0x0f0f, 0x1f0f, 0x1f0f, 0x1f0f, 0x1f0f, 0x1f0f, 0x1f0f, 0x1f0f,
|
||||
0x1f0f, 0x1f0f, 0x1f0f, 0x1f0f, 0x1f0f, 0x1f0f, 0x1f0f, 0x1f0f,
|
||||
0x1f0f, 0x1f0f, 0x1f0f, 0x1f0f, 0x1f0f, 0x1f0f, 0x1f0f, 0x1f0f,
|
||||
};
|
||||
|
||||
DECLARE_ALIGNED(16, static const int16_t,
|
||||
av1_eob_to_eobxy_32x16_default[16]) = {
|
||||
0x0707, 0x0f0f, 0x0f0f, 0x0f0f, 0x0f1f, 0x0f1f, 0x0f1f, 0x0f1f,
|
||||
0x0f1f, 0x0f1f, 0x0f1f, 0x0f1f, 0x0f1f, 0x0f1f, 0x0f1f, 0x0f1f,
|
||||
};
|
||||
|
||||
DECLARE_ALIGNED(16, static const int16_t, av1_eob_to_eobxy_8x32_default[32]) = {
|
||||
0x0707, 0x0707, 0x0707, 0x0707, 0x0707, 0x0f07, 0x0f07, 0x0f07,
|
||||
0x0f07, 0x0f07, 0x0f07, 0x0f07, 0x0f07, 0x1f07, 0x1f07, 0x1f07,
|
||||
0x1f07, 0x1f07, 0x1f07, 0x1f07, 0x1f07, 0x1f07, 0x1f07, 0x1f07,
|
||||
0x1f07, 0x1f07, 0x1f07, 0x1f07, 0x1f07, 0x1f07, 0x1f07, 0x1f07,
|
||||
};
|
||||
|
||||
DECLARE_ALIGNED(16, static const int16_t, av1_eob_to_eobxy_32x8_default[8]) = {
|
||||
0x0707, 0x070f, 0x070f, 0x071f, 0x071f, 0x071f, 0x071f, 0x071f,
|
||||
};
|
||||
|
||||
DECLARE_ALIGNED(16, static const int16_t *,
|
||||
av1_eob_to_eobxy_default[TX_SIZES_ALL]) = {
|
||||
NULL,
|
||||
av1_eob_to_eobxy_8x8_default,
|
||||
av1_eob_to_eobxy_16x16_default,
|
||||
av1_eob_to_eobxy_32x32_default,
|
||||
av1_eob_to_eobxy_32x32_default,
|
||||
NULL,
|
||||
NULL,
|
||||
av1_eob_to_eobxy_8x16_default,
|
||||
av1_eob_to_eobxy_16x8_default,
|
||||
av1_eob_to_eobxy_16x32_default,
|
||||
av1_eob_to_eobxy_32x16_default,
|
||||
av1_eob_to_eobxy_32x32_default,
|
||||
av1_eob_to_eobxy_32x32_default,
|
||||
NULL,
|
||||
NULL,
|
||||
av1_eob_to_eobxy_8x32_default,
|
||||
av1_eob_to_eobxy_32x8_default,
|
||||
av1_eob_to_eobxy_16x32_default,
|
||||
av1_eob_to_eobxy_32x16_default,
|
||||
};
|
||||
|
||||
static const int lowbd_txfm_all_1d_zeros_idx[32] = {
|
||||
0, 1, 1, 1, 1, 1, 1, 1, 2, 2, 2, 2, 2, 2, 2, 2,
|
||||
3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3,
|
||||
};
|
||||
|
||||
// Transform block width in log2 for eob (size of 64 map to 32)
|
||||
static const int tx_size_wide_log2_eob[TX_SIZES_ALL] = {
|
||||
2, 3, 4, 5, 5, 2, 3, 3, 4, 4, 5, 5, 5, 2, 4, 3, 5, 4, 5,
|
||||
};
|
||||
|
||||
static int eob_fill[32] = {
|
||||
0, 7, 7, 7, 7, 7, 7, 7, 15, 15, 15, 15, 15, 15, 15, 15,
|
||||
31, 31, 31, 31, 31, 31, 31, 31, 31, 31, 31, 31, 31, 31, 31, 31,
|
||||
};
|
||||
|
||||
static INLINE void get_eobx_eoby_scan_default(int *eobx, int *eoby,
|
||||
TX_SIZE tx_size, int eob) {
|
||||
if (eob == 1) {
|
||||
*eobx = 0;
|
||||
*eoby = 0;
|
||||
return;
|
||||
}
|
||||
|
||||
const int tx_w_log2 = tx_size_wide_log2_eob[tx_size];
|
||||
const int eob_row = (eob - 1) >> tx_w_log2;
|
||||
const int eobxy = av1_eob_to_eobxy_default[tx_size][eob_row];
|
||||
*eobx = eobxy & 0xFF;
|
||||
*eoby = eobxy >> 8;
|
||||
}
|
||||
|
||||
static INLINE void get_eobx_eoby_scan_v_identity(int *eobx, int *eoby,
|
||||
TX_SIZE tx_size, int eob) {
|
||||
eob -= 1;
|
||||
const int txfm_size_row = tx_size_high[tx_size];
|
||||
const int eoby_max = AOMMIN(32, txfm_size_row) - 1;
|
||||
*eobx = eob / (eoby_max + 1);
|
||||
*eoby = (eob >= eoby_max) ? eoby_max : eob_fill[eob];
|
||||
}
|
||||
|
||||
static INLINE void get_eobx_eoby_scan_h_identity(int *eobx, int *eoby,
|
||||
TX_SIZE tx_size, int eob) {
|
||||
eob -= 1;
|
||||
const int txfm_size_col = tx_size_wide[tx_size];
|
||||
const int eobx_max = AOMMIN(32, txfm_size_col) - 1;
|
||||
*eobx = (eob >= eobx_max) ? eobx_max : eob_fill[eob];
|
||||
const int temp_eoby = eob / (eobx_max + 1);
|
||||
assert(temp_eoby < 32);
|
||||
*eoby = eob_fill[temp_eoby];
|
||||
}
|
||||
|
||||
#endif // AV1_COMMON_ARM_AV1_INV_TXFM_NEON_H_
|
||||
24
third_party/aom/av1/common/arm/convolve_neon.c
vendored
24
third_party/aom/av1/common/arm/convolve_neon.c
vendored
|
|
@ -164,8 +164,8 @@ static INLINE uint8x8_t convolve8_vert_8x4_s32(
|
|||
|
||||
void av1_convolve_x_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
|
||||
int dst_stride, int w, int h,
|
||||
InterpFilterParams *filter_params_x,
|
||||
InterpFilterParams *filter_params_y,
|
||||
const InterpFilterParams *filter_params_x,
|
||||
const InterpFilterParams *filter_params_y,
|
||||
const int subpel_x_q4, const int subpel_y_q4,
|
||||
ConvolveParams *conv_params) {
|
||||
const uint8_t horiz_offset = filter_params_x->taps / 2 - 1;
|
||||
|
|
@ -182,7 +182,7 @@ void av1_convolve_x_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
|
|||
((conv_params->round_0 + conv_params->round_1) == 2 * FILTER_BITS));
|
||||
|
||||
const int16_t *x_filter = av1_get_interp_filter_subpel_kernel(
|
||||
*filter_params_x, subpel_x_q4 & SUBPEL_MASK);
|
||||
filter_params_x, subpel_x_q4 & SUBPEL_MASK);
|
||||
|
||||
const int16x8_t shift_round_0 = vdupq_n_s16(-conv_params->round_0);
|
||||
const int16x8_t shift_by_bits = vdupq_n_s16(-bits);
|
||||
|
|
@ -485,8 +485,8 @@ void av1_convolve_x_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
|
|||
|
||||
void av1_convolve_y_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
|
||||
int dst_stride, int w, int h,
|
||||
InterpFilterParams *filter_params_x,
|
||||
InterpFilterParams *filter_params_y,
|
||||
const InterpFilterParams *filter_params_x,
|
||||
const InterpFilterParams *filter_params_y,
|
||||
const int subpel_x_q4, const int subpel_y_q4,
|
||||
ConvolveParams *conv_params) {
|
||||
const int vert_offset = filter_params_y->taps / 2 - 1;
|
||||
|
|
@ -502,7 +502,7 @@ void av1_convolve_y_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
|
|||
((conv_params->round_0 + conv_params->round_1) == (2 * FILTER_BITS)));
|
||||
|
||||
const int16_t *y_filter = av1_get_interp_filter_subpel_kernel(
|
||||
*filter_params_y, subpel_y_q4 & SUBPEL_MASK);
|
||||
filter_params_y, subpel_y_q4 & SUBPEL_MASK);
|
||||
|
||||
if (w <= 4) {
|
||||
uint8x8_t d01, d23;
|
||||
|
|
@ -680,8 +680,8 @@ void av1_convolve_y_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
|
|||
|
||||
void av1_convolve_2d_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
|
||||
int dst_stride, int w, int h,
|
||||
InterpFilterParams *filter_params_x,
|
||||
InterpFilterParams *filter_params_y,
|
||||
const InterpFilterParams *filter_params_x,
|
||||
const InterpFilterParams *filter_params_y,
|
||||
const int subpel_x_q4, const int subpel_y_q4,
|
||||
ConvolveParams *conv_params) {
|
||||
int im_dst_stride;
|
||||
|
|
@ -711,7 +711,7 @@ void av1_convolve_2d_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
|
|||
const int16x8_t vec_round_bits = vdupq_n_s16(-round_bits);
|
||||
const int offset_bits = bd + 2 * FILTER_BITS - conv_params->round_0;
|
||||
const int16_t *x_filter = av1_get_interp_filter_subpel_kernel(
|
||||
*filter_params_x, subpel_x_q4 & SUBPEL_MASK);
|
||||
filter_params_x, subpel_x_q4 & SUBPEL_MASK);
|
||||
|
||||
int16_t x_filter_tmp[8];
|
||||
int16x8_t filter_x_coef = vld1q_s16(x_filter);
|
||||
|
|
@ -896,7 +896,7 @@ void av1_convolve_2d_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
|
|||
const int32_t sub_const = (1 << (offset_bits - conv_params->round_1)) +
|
||||
(1 << (offset_bits - conv_params->round_1 - 1));
|
||||
const int16_t *y_filter = av1_get_interp_filter_subpel_kernel(
|
||||
*filter_params_y, subpel_y_q4 & SUBPEL_MASK);
|
||||
filter_params_y, subpel_y_q4 & SUBPEL_MASK);
|
||||
|
||||
const int32x4_t round_shift_vec = vdupq_n_s32(-(conv_params->round_1));
|
||||
const int32x4_t offset_const = vdupq_n_s32(1 << offset_bits);
|
||||
|
|
@ -1086,8 +1086,8 @@ void av1_convolve_2d_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
|
|||
}
|
||||
void av1_convolve_2d_copy_sr_neon(const uint8_t *src, int src_stride,
|
||||
uint8_t *dst, int dst_stride, int w, int h,
|
||||
InterpFilterParams *filter_params_x,
|
||||
InterpFilterParams *filter_params_y,
|
||||
const InterpFilterParams *filter_params_x,
|
||||
const InterpFilterParams *filter_params_y,
|
||||
const int subpel_x_q4, const int subpel_y_q4,
|
||||
ConvolveParams *conv_params) {
|
||||
(void)filter_params_x;
|
||||
|
|
|
|||
79
third_party/aom/av1/common/arm/intrapred_neon.c
vendored
79
third_party/aom/av1/common/arm/intrapred_neon.c
vendored
|
|
@ -1,79 +0,0 @@
|
|||
/*
|
||||
*
|
||||
* Copyright (c) 2018, Alliance for Open Media. All rights reserved
|
||||
*
|
||||
* This source code is subject to the terms of the BSD 2 Clause License and
|
||||
* the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
|
||||
* was not distributed with this source code in the LICENSE file, you can
|
||||
* obtain it at www.aomedia.org/license/software. If the Alliance for Open
|
||||
* Media Patent License 1.0 was not distributed with this source code in the
|
||||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
#include <arm_neon.h>
|
||||
#include <assert.h>
|
||||
|
||||
#include "aom_mem/aom_mem.h"
|
||||
#include "aom_ports/mem.h"
|
||||
#include "av1/common/arm/mem_neon.h"
|
||||
#include "config/aom_dsp_rtcd.h"
|
||||
|
||||
static INLINE void highbd_dc_predictor_neon(uint16_t *dst, ptrdiff_t stride,
|
||||
int bw, const uint16_t *above,
|
||||
const uint16_t *left) {
|
||||
assert(bw >= 4);
|
||||
assert(IS_POWER_OF_TWO(bw));
|
||||
int expected_dc, sum = 0;
|
||||
const int count = bw * 2;
|
||||
uint32x4_t sum_q = vdupq_n_u32(0);
|
||||
uint32x2_t sum_d;
|
||||
uint16_t *dst_1;
|
||||
if (bw >= 8) {
|
||||
for (int i = 0; i < bw; i += 8) {
|
||||
sum_q = vpadalq_u16(sum_q, vld1q_u16(above));
|
||||
sum_q = vpadalq_u16(sum_q, vld1q_u16(left));
|
||||
above += 8;
|
||||
left += 8;
|
||||
}
|
||||
sum_d = vadd_u32(vget_low_u32(sum_q), vget_high_u32(sum_q));
|
||||
sum = vget_lane_s32(vreinterpret_s32_u64(vpaddl_u32(sum_d)), 0);
|
||||
expected_dc = (sum + (count >> 1)) / count;
|
||||
const uint16x8_t dc = vdupq_n_u16((uint16_t)expected_dc);
|
||||
for (int r = 0; r < bw; r++) {
|
||||
dst_1 = dst;
|
||||
for (int i = 0; i < bw; i += 8) {
|
||||
vst1q_u16(dst_1, dc);
|
||||
dst_1 += 8;
|
||||
}
|
||||
dst += stride;
|
||||
}
|
||||
} else { // 4x4
|
||||
sum_q = vaddl_u16(vld1_u16(above), vld1_u16(left));
|
||||
sum_d = vadd_u32(vget_low_u32(sum_q), vget_high_u32(sum_q));
|
||||
sum = vget_lane_s32(vreinterpret_s32_u64(vpaddl_u32(sum_d)), 0);
|
||||
expected_dc = (sum + (count >> 1)) / count;
|
||||
const uint16x4_t dc = vdup_n_u16((uint16_t)expected_dc);
|
||||
for (int r = 0; r < bw; r++) {
|
||||
vst1_u16(dst, dc);
|
||||
dst += stride;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#define intra_pred_highbd_sized(type, width) \
|
||||
void aom_highbd_##type##_predictor_##width##x##width##_neon( \
|
||||
uint16_t *dst, ptrdiff_t stride, const uint16_t *above, \
|
||||
const uint16_t *left, int bd) { \
|
||||
(void)bd; \
|
||||
highbd_##type##_predictor_neon(dst, stride, width, above, left); \
|
||||
}
|
||||
|
||||
#define intra_pred_square(type) \
|
||||
intra_pred_highbd_sized(type, 4); \
|
||||
intra_pred_highbd_sized(type, 8); \
|
||||
intra_pred_highbd_sized(type, 16); \
|
||||
intra_pred_highbd_sized(type, 32); \
|
||||
intra_pred_highbd_sized(type, 64);
|
||||
|
||||
intra_pred_square(dc);
|
||||
|
||||
#undef intra_pred_square
|
||||
|
|
@ -515,8 +515,8 @@ static INLINE void jnt_convolve_2d_vert_neon(
|
|||
|
||||
void av1_jnt_convolve_2d_neon(const uint8_t *src, int src_stride, uint8_t *dst8,
|
||||
int dst8_stride, int w, int h,
|
||||
InterpFilterParams *filter_params_x,
|
||||
InterpFilterParams *filter_params_y,
|
||||
const InterpFilterParams *filter_params_x,
|
||||
const InterpFilterParams *filter_params_y,
|
||||
const int subpel_x_q4, const int subpel_y_q4,
|
||||
ConvolveParams *conv_params) {
|
||||
assert(!(w % 4));
|
||||
|
|
@ -532,9 +532,9 @@ void av1_jnt_convolve_2d_neon(const uint8_t *src, int src_stride, uint8_t *dst8,
|
|||
const int round_0 = conv_params->round_0 - 1;
|
||||
const uint8_t *src_ptr = src - vert_offset * src_stride - horiz_offset;
|
||||
const int16_t *x_filter = av1_get_interp_filter_subpel_kernel(
|
||||
*filter_params_x, subpel_x_q4 & SUBPEL_MASK);
|
||||
filter_params_x, subpel_x_q4 & SUBPEL_MASK);
|
||||
const int16_t *y_filter = av1_get_interp_filter_subpel_kernel(
|
||||
*filter_params_y, subpel_y_q4 & SUBPEL_MASK);
|
||||
filter_params_y, subpel_y_q4 & SUBPEL_MASK);
|
||||
|
||||
int16_t x_filter_tmp[8];
|
||||
int16x8_t filter_x_coef = vld1q_s16(x_filter);
|
||||
|
|
@ -553,8 +553,8 @@ void av1_jnt_convolve_2d_neon(const uint8_t *src, int src_stride, uint8_t *dst8,
|
|||
|
||||
void av1_jnt_convolve_2d_copy_neon(const uint8_t *src, int src_stride,
|
||||
uint8_t *dst8, int dst8_stride, int w, int h,
|
||||
InterpFilterParams *filter_params_x,
|
||||
InterpFilterParams *filter_params_y,
|
||||
const InterpFilterParams *filter_params_x,
|
||||
const InterpFilterParams *filter_params_y,
|
||||
const int subpel_x_q4, const int subpel_y_q4,
|
||||
ConvolveParams *conv_params) {
|
||||
uint8x8_t res0_8, res1_8, res2_8, res3_8, tmp_shift0, tmp_shift1, tmp_shift2,
|
||||
|
|
@ -679,8 +679,8 @@ void av1_jnt_convolve_2d_copy_neon(const uint8_t *src, int src_stride,
|
|||
|
||||
void av1_jnt_convolve_x_neon(const uint8_t *src, int src_stride, uint8_t *dst8,
|
||||
int dst8_stride, int w, int h,
|
||||
InterpFilterParams *filter_params_x,
|
||||
InterpFilterParams *filter_params_y,
|
||||
const InterpFilterParams *filter_params_x,
|
||||
const InterpFilterParams *filter_params_y,
|
||||
const int subpel_x_q4, const int subpel_y_q4,
|
||||
ConvolveParams *conv_params) {
|
||||
assert(!(w % 4));
|
||||
|
|
@ -705,7 +705,7 @@ void av1_jnt_convolve_x_neon(const uint8_t *src, int src_stride, uint8_t *dst8,
|
|||
|
||||
// horizontal filter
|
||||
const int16_t *x_filter = av1_get_interp_filter_subpel_kernel(
|
||||
*filter_params_x, subpel_x_q4 & SUBPEL_MASK);
|
||||
filter_params_x, subpel_x_q4 & SUBPEL_MASK);
|
||||
|
||||
const uint8_t *src_ptr = src - horiz_offset;
|
||||
|
||||
|
|
@ -1013,8 +1013,8 @@ void av1_jnt_convolve_x_neon(const uint8_t *src, int src_stride, uint8_t *dst8,
|
|||
|
||||
void av1_jnt_convolve_y_neon(const uint8_t *src, int src_stride, uint8_t *dst8,
|
||||
int dst8_stride, int w, int h,
|
||||
InterpFilterParams *filter_params_x,
|
||||
InterpFilterParams *filter_params_y,
|
||||
const InterpFilterParams *filter_params_x,
|
||||
const InterpFilterParams *filter_params_y,
|
||||
const int subpel_x_q4, const int subpel_y_q4,
|
||||
ConvolveParams *conv_params) {
|
||||
assert(!(w % 4));
|
||||
|
|
@ -1040,7 +1040,7 @@ void av1_jnt_convolve_y_neon(const uint8_t *src, int src_stride, uint8_t *dst8,
|
|||
|
||||
// vertical filter
|
||||
const int16_t *y_filter = av1_get_interp_filter_subpel_kernel(
|
||||
*filter_params_y, subpel_y_q4 & SUBPEL_MASK);
|
||||
filter_params_y, subpel_y_q4 & SUBPEL_MASK);
|
||||
|
||||
const uint8_t *src_ptr = src - (vert_offset * src_stride);
|
||||
|
||||
|
|
|
|||
84
third_party/aom/av1/common/arm/mem_neon.h
vendored
84
third_party/aom/av1/common/arm/mem_neon.h
vendored
|
|
@ -22,6 +22,14 @@ static INLINE void store_row2_u8_8x8(uint8_t *s, int p, const uint8x8_t s0,
|
|||
s += p;
|
||||
}
|
||||
|
||||
/* These intrinsics require immediate values, so we must use #defines
|
||||
to enforce that. */
|
||||
#define load_u8_4x1(s, s0, lane) \
|
||||
do { \
|
||||
*(s0) = vreinterpret_u8_u32( \
|
||||
vld1_lane_u32((uint32_t *)(s), vreinterpret_u32_u8(*(s0)), lane)); \
|
||||
} while (0)
|
||||
|
||||
static INLINE void load_u8_8x8(const uint8_t *s, ptrdiff_t p,
|
||||
uint8x8_t *const s0, uint8x8_t *const s1,
|
||||
uint8x8_t *const s2, uint8x8_t *const s3,
|
||||
|
|
@ -128,6 +136,13 @@ static INLINE void load_s16_4x4(const int16_t *s, ptrdiff_t p,
|
|||
*s3 = vld1_s16(s);
|
||||
}
|
||||
|
||||
/* These intrinsics require immediate values, so we must use #defines
|
||||
to enforce that. */
|
||||
#define store_u8_4x1(s, s0, lane) \
|
||||
do { \
|
||||
vst1_lane_u32((uint32_t *)(s), vreinterpret_u32_u8(s0), lane); \
|
||||
} while (0)
|
||||
|
||||
static INLINE void store_u8_8x8(uint8_t *s, ptrdiff_t p, const uint8x8_t s0,
|
||||
const uint8x8_t s1, const uint8x8_t s2,
|
||||
const uint8x8_t s3, const uint8x8_t s4,
|
||||
|
|
@ -242,6 +257,30 @@ static INLINE void store_s16_8x8(int16_t *s, ptrdiff_t dst_stride,
|
|||
vst1q_s16(s, s7);
|
||||
}
|
||||
|
||||
static INLINE void store_s16_4x4(int16_t *s, ptrdiff_t dst_stride,
|
||||
const int16x4_t s0, const int16x4_t s1,
|
||||
const int16x4_t s2, const int16x4_t s3) {
|
||||
vst1_s16(s, s0);
|
||||
s += dst_stride;
|
||||
vst1_s16(s, s1);
|
||||
s += dst_stride;
|
||||
vst1_s16(s, s2);
|
||||
s += dst_stride;
|
||||
vst1_s16(s, s3);
|
||||
}
|
||||
|
||||
static INLINE void store_s16_8x4(int16_t *s, ptrdiff_t dst_stride,
|
||||
const int16x8_t s0, const int16x8_t s1,
|
||||
const int16x8_t s2, const int16x8_t s3) {
|
||||
vst1q_s16(s, s0);
|
||||
s += dst_stride;
|
||||
vst1q_s16(s, s1);
|
||||
s += dst_stride;
|
||||
vst1q_s16(s, s2);
|
||||
s += dst_stride;
|
||||
vst1q_s16(s, s3);
|
||||
}
|
||||
|
||||
static INLINE void load_s16_8x8(const int16_t *s, ptrdiff_t p,
|
||||
int16x8_t *const s0, int16x8_t *const s1,
|
||||
int16x8_t *const s2, int16x8_t *const s3,
|
||||
|
|
@ -398,4 +437,49 @@ static INLINE void load_unaligned_u16_4x4(const uint16_t *buf, uint32_t stride,
|
|||
*tu1 = vsetq_lane_u64(a, *tu1, 1);
|
||||
}
|
||||
|
||||
static INLINE void load_s32_4x4(int32_t *s, int32_t p, int32x4_t *s1,
|
||||
int32x4_t *s2, int32x4_t *s3, int32x4_t *s4) {
|
||||
*s1 = vld1q_s32(s);
|
||||
s += p;
|
||||
*s2 = vld1q_s32(s);
|
||||
s += p;
|
||||
*s3 = vld1q_s32(s);
|
||||
s += p;
|
||||
*s4 = vld1q_s32(s);
|
||||
}
|
||||
|
||||
static INLINE void store_s32_4x4(int32_t *s, int32_t p, int32x4_t s1,
|
||||
int32x4_t s2, int32x4_t s3, int32x4_t s4) {
|
||||
vst1q_s32(s, s1);
|
||||
s += p;
|
||||
vst1q_s32(s, s2);
|
||||
s += p;
|
||||
vst1q_s32(s, s3);
|
||||
s += p;
|
||||
vst1q_s32(s, s4);
|
||||
}
|
||||
|
||||
static INLINE void load_u32_4x4(uint32_t *s, int32_t p, uint32x4_t *s1,
|
||||
uint32x4_t *s2, uint32x4_t *s3,
|
||||
uint32x4_t *s4) {
|
||||
*s1 = vld1q_u32(s);
|
||||
s += p;
|
||||
*s2 = vld1q_u32(s);
|
||||
s += p;
|
||||
*s3 = vld1q_u32(s);
|
||||
s += p;
|
||||
*s4 = vld1q_u32(s);
|
||||
}
|
||||
|
||||
static INLINE void store_u32_4x4(uint32_t *s, int32_t p, uint32x4_t s1,
|
||||
uint32x4_t s2, uint32x4_t s3, uint32x4_t s4) {
|
||||
vst1q_u32(s, s1);
|
||||
s += p;
|
||||
vst1q_u32(s, s2);
|
||||
s += p;
|
||||
vst1q_u32(s, s3);
|
||||
s += p;
|
||||
vst1q_u32(s, s4);
|
||||
}
|
||||
|
||||
#endif // AV1_COMMON_ARM_MEM_NEON_H_
|
||||
|
|
|
|||
1506
third_party/aom/av1/common/arm/selfguided_neon.c
vendored
Normal file
1506
third_party/aom/av1/common/arm/selfguided_neon.c
vendored
Normal file
File diff suppressed because it is too large
Load diff
38
third_party/aom/av1/common/arm/transpose_neon.h
vendored
38
third_party/aom/av1/common/arm/transpose_neon.h
vendored
|
|
@ -419,4 +419,42 @@ static INLINE void transpose_s16_4x4d(int16x4_t *a0, int16x4_t *a1,
|
|||
*a3 = vreinterpret_s16_s32(c1.val[1]);
|
||||
}
|
||||
|
||||
static INLINE int32x4x2_t aom_vtrnq_s64_to_s32(int32x4_t a0, int32x4_t a1) {
|
||||
int32x4x2_t b0;
|
||||
b0.val[0] = vcombine_s32(vget_low_s32(a0), vget_low_s32(a1));
|
||||
b0.val[1] = vcombine_s32(vget_high_s32(a0), vget_high_s32(a1));
|
||||
return b0;
|
||||
}
|
||||
|
||||
static INLINE void transpose_s32_4x4(int32x4_t *a0, int32x4_t *a1,
|
||||
int32x4_t *a2, int32x4_t *a3) {
|
||||
// Swap 32 bit elements. Goes from:
|
||||
// a0: 00 01 02 03
|
||||
// a1: 10 11 12 13
|
||||
// a2: 20 21 22 23
|
||||
// a3: 30 31 32 33
|
||||
// to:
|
||||
// b0.val[0]: 00 10 02 12
|
||||
// b0.val[1]: 01 11 03 13
|
||||
// b1.val[0]: 20 30 22 32
|
||||
// b1.val[1]: 21 31 23 33
|
||||
|
||||
const int32x4x2_t b0 = vtrnq_s32(*a0, *a1);
|
||||
const int32x4x2_t b1 = vtrnq_s32(*a2, *a3);
|
||||
|
||||
// Swap 64 bit elements resulting in:
|
||||
// c0.val[0]: 00 10 20 30
|
||||
// c0.val[1]: 02 12 22 32
|
||||
// c1.val[0]: 01 11 21 31
|
||||
// c1.val[1]: 03 13 23 33
|
||||
|
||||
const int32x4x2_t c0 = aom_vtrnq_s64_to_s32(b0.val[0], b1.val[0]);
|
||||
const int32x4x2_t c1 = aom_vtrnq_s64_to_s32(b0.val[1], b1.val[1]);
|
||||
|
||||
*a0 = c0.val[0];
|
||||
*a1 = c1.val[0];
|
||||
*a2 = c0.val[1];
|
||||
*a3 = c1.val[1];
|
||||
}
|
||||
|
||||
#endif // AV1_COMMON_ARM_TRANSPOSE_NEON_H_
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue