mirror of
https://repo.dactyloidae.xyz/Dactyloidae/UXP.git
synced 2026-10-09 08:47:31 +09:00
Update libaom to commit ID 1e227d41f0616de9548a673a83a21ef990b62591
This commit is contained in:
parent
77c9089b7d
commit
368651059a
526 changed files with 34535 additions and 15900 deletions
14
third_party/aom/av1/av1.cmake
vendored
14
third_party/aom/av1/av1.cmake
vendored
|
|
@ -53,6 +53,8 @@ list(APPEND AOM_AV1_COMMON_SOURCES
|
|||
"${AOM_ROOT}/av1/common/mv.h"
|
||||
"${AOM_ROOT}/av1/common/mvref_common.c"
|
||||
"${AOM_ROOT}/av1/common/mvref_common.h"
|
||||
"${AOM_ROOT}/av1/common/obu_util.c"
|
||||
"${AOM_ROOT}/av1/common/obu_util.h"
|
||||
"${AOM_ROOT}/av1/common/odintrin.c"
|
||||
"${AOM_ROOT}/av1/common/odintrin.h"
|
||||
"${AOM_ROOT}/av1/common/onyxc_int.h"
|
||||
|
|
@ -78,8 +80,8 @@ list(APPEND AOM_AV1_COMMON_SOURCES
|
|||
"${AOM_ROOT}/av1/common/thread_common.h"
|
||||
"${AOM_ROOT}/av1/common/tile_common.c"
|
||||
"${AOM_ROOT}/av1/common/tile_common.h"
|
||||
"${AOM_ROOT}/av1/common/timing.h"
|
||||
"${AOM_ROOT}/av1/common/timing.c"
|
||||
"${AOM_ROOT}/av1/common/timing.h"
|
||||
"${AOM_ROOT}/av1/common/token_cdfs.h"
|
||||
"${AOM_ROOT}/av1/common/txb_common.c"
|
||||
"${AOM_ROOT}/av1/common/txb_common.h"
|
||||
|
|
@ -176,6 +178,8 @@ list(APPEND AOM_AV1_ENCODER_SOURCES
|
|||
"${AOM_ROOT}/av1/encoder/rd.h"
|
||||
"${AOM_ROOT}/av1/encoder/rdopt.c"
|
||||
"${AOM_ROOT}/av1/encoder/rdopt.h"
|
||||
"${AOM_ROOT}/av1/encoder/reconinter_enc.c"
|
||||
"${AOM_ROOT}/av1/encoder/reconinter_enc.h"
|
||||
"${AOM_ROOT}/av1/encoder/segmentation.c"
|
||||
"${AOM_ROOT}/av1/encoder/segmentation.h"
|
||||
"${AOM_ROOT}/av1/encoder/speed_features.c"
|
||||
|
|
@ -268,7 +272,8 @@ list(APPEND AOM_AV1_ENCODER_INTRIN_SSE4_1
|
|||
"${AOM_ROOT}/av1/encoder/x86/av1_highbd_quantize_sse4.c"
|
||||
"${AOM_ROOT}/av1/encoder/x86/corner_match_sse4.c"
|
||||
"${AOM_ROOT}/av1/encoder/x86/encodetxb_sse4.c"
|
||||
"${AOM_ROOT}/av1/encoder/x86/highbd_fwd_txfm_sse4.c")
|
||||
"${AOM_ROOT}/av1/encoder/x86/highbd_fwd_txfm_sse4.c"
|
||||
"${AOM_ROOT}/av1/encoder/x86/pickrst_sse4.c")
|
||||
|
||||
list(APPEND AOM_AV1_ENCODER_INTRIN_AVX2
|
||||
"${AOM_ROOT}/av1/encoder/x86/av1_quantize_avx2.c"
|
||||
|
|
@ -276,7 +281,9 @@ list(APPEND AOM_AV1_ENCODER_INTRIN_AVX2
|
|||
"${AOM_ROOT}/av1/encoder/x86/error_intrin_avx2.c"
|
||||
"${AOM_ROOT}/av1/encoder/x86/av1_fwd_txfm_avx2.h"
|
||||
"${AOM_ROOT}/av1/encoder/x86/av1_fwd_txfm2d_avx2.c"
|
||||
"${AOM_ROOT}/av1/encoder/x86/wedge_utils_avx2.c")
|
||||
"${AOM_ROOT}/av1/encoder/x86/wedge_utils_avx2.c"
|
||||
"${AOM_ROOT}/av1/encoder/x86/encodetxb_avx2.c"
|
||||
"${AOM_ROOT}/av1/encoder/x86/pickrst_avx2.c")
|
||||
|
||||
list(APPEND AOM_AV1_ENCODER_INTRIN_NEON
|
||||
"${AOM_ROOT}/av1/encoder/arm/neon/quantize_neon.c")
|
||||
|
|
@ -301,6 +308,7 @@ list(APPEND AOM_AV1_COMMON_INTRIN_NEON
|
|||
"${AOM_ROOT}/av1/common/arm/selfguided_neon.c"
|
||||
"${AOM_ROOT}/av1/common/arm/av1_inv_txfm_neon.c"
|
||||
"${AOM_ROOT}/av1/common/arm/av1_inv_txfm_neon.h"
|
||||
"${AOM_ROOT}/av1/common/arm/warp_plane_neon.c"
|
||||
"${AOM_ROOT}/av1/common/cdef_block_neon.c")
|
||||
|
||||
list(APPEND AOM_AV1_ENCODER_INTRIN_SSE4_2
|
||||
|
|
|
|||
160
third_party/aom/av1/av1_cx_iface.c
vendored
160
third_party/aom/av1/av1_cx_iface.c
vendored
|
|
@ -14,28 +14,29 @@
|
|||
#include "config/aom_config.h"
|
||||
#include "config/aom_version.h"
|
||||
|
||||
#include "aom/aom_encoder.h"
|
||||
#include "aom_ports/aom_once.h"
|
||||
#include "aom_ports/mem_ops.h"
|
||||
#include "aom_ports/system_state.h"
|
||||
|
||||
#include "aom/aom_encoder.h"
|
||||
#include "aom/internal/aom_codec_internal.h"
|
||||
#include "av1/encoder/encoder.h"
|
||||
#include "aom/aomcx.h"
|
||||
#include "av1/encoder/firstpass.h"
|
||||
|
||||
#include "av1/av1_iface_common.h"
|
||||
#include "av1/encoder/bitstream.h"
|
||||
#include "aom_ports/mem_ops.h"
|
||||
#include "av1/encoder/encoder.h"
|
||||
#include "av1/encoder/firstpass.h"
|
||||
|
||||
#define MAG_SIZE (4)
|
||||
#define MAX_NUM_ENHANCEMENT_LAYERS 3
|
||||
|
||||
struct av1_extracfg {
|
||||
int cpu_used; // available cpu percentage in 1/16
|
||||
int dev_sf;
|
||||
unsigned int enable_auto_alt_ref;
|
||||
unsigned int enable_auto_bwd_ref;
|
||||
unsigned int noise_sensitivity;
|
||||
unsigned int sharpness;
|
||||
unsigned int static_thresh;
|
||||
unsigned int row_mt;
|
||||
unsigned int tile_columns; // log2 number of tile columns
|
||||
unsigned int tile_rows; // log2 number of tile rows
|
||||
unsigned int arnr_max_frames;
|
||||
|
|
@ -98,37 +99,40 @@ struct av1_extracfg {
|
|||
float noise_level;
|
||||
int noise_block_size;
|
||||
#endif
|
||||
|
||||
unsigned int chroma_subsampling_x;
|
||||
unsigned int chroma_subsampling_y;
|
||||
};
|
||||
|
||||
static struct av1_extracfg default_extra_cfg = {
|
||||
0, // cpu_used
|
||||
0, // dev_sf
|
||||
1, // enable_auto_alt_ref
|
||||
0, // enable_auto_bwd_ref
|
||||
0, // noise_sensitivity
|
||||
0, // sharpness
|
||||
0, // static_thresh
|
||||
0, // tile_columns
|
||||
0, // tile_rows
|
||||
7, // arnr_max_frames
|
||||
5, // arnr_strength
|
||||
0, // min_gf_interval; 0 -> default decision
|
||||
0, // max_gf_interval; 0 -> default decision
|
||||
AOM_TUNE_PSNR, // tuning
|
||||
10, // cq_level
|
||||
0, // rc_max_intra_bitrate_pct
|
||||
0, // rc_max_inter_bitrate_pct
|
||||
0, // gf_cbr_boost_pct
|
||||
0, // lossless
|
||||
1, // enable_cdef
|
||||
1, // enable_restoration
|
||||
0, // disable_trellis_quant
|
||||
0, // enable_qm
|
||||
DEFAULT_QM_Y, // qm_y
|
||||
DEFAULT_QM_U, // qm_u
|
||||
DEFAULT_QM_V, // qm_v
|
||||
DEFAULT_QM_FIRST, // qm_min
|
||||
DEFAULT_QM_LAST, // qm_max
|
||||
0, // cpu_used
|
||||
1, // enable_auto_alt_ref
|
||||
0, // enable_auto_bwd_ref
|
||||
0, // noise_sensitivity
|
||||
CONFIG_SHARP_SETTINGS, // sharpness
|
||||
0, // static_thresh
|
||||
0, // row_mt
|
||||
0, // tile_columns
|
||||
0, // tile_rows
|
||||
7, // arnr_max_frames
|
||||
5, // arnr_strength
|
||||
0, // min_gf_interval; 0 -> default decision
|
||||
0, // max_gf_interval; 0 -> default decision
|
||||
AOM_TUNE_PSNR, // tuning
|
||||
10, // cq_level
|
||||
0, // rc_max_intra_bitrate_pct
|
||||
0, // rc_max_inter_bitrate_pct
|
||||
0, // gf_cbr_boost_pct
|
||||
0, // lossless
|
||||
!CONFIG_SHARP_SETTINGS, // enable_cdef
|
||||
1, // enable_restoration
|
||||
0, // disable_trellis_quant
|
||||
0, // enable_qm
|
||||
DEFAULT_QM_Y, // qm_y
|
||||
DEFAULT_QM_U, // qm_u
|
||||
DEFAULT_QM_V, // qm_v
|
||||
DEFAULT_QM_FIRST, // qm_min
|
||||
DEFAULT_QM_LAST, // qm_max
|
||||
#if CONFIG_DIST_8X8
|
||||
0,
|
||||
#endif
|
||||
|
|
@ -150,7 +154,7 @@ static struct av1_extracfg default_extra_cfg = {
|
|||
0, // render width
|
||||
0, // render height
|
||||
AOM_SUPERBLOCK_SIZE_DYNAMIC, // superblock_size
|
||||
0, // Single tile decoding is off by default.
|
||||
1, // this depends on large_scale_tile.
|
||||
0, // error_resilient_mode off by default.
|
||||
0, // s_frame_mode off by default.
|
||||
0, // film_grain_test_vector
|
||||
|
|
@ -168,6 +172,8 @@ static struct av1_extracfg default_extra_cfg = {
|
|||
0, // noise_level
|
||||
32, // noise_block_size
|
||||
#endif
|
||||
0, // chroma_subsampling_x
|
||||
0, // chroma_subsampling_y
|
||||
};
|
||||
|
||||
struct aom_codec_alg_priv {
|
||||
|
|
@ -251,10 +257,7 @@ static aom_codec_err_t validate_config(aom_codec_alg_priv_t *ctx,
|
|||
RANGE_CHECK_HI(extra_cfg, min_gf_interval, MAX_LAG_BUFFERS - 1);
|
||||
RANGE_CHECK_HI(extra_cfg, max_gf_interval, MAX_LAG_BUFFERS - 1);
|
||||
if (extra_cfg->max_gf_interval > 0) {
|
||||
RANGE_CHECK(extra_cfg, max_gf_interval, 2, (MAX_LAG_BUFFERS - 1));
|
||||
}
|
||||
if (extra_cfg->min_gf_interval > 0 && extra_cfg->max_gf_interval > 0) {
|
||||
RANGE_CHECK(extra_cfg, max_gf_interval, extra_cfg->min_gf_interval,
|
||||
RANGE_CHECK(extra_cfg, max_gf_interval, MAX(2, extra_cfg->min_gf_interval),
|
||||
(MAX_LAG_BUFFERS - 1));
|
||||
}
|
||||
|
||||
|
|
@ -284,13 +287,14 @@ static aom_codec_err_t validate_config(aom_codec_alg_priv_t *ctx,
|
|||
RANGE_CHECK_HI(extra_cfg, enable_auto_alt_ref, 2);
|
||||
RANGE_CHECK_HI(extra_cfg, enable_auto_bwd_ref, 2);
|
||||
RANGE_CHECK(extra_cfg, cpu_used, 0, 8);
|
||||
RANGE_CHECK(extra_cfg, dev_sf, 0, UINT8_MAX);
|
||||
RANGE_CHECK_HI(extra_cfg, noise_sensitivity, 6);
|
||||
RANGE_CHECK(extra_cfg, superblock_size, AOM_SUPERBLOCK_SIZE_64X64,
|
||||
AOM_SUPERBLOCK_SIZE_DYNAMIC);
|
||||
RANGE_CHECK_HI(cfg, large_scale_tile, 1);
|
||||
RANGE_CHECK_HI(extra_cfg, single_tile_decoding, 1);
|
||||
|
||||
RANGE_CHECK_HI(extra_cfg, row_mt, 1);
|
||||
|
||||
RANGE_CHECK_HI(extra_cfg, tile_columns, 6);
|
||||
RANGE_CHECK_HI(extra_cfg, tile_rows, 6);
|
||||
|
||||
|
|
@ -372,6 +376,9 @@ static aom_codec_err_t validate_config(aom_codec_alg_priv_t *ctx,
|
|||
#endif
|
||||
}
|
||||
|
||||
RANGE_CHECK_HI(extra_cfg, chroma_subsampling_x, 1);
|
||||
RANGE_CHECK_HI(extra_cfg, chroma_subsampling_y, 1);
|
||||
|
||||
return AOM_CODEC_OK;
|
||||
}
|
||||
|
||||
|
|
@ -581,7 +588,6 @@ static aom_codec_err_t set_encoder_config(
|
|||
oxcf->sframe_mode = cfg->sframe_mode;
|
||||
oxcf->sframe_enabled = cfg->sframe_dist != 0;
|
||||
oxcf->speed = extra_cfg->cpu_used;
|
||||
oxcf->dev_sf = extra_cfg->dev_sf;
|
||||
oxcf->enable_auto_arf = extra_cfg->enable_auto_alt_ref;
|
||||
oxcf->enable_auto_brf = extra_cfg->enable_auto_bwd_ref;
|
||||
oxcf->noise_sensitivity = extra_cfg->noise_sensitivity;
|
||||
|
|
@ -637,6 +643,8 @@ static aom_codec_err_t set_encoder_config(
|
|||
oxcf->superblock_size = AOM_SUPERBLOCK_SIZE_64X64;
|
||||
}
|
||||
|
||||
oxcf->row_mt = extra_cfg->row_mt;
|
||||
|
||||
oxcf->tile_columns = extra_cfg->tile_columns;
|
||||
oxcf->tile_rows = extra_cfg->tile_rows;
|
||||
|
||||
|
|
@ -692,6 +700,24 @@ static aom_codec_err_t set_encoder_config(
|
|||
|
||||
oxcf->frame_periodic_boost = extra_cfg->frame_periodic_boost;
|
||||
oxcf->motion_vector_unit_test = extra_cfg->motion_vector_unit_test;
|
||||
|
||||
#if CONFIG_REDUCED_ENCODER_BORDER
|
||||
if (oxcf->superres_mode != SUPERRES_NONE ||
|
||||
oxcf->resize_mode != RESIZE_NONE) {
|
||||
warn(
|
||||
"Superres / resize cannot be used with CONFIG_REDUCED_ENCODER_BORDER. "
|
||||
"Disabling superres/resize.\n");
|
||||
// return AOM_CODEC_INVALID_PARAM;
|
||||
disable_superres(oxcf);
|
||||
oxcf->resize_mode = RESIZE_NONE;
|
||||
oxcf->resize_scale_denominator = SCALE_NUMERATOR;
|
||||
oxcf->resize_kf_scale_denominator = SCALE_NUMERATOR;
|
||||
}
|
||||
#endif // CONFIG_REDUCED_ENCODER_BORDER
|
||||
|
||||
oxcf->chroma_subsampling_x = extra_cfg->chroma_subsampling_x;
|
||||
oxcf->chroma_subsampling_y = extra_cfg->chroma_subsampling_y;
|
||||
|
||||
return AOM_CODEC_OK;
|
||||
}
|
||||
|
||||
|
|
@ -731,6 +757,10 @@ static aom_codec_err_t encoder_set_config(aom_codec_alg_priv_t *ctx,
|
|||
return res;
|
||||
}
|
||||
|
||||
static aom_fixed_buf_t *encoder_get_global_headers(aom_codec_alg_priv_t *ctx) {
|
||||
return av1_get_global_headers(ctx->cpi);
|
||||
}
|
||||
|
||||
static aom_codec_err_t ctrl_get_quantizer(aom_codec_alg_priv_t *ctx,
|
||||
va_list args) {
|
||||
int *const arg = va_arg(args, int *);
|
||||
|
|
@ -765,12 +795,6 @@ static aom_codec_err_t ctrl_set_cpuused(aom_codec_alg_priv_t *ctx,
|
|||
return update_extra_cfg(ctx, &extra_cfg);
|
||||
}
|
||||
|
||||
static aom_codec_err_t ctrl_set_devsf(aom_codec_alg_priv_t *ctx, va_list args) {
|
||||
struct av1_extracfg extra_cfg = ctx->extra_cfg;
|
||||
extra_cfg.dev_sf = CAST(AOME_SET_DEVSF, args);
|
||||
return update_extra_cfg(ctx, &extra_cfg);
|
||||
}
|
||||
|
||||
static aom_codec_err_t ctrl_set_enable_auto_alt_ref(aom_codec_alg_priv_t *ctx,
|
||||
va_list args) {
|
||||
struct av1_extracfg extra_cfg = ctx->extra_cfg;
|
||||
|
|
@ -806,6 +830,13 @@ static aom_codec_err_t ctrl_set_static_thresh(aom_codec_alg_priv_t *ctx,
|
|||
return update_extra_cfg(ctx, &extra_cfg);
|
||||
}
|
||||
|
||||
static aom_codec_err_t ctrl_set_row_mt(aom_codec_alg_priv_t *ctx,
|
||||
va_list args) {
|
||||
struct av1_extracfg extra_cfg = ctx->extra_cfg;
|
||||
extra_cfg.row_mt = CAST(AV1E_SET_ROW_MT, args);
|
||||
return update_extra_cfg(ctx, &extra_cfg);
|
||||
}
|
||||
|
||||
static aom_codec_err_t ctrl_set_tile_columns(aom_codec_alg_priv_t *ctx,
|
||||
va_list args) {
|
||||
struct av1_extracfg extra_cfg = ctx->extra_cfg;
|
||||
|
|
@ -1669,6 +1700,20 @@ static aom_codec_err_t ctrl_set_superblock_size(aom_codec_alg_priv_t *ctx,
|
|||
return update_extra_cfg(ctx, &extra_cfg);
|
||||
}
|
||||
|
||||
static aom_codec_err_t ctrl_set_chroma_subsampling_x(aom_codec_alg_priv_t *ctx,
|
||||
va_list args) {
|
||||
struct av1_extracfg extra_cfg = ctx->extra_cfg;
|
||||
extra_cfg.chroma_subsampling_x = CAST(AV1E_SET_CHROMA_SUBSAMPLING_X, args);
|
||||
return update_extra_cfg(ctx, &extra_cfg);
|
||||
}
|
||||
|
||||
static aom_codec_err_t ctrl_set_chroma_subsampling_y(aom_codec_alg_priv_t *ctx,
|
||||
va_list args) {
|
||||
struct av1_extracfg extra_cfg = ctx->extra_cfg;
|
||||
extra_cfg.chroma_subsampling_y = CAST(AV1E_SET_CHROMA_SUBSAMPLING_Y, args);
|
||||
return update_extra_cfg(ctx, &extra_cfg);
|
||||
}
|
||||
|
||||
static aom_codec_ctrl_fn_map_t encoder_ctrl_maps[] = {
|
||||
{ AV1_COPY_REFERENCE, ctrl_copy_reference },
|
||||
{ AOME_USE_REFERENCE, ctrl_use_reference },
|
||||
|
|
@ -1681,11 +1726,11 @@ static aom_codec_ctrl_fn_map_t encoder_ctrl_maps[] = {
|
|||
{ AOME_SET_SCALEMODE, ctrl_set_scale_mode },
|
||||
{ AOME_SET_SPATIAL_LAYER_ID, ctrl_set_spatial_layer_id },
|
||||
{ AOME_SET_CPUUSED, ctrl_set_cpuused },
|
||||
{ AOME_SET_DEVSF, ctrl_set_devsf },
|
||||
{ AOME_SET_ENABLEAUTOALTREF, ctrl_set_enable_auto_alt_ref },
|
||||
{ AOME_SET_ENABLEAUTOBWDREF, ctrl_set_enable_auto_bwd_ref },
|
||||
{ AOME_SET_SHARPNESS, ctrl_set_sharpness },
|
||||
{ AOME_SET_STATIC_THRESHOLD, ctrl_set_static_thresh },
|
||||
{ AV1E_SET_ROW_MT, ctrl_set_row_mt },
|
||||
{ AV1E_SET_TILE_COLUMNS, ctrl_set_tile_columns },
|
||||
{ AV1E_SET_TILE_ROWS, ctrl_set_tile_rows },
|
||||
{ AOME_SET_ARNR_MAXFRAMES, ctrl_set_arnr_max_frames },
|
||||
|
|
@ -1754,7 +1799,8 @@ static aom_codec_ctrl_fn_map_t encoder_ctrl_maps[] = {
|
|||
{ AV1E_GET_ACTIVEMAP, ctrl_get_active_map },
|
||||
{ AV1_GET_NEW_FRAME_IMAGE, ctrl_get_new_frame_image },
|
||||
{ AV1_COPY_NEW_FRAME_IMAGE, ctrl_copy_new_frame_image },
|
||||
|
||||
{ AV1E_SET_CHROMA_SUBSAMPLING_X, ctrl_set_chroma_subsampling_x },
|
||||
{ AV1E_SET_CHROMA_SUBSAMPLING_Y, ctrl_set_chroma_subsampling_y },
|
||||
{ -1, NULL },
|
||||
};
|
||||
|
||||
|
|
@ -1850,13 +1896,13 @@ CODEC_INTERFACE(aom_codec_av1_cx) = {
|
|||
},
|
||||
{
|
||||
// NOLINT
|
||||
1, // 1 cfg map
|
||||
encoder_usage_cfg_map, // aom_codec_enc_cfg_map_t
|
||||
encoder_encode, // aom_codec_encode_fn_t
|
||||
encoder_get_cxdata, // aom_codec_get_cx_data_fn_t
|
||||
encoder_set_config, // aom_codec_enc_config_set_fn_t
|
||||
NULL, // aom_codec_get_global_headers_fn_t
|
||||
encoder_get_preview, // aom_codec_get_preview_frame_fn_t
|
||||
NULL // aom_codec_enc_mr_get_mem_loc_fn_t
|
||||
1, // 1 cfg map
|
||||
encoder_usage_cfg_map, // aom_codec_enc_cfg_map_t
|
||||
encoder_encode, // aom_codec_encode_fn_t
|
||||
encoder_get_cxdata, // aom_codec_get_cx_data_fn_t
|
||||
encoder_set_config, // aom_codec_enc_config_set_fn_t
|
||||
encoder_get_global_headers, // aom_codec_get_global_headers_fn_t
|
||||
encoder_get_preview, // aom_codec_get_preview_frame_fn_t
|
||||
NULL // aom_codec_enc_mr_get_mem_loc_fn_t
|
||||
}
|
||||
};
|
||||
|
|
|
|||
62
third_party/aom/av1/av1_dx_iface.c
vendored
62
third_party/aom/av1/av1_dx_iface.c
vendored
|
|
@ -26,6 +26,7 @@
|
|||
#include "av1/common/alloccommon.h"
|
||||
#include "av1/common/frame_buffers.h"
|
||||
#include "av1/common/enums.h"
|
||||
#include "av1/common/obu_util.h"
|
||||
|
||||
#include "av1/decoder/decoder.h"
|
||||
#include "av1/decoder/decodeframe.h"
|
||||
|
|
@ -46,6 +47,7 @@ struct aom_codec_alg_priv {
|
|||
int last_show_frame; // Index of last output frame.
|
||||
int byte_alignment;
|
||||
int skip_loop_filter;
|
||||
int skip_film_grain;
|
||||
int decode_tile_row;
|
||||
int decode_tile_col;
|
||||
unsigned int tile_mode;
|
||||
|
|
@ -103,6 +105,15 @@ static aom_codec_err_t decoder_init(aom_codec_ctx_t *ctx,
|
|||
priv->cfg.cfg.ext_partition = 1;
|
||||
}
|
||||
av1_zero(priv->image_with_grain);
|
||||
// Turn row_mt on by default.
|
||||
priv->row_mt = 1;
|
||||
|
||||
// Turn on normal tile coding mode by default.
|
||||
// 0 is for normal tile coding mode, and 1 is for large scale tile coding
|
||||
// mode(refer to lightfield example).
|
||||
priv->tile_mode = 0;
|
||||
priv->decode_tile_row = -1;
|
||||
priv->decode_tile_col = -1;
|
||||
}
|
||||
|
||||
return AOM_CODEC_OK;
|
||||
|
|
@ -216,7 +227,7 @@ static aom_codec_err_t decoder_peek_si_internal(const uint8_t *data,
|
|||
while (1) {
|
||||
data += bytes_read;
|
||||
data_sz -= bytes_read;
|
||||
const uint8_t *payload_start = data;
|
||||
if (data_sz < payload_size) return AOM_CODEC_CORRUPT_FRAME;
|
||||
// Check that the selected OBU is a sequence header
|
||||
if (obu_header.type == OBU_SEQUENCE_HEADER) {
|
||||
// Sanity check on sequence header size
|
||||
|
|
@ -264,9 +275,9 @@ static aom_codec_err_t decoder_peek_si_internal(const uint8_t *data,
|
|||
}
|
||||
}
|
||||
// skip past any unread OBU header data
|
||||
data = payload_start + payload_size;
|
||||
data += payload_size;
|
||||
data_sz -= payload_size;
|
||||
if (data_sz <= 0) break; // exit if we're out of OBUs
|
||||
if (data_sz == 0) break; // exit if we're out of OBUs
|
||||
status = aom_read_obu_header_and_size(
|
||||
data, data_sz, si->is_annexb, &obu_header, &payload_size, &bytes_read);
|
||||
if (status != AOM_CODEC_OK) return status;
|
||||
|
|
@ -313,6 +324,7 @@ static void init_buffer_callbacks(aom_codec_alg_priv_t *ctx) {
|
|||
cm->new_fb_idx = INVALID_IDX;
|
||||
cm->byte_alignment = ctx->byte_alignment;
|
||||
cm->skip_loop_filter = ctx->skip_loop_filter;
|
||||
cm->skip_film_grain = ctx->skip_film_grain;
|
||||
|
||||
if (ctx->get_ext_fb_cb != NULL && ctx->release_ext_fb_cb != NULL) {
|
||||
pool->get_fb_cb = ctx->get_ext_fb_cb;
|
||||
|
|
@ -434,7 +446,7 @@ static aom_codec_err_t init_decoder(aom_codec_alg_priv_t *ctx) {
|
|||
frame_worker_data->pbi->ext_tile_debug = ctx->ext_tile_debug;
|
||||
frame_worker_data->pbi->row_mt = ctx->row_mt;
|
||||
|
||||
worker->hook = (AVxWorkerHook)frame_worker_hook;
|
||||
worker->hook = frame_worker_hook;
|
||||
if (!winterface->reset(worker)) {
|
||||
set_error_detail(ctx, "Frame Worker thread creation failed");
|
||||
return AOM_CODEC_MEM_ERROR;
|
||||
|
|
@ -515,12 +527,11 @@ static aom_codec_err_t decode_one(aom_codec_alg_priv_t *ctx,
|
|||
static aom_codec_err_t decoder_decode(aom_codec_alg_priv_t *ctx,
|
||||
const uint8_t *data, size_t data_sz,
|
||||
void *user_priv) {
|
||||
const uint8_t *data_start = data;
|
||||
const uint8_t *data_end = data + data_sz;
|
||||
aom_codec_err_t res = AOM_CODEC_OK;
|
||||
|
||||
// Release any pending output frames from the previous decoder call.
|
||||
// We need to do this even if the decoder is being flushed
|
||||
// Release any pending output frames from the previous decoder_decode call.
|
||||
// We need to do this even if the decoder is being flushed or the input
|
||||
// arguments are invalid.
|
||||
if (ctx->frame_workers) {
|
||||
BufferPool *const pool = ctx->buffer_pool;
|
||||
RefCntBuffer *const frame_bufs = pool->frame_bufs;
|
||||
|
|
@ -538,10 +549,13 @@ static aom_codec_err_t decoder_decode(aom_codec_alg_priv_t *ctx,
|
|||
unlock_buffer_pool(ctx->buffer_pool);
|
||||
}
|
||||
|
||||
/* Sanity checks */
|
||||
/* NULL data ptr allowed if data_sz is 0 too */
|
||||
if (data == NULL && data_sz == 0) {
|
||||
ctx->flushed = 1;
|
||||
return AOM_CODEC_OK;
|
||||
}
|
||||
if (data == NULL || data_sz == 0) return AOM_CODEC_INVALID_PARAM;
|
||||
|
||||
// Reset flushed when receiving a valid frame.
|
||||
ctx->flushed = 0;
|
||||
|
|
@ -552,6 +566,9 @@ static aom_codec_err_t decoder_decode(aom_codec_alg_priv_t *ctx,
|
|||
if (res != AOM_CODEC_OK) return res;
|
||||
}
|
||||
|
||||
const uint8_t *data_start = data;
|
||||
const uint8_t *data_end = data + data_sz;
|
||||
|
||||
if (ctx->is_annexb) {
|
||||
// read the size of this temporal unit
|
||||
size_t length_of_size;
|
||||
|
|
@ -617,6 +634,7 @@ static aom_image_t *add_grain_if_needed(aom_image_t *img,
|
|||
img->fmt != grain_img_buf->fmt) {
|
||||
aom_img_free(grain_img_buf);
|
||||
grain_img_buf = NULL;
|
||||
*grain_img_ptr = NULL;
|
||||
}
|
||||
}
|
||||
if (!grain_img_buf) {
|
||||
|
|
@ -624,7 +642,14 @@ static aom_image_t *add_grain_if_needed(aom_image_t *img,
|
|||
*grain_img_ptr = grain_img_buf;
|
||||
}
|
||||
|
||||
av1_add_film_grain(grain_params, img, grain_img_buf);
|
||||
if (grain_img_buf) {
|
||||
grain_img_buf->user_priv = img->user_priv;
|
||||
if (av1_add_film_grain(grain_params, img, grain_img_buf)) {
|
||||
aom_img_free(grain_img_buf);
|
||||
grain_img_buf = NULL;
|
||||
*grain_img_ptr = NULL;
|
||||
}
|
||||
}
|
||||
|
||||
return grain_img_buf;
|
||||
}
|
||||
|
|
@ -720,8 +745,13 @@ static aom_image_t *decoder_get_frame(aom_codec_alg_priv_t *ctx,
|
|||
img = &ctx->img;
|
||||
img->temporal_id = cm->temporal_layer_id;
|
||||
img->spatial_id = cm->spatial_layer_id;
|
||||
if (cm->skip_film_grain) grain_params->apply_grain = 0;
|
||||
aom_image_t *res = add_grain_if_needed(
|
||||
img, &ctx->image_with_grain[*index], grain_params);
|
||||
if (!res) {
|
||||
aom_internal_error(&pbi->common.error, AOM_CODEC_CORRUPT_FRAME,
|
||||
"Grain systhesis failed\n");
|
||||
}
|
||||
*index += 1; // Advance the iterator to point to the next image
|
||||
return res;
|
||||
}
|
||||
|
|
@ -1128,6 +1158,19 @@ static aom_codec_err_t ctrl_set_skip_loop_filter(aom_codec_alg_priv_t *ctx,
|
|||
return AOM_CODEC_OK;
|
||||
}
|
||||
|
||||
static aom_codec_err_t ctrl_set_skip_film_grain(aom_codec_alg_priv_t *ctx,
|
||||
va_list args) {
|
||||
ctx->skip_film_grain = va_arg(args, int);
|
||||
|
||||
if (ctx->frame_workers) {
|
||||
AVxWorker *const worker = ctx->frame_workers;
|
||||
FrameWorkerData *const frame_worker_data = (FrameWorkerData *)worker->data1;
|
||||
frame_worker_data->pbi->common.skip_film_grain = ctx->skip_film_grain;
|
||||
}
|
||||
|
||||
return AOM_CODEC_OK;
|
||||
}
|
||||
|
||||
static aom_codec_err_t ctrl_get_accounting(aom_codec_alg_priv_t *ctx,
|
||||
va_list args) {
|
||||
#if !CONFIG_ACCOUNTING
|
||||
|
|
@ -1231,6 +1274,7 @@ static aom_codec_ctrl_fn_map_t decoder_ctrl_maps[] = {
|
|||
{ AV1D_EXT_TILE_DEBUG, ctrl_ext_tile_debug },
|
||||
{ AV1D_SET_ROW_MT, ctrl_set_row_mt },
|
||||
{ AV1D_SET_EXT_REF_PTR, ctrl_set_ext_ref_ptr },
|
||||
{ AV1D_SET_SKIP_FILM_GRAIN, ctrl_set_skip_film_grain },
|
||||
|
||||
// Getters
|
||||
{ AOMD_GET_FRAME_CORRUPTED, ctrl_get_frame_corrupted },
|
||||
|
|
|
|||
7
third_party/aom/av1/av1_iface_common.h
vendored
7
third_party/aom/av1/av1_iface_common.h
vendored
|
|
@ -8,10 +8,11 @@
|
|||
* Media Patent License 1.0 was not distributed with this source code in the
|
||||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
#ifndef AV1_AV1_IFACE_COMMON_H_
|
||||
#define AV1_AV1_IFACE_COMMON_H_
|
||||
#ifndef AOM_AV1_AV1_IFACE_COMMON_H_
|
||||
#define AOM_AV1_AV1_IFACE_COMMON_H_
|
||||
|
||||
#include "aom_ports/mem.h"
|
||||
#include "aom_scale/yv12config.h"
|
||||
|
||||
static void yuvconfig2image(aom_image_t *img, const YV12_BUFFER_CONFIG *yv12,
|
||||
void *user_priv) {
|
||||
|
|
@ -132,4 +133,4 @@ static aom_codec_err_t image2yuvconfig(const aom_image_t *img,
|
|||
return AOM_CODEC_OK;
|
||||
}
|
||||
|
||||
#endif // AV1_AV1_IFACE_COMMON_H_
|
||||
#endif // AOM_AV1_AV1_IFACE_COMMON_H_
|
||||
|
|
|
|||
6
third_party/aom/av1/common/alloccommon.h
vendored
6
third_party/aom/av1/common/alloccommon.h
vendored
|
|
@ -9,8 +9,8 @@
|
|||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#ifndef AV1_COMMON_ALLOCCOMMON_H_
|
||||
#define AV1_COMMON_ALLOCCOMMON_H_
|
||||
#ifndef AOM_AV1_COMMON_ALLOCCOMMON_H_
|
||||
#define AOM_AV1_COMMON_ALLOCCOMMON_H_
|
||||
|
||||
#define INVALID_IDX -1 // Invalid buffer index.
|
||||
|
||||
|
|
@ -45,4 +45,4 @@ int av1_get_MBs(int width, int height);
|
|||
} // extern "C"
|
||||
#endif
|
||||
|
||||
#endif // AV1_COMMON_ALLOCCOMMON_H_
|
||||
#endif // AOM_AV1_COMMON_ALLOCCOMMON_H_
|
||||
|
|
|
|||
2447
third_party/aom/av1/common/arm/av1_inv_txfm_neon.c
vendored
2447
third_party/aom/av1/common/arm/av1_inv_txfm_neon.c
vendored
File diff suppressed because it is too large
Load diff
|
|
@ -8,8 +8,8 @@
|
|||
* Media Patent License 1.0 was not distributed with this source code in the
|
||||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
#ifndef AV1_COMMON_ARM_AV1_INV_TXFM_NEON_H_
|
||||
#define AV1_COMMON_ARM_AV1_INV_TXFM_NEON_H_
|
||||
#ifndef AOM_AV1_COMMON_ARM_AV1_INV_TXFM_NEON_H_
|
||||
#define AOM_AV1_COMMON_ARM_AV1_INV_TXFM_NEON_H_
|
||||
|
||||
#include "config/aom_config.h"
|
||||
#include "config/av1_rtcd.h"
|
||||
|
|
@ -23,6 +23,8 @@
|
|||
typedef void (*transform_1d_neon)(const int32_t *input, int32_t *output,
|
||||
const int8_t cos_bit,
|
||||
const int8_t *stage_ptr);
|
||||
typedef void (*transform_neon)(int16x8_t *input, int16x8_t *output,
|
||||
int8_t cos_bit, int bit);
|
||||
|
||||
DECLARE_ALIGNED(16, static const int16_t, av1_eob_to_eobxy_8x8_default[8]) = {
|
||||
0x0707, 0x0707, 0x0707, 0x0707, 0x0707, 0x0707, 0x0707, 0x0707,
|
||||
|
|
@ -149,4 +151,4 @@ static INLINE void get_eobx_eoby_scan_h_identity(int *eobx, int *eoby,
|
|||
*eoby = eob_fill[temp_eoby];
|
||||
}
|
||||
|
||||
#endif // AV1_COMMON_ARM_AV1_INV_TXFM_NEON_H_
|
||||
#endif // AOM_AV1_COMMON_ARM_AV1_INV_TXFM_NEON_H_
|
||||
|
|
|
|||
|
|
@ -34,8 +34,8 @@ void aom_blend_a64_hmask_neon(uint8_t *dst, uint32_t dst_stride,
|
|||
uint8x8_t tmp0, tmp1;
|
||||
uint8x16_t res_q;
|
||||
uint16x8_t res, res_low, res_high;
|
||||
uint32x2_t tmp0_32, tmp1_32;
|
||||
uint16x4_t tmp0_16, tmp1_16;
|
||||
uint32x2_t tmp0_32 = vdup_n_u32(0), tmp1_32 = vdup_n_u32(0);
|
||||
uint16x4_t tmp0_16 = vdup_n_u16(0), tmp1_16 = vdup_n_u16(0);
|
||||
const uint8x8_t vdup_64 = vdup_n_u8((uint8_t)64);
|
||||
|
||||
if (w >= 16) {
|
||||
|
|
|
|||
|
|
@ -27,8 +27,8 @@ void aom_blend_a64_vmask_neon(uint8_t *dst, uint32_t dst_stride,
|
|||
uint8x8_t tmp0, tmp1;
|
||||
uint8x16_t tmp0_q, tmp1_q, res_q;
|
||||
uint16x8_t res, res_low, res_high;
|
||||
uint32x2_t tmp0_32, tmp1_32;
|
||||
uint16x4_t tmp0_16, tmp1_16;
|
||||
uint32x2_t tmp0_32 = vdup_n_u32(0), tmp1_32 = vdup_n_u32(0);
|
||||
uint16x4_t tmp0_16 = vdup_n_u16(0), tmp1_16 = vdup_n_u16(0);
|
||||
assert(IMPLIES(src0 == dst, src0_stride == dst_stride));
|
||||
assert(IMPLIES(src1 == dst, src1_stride == dst_stride));
|
||||
|
||||
|
|
|
|||
4
third_party/aom/av1/common/arm/cfl_neon.c
vendored
4
third_party/aom/av1/common/arm/cfl_neon.c
vendored
|
|
@ -131,7 +131,7 @@ static void cfl_luma_subsampling_444_lbd_neon(const uint8_t *input,
|
|||
} while ((pred_buf_q3 += CFL_BUF_LINE) < end);
|
||||
}
|
||||
|
||||
#if __ARM_ARCH <= 7
|
||||
#ifndef __aarch64__
|
||||
uint16x8_t vpaddq_u16(uint16x8_t a, uint16x8_t b) {
|
||||
return vcombine_u16(vpadd_u16(vget_low_u16(a), vget_high_u16(a)),
|
||||
vpadd_u16(vget_low_u16(b), vget_high_u16(b)));
|
||||
|
|
@ -311,7 +311,7 @@ static INLINE void subtract_average_neon(const uint16_t *src, int16_t *dst,
|
|||
|
||||
// Permute and add in such a way that each lane contains the block sum.
|
||||
// [A+C+B+D, B+D+A+C, C+A+D+B, D+B+C+A]
|
||||
#if __ARM_ARCH >= 8
|
||||
#ifdef __aarch64__
|
||||
sum_32x4 = vpaddq_u32(sum_32x4, sum_32x4);
|
||||
sum_32x4 = vpaddq_u32(sum_32x4, sum_32x4);
|
||||
#else
|
||||
|
|
|
|||
363
third_party/aom/av1/common/arm/convolve_neon.c
vendored
363
third_party/aom/av1/common/arm/convolve_neon.c
vendored
|
|
@ -13,6 +13,8 @@
|
|||
#include <assert.h>
|
||||
#include <arm_neon.h>
|
||||
|
||||
#include "config/av1_rtcd.h"
|
||||
|
||||
#include "aom_dsp/aom_dsp_common.h"
|
||||
#include "aom_ports/mem.h"
|
||||
#include "av1/common/convolve.h"
|
||||
|
|
@ -68,6 +70,33 @@ static INLINE uint8x8_t convolve8_horiz_8x8(
|
|||
return vqmovun_s16(sum);
|
||||
}
|
||||
|
||||
#if !defined(__aarch64__)
|
||||
static INLINE uint8x8_t convolve8_horiz_4x1(
|
||||
const int16x4_t s0, const int16x4_t s1, const int16x4_t s2,
|
||||
const int16x4_t s3, const int16x4_t s4, const int16x4_t s5,
|
||||
const int16x4_t s6, const int16x4_t s7, const int16_t *filter,
|
||||
const int16x4_t shift_round_0, const int16x4_t shift_by_bits) {
|
||||
int16x4_t sum;
|
||||
|
||||
sum = vmul_n_s16(s0, filter[0]);
|
||||
sum = vmla_n_s16(sum, s1, filter[1]);
|
||||
sum = vmla_n_s16(sum, s2, filter[2]);
|
||||
sum = vmla_n_s16(sum, s5, filter[5]);
|
||||
sum = vmla_n_s16(sum, s6, filter[6]);
|
||||
sum = vmla_n_s16(sum, s7, filter[7]);
|
||||
/* filter[3] can take a max value of 128. So the max value of the result :
|
||||
* 128*255 + sum > 16 bits
|
||||
*/
|
||||
sum = vqadd_s16(sum, vmul_n_s16(s3, filter[3]));
|
||||
sum = vqadd_s16(sum, vmul_n_s16(s4, filter[4]));
|
||||
|
||||
sum = vqrshl_s16(sum, shift_round_0);
|
||||
sum = vqrshl_s16(sum, shift_by_bits);
|
||||
|
||||
return vqmovun_s16(vcombine_s16(sum, sum));
|
||||
}
|
||||
#endif // !defined(__arch64__)
|
||||
|
||||
static INLINE uint8x8_t convolve8_vert_8x4(
|
||||
const int16x8_t s0, const int16x8_t s1, const int16x8_t s2,
|
||||
const int16x8_t s3, const int16x8_t s4, const int16x8_t s5,
|
||||
|
|
@ -175,7 +204,10 @@ void av1_convolve_x_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
|
|||
(void)conv_params;
|
||||
(void)filter_params_y;
|
||||
|
||||
uint8x8_t t0, t1, t2, t3;
|
||||
uint8x8_t t0;
|
||||
#if defined(__aarch64__)
|
||||
uint8x8_t t1, t2, t3;
|
||||
#endif
|
||||
|
||||
assert(bits >= 0);
|
||||
assert((FILTER_BITS - conv_params->round_1) >= 0 ||
|
||||
|
|
@ -188,7 +220,7 @@ void av1_convolve_x_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
|
|||
const int16x8_t shift_by_bits = vdupq_n_s16(-bits);
|
||||
|
||||
src -= horiz_offset;
|
||||
|
||||
#if defined(__aarch64__)
|
||||
if (h == 4) {
|
||||
uint8x8_t d01, d23;
|
||||
int16x4_t s0, s1, s2, s3, s4, s5, s6, s7, s8, s9, s10, d0, d1, d2, d3;
|
||||
|
|
@ -275,12 +307,18 @@ void av1_convolve_x_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
|
|||
w -= 4;
|
||||
} while (w > 0);
|
||||
} else {
|
||||
#endif
|
||||
int width;
|
||||
const uint8_t *s;
|
||||
int16x8_t s0, s1, s2, s3, s4, s5, s6, s7;
|
||||
|
||||
#if defined(__aarch64__)
|
||||
int16x8_t s8, s9, s10;
|
||||
uint8x8_t t4, t5, t6, t7;
|
||||
int16x8_t s0, s1, s2, s3, s4, s5, s6, s7, s8, s9, s10;
|
||||
#endif
|
||||
|
||||
if (w <= 4) {
|
||||
#if defined(__aarch64__)
|
||||
do {
|
||||
load_u8_8x8(src, src_stride, &t0, &t1, &t2, &t3, &t4, &t5, &t6, &t7);
|
||||
transpose_u8_8x8(&t0, &t1, &t2, &t3, &t4, &t5, &t6, &t7);
|
||||
|
|
@ -387,10 +425,49 @@ void av1_convolve_x_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
|
|||
}
|
||||
h -= 8;
|
||||
} while (h > 0);
|
||||
#else
|
||||
int16x8_t tt0;
|
||||
int16x4_t x0, x1, x2, x3, x4, x5, x6, x7;
|
||||
const int16x4_t shift_round_0_low = vget_low_s16(shift_round_0);
|
||||
const int16x4_t shift_by_bits_low = vget_low_s16(shift_by_bits);
|
||||
do {
|
||||
t0 = vld1_u8(src); // a0 a1 a2 a3 a4 a5 a6 a7
|
||||
tt0 = vreinterpretq_s16_u16(vmovl_u8(t0));
|
||||
x0 = vget_low_s16(tt0); // a0 a1 a2 a3
|
||||
x4 = vget_high_s16(tt0); // a4 a5 a6 a7
|
||||
|
||||
t0 = vld1_u8(src + 8); // a8 a9 a10 a11 a12 a13 a14 a15
|
||||
tt0 = vreinterpretq_s16_u16(vmovl_u8(t0));
|
||||
x7 = vget_low_s16(tt0); // a8 a9 a10 a11
|
||||
|
||||
x1 = vext_s16(x0, x4, 1); // a1 a2 a3 a4
|
||||
x2 = vext_s16(x0, x4, 2); // a2 a3 a4 a5
|
||||
x3 = vext_s16(x0, x4, 3); // a3 a4 a5 a6
|
||||
x5 = vext_s16(x4, x7, 1); // a5 a6 a7 a8
|
||||
x6 = vext_s16(x4, x7, 2); // a6 a7 a8 a9
|
||||
x7 = vext_s16(x4, x7, 3); // a7 a8 a9 a10
|
||||
|
||||
src += src_stride;
|
||||
|
||||
t0 = convolve8_horiz_4x1(x0, x1, x2, x3, x4, x5, x6, x7, x_filter,
|
||||
shift_round_0_low, shift_by_bits_low);
|
||||
|
||||
if (w == 4) {
|
||||
vst1_lane_u32((uint32_t *)dst, vreinterpret_u32_u8(t0),
|
||||
0); // 00 01 02 03
|
||||
dst += dst_stride;
|
||||
} else if (w == 2) {
|
||||
vst1_lane_u16((uint16_t *)dst, vreinterpret_u16_u8(t0), 0); // 00 01
|
||||
dst += dst_stride;
|
||||
}
|
||||
h -= 1;
|
||||
} while (h > 0);
|
||||
#endif
|
||||
} else {
|
||||
uint8_t *d;
|
||||
int16x8_t s11, s12, s13, s14;
|
||||
|
||||
int16x8_t s11;
|
||||
#if defined(__aarch64__)
|
||||
int16x8_t s12, s13, s14;
|
||||
do {
|
||||
__builtin_prefetch(src + 0 * src_stride);
|
||||
__builtin_prefetch(src + 1 * src_stride);
|
||||
|
|
@ -479,8 +556,47 @@ void av1_convolve_x_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
|
|||
dst += 8 * dst_stride;
|
||||
h -= 8;
|
||||
} while (h > 0);
|
||||
#else
|
||||
do {
|
||||
t0 = vld1_u8(src); // a0 a1 a2 a3 a4 a5 a6 a7
|
||||
s0 = vreinterpretq_s16_u16(vmovl_u8(t0));
|
||||
|
||||
width = w;
|
||||
s = src + 8;
|
||||
d = dst;
|
||||
__builtin_prefetch(dst);
|
||||
|
||||
do {
|
||||
t0 = vld1_u8(s); // a8 a9 a10 a11 a12 a13 a14 a15
|
||||
s7 = vreinterpretq_s16_u16(vmovl_u8(t0));
|
||||
s11 = s0;
|
||||
s0 = s7;
|
||||
|
||||
s1 = vextq_s16(s11, s7, 1); // a1 a2 a3 a4 a5 a6 a7 a8
|
||||
s2 = vextq_s16(s11, s7, 2); // a2 a3 a4 a5 a6 a7 a8 a9
|
||||
s3 = vextq_s16(s11, s7, 3); // a3 a4 a5 a6 a7 a8 a9 a10
|
||||
s4 = vextq_s16(s11, s7, 4); // a4 a5 a6 a7 a8 a9 a10 a11
|
||||
s5 = vextq_s16(s11, s7, 5); // a5 a6 a7 a8 a9 a10 a11 a12
|
||||
s6 = vextq_s16(s11, s7, 6); // a6 a7 a8 a9 a10 a11 a12 a13
|
||||
s7 = vextq_s16(s11, s7, 7); // a7 a8 a9 a10 a11 a12 a13 a14
|
||||
|
||||
t0 = convolve8_horiz_8x8(s11, s1, s2, s3, s4, s5, s6, s7, x_filter,
|
||||
shift_round_0, shift_by_bits);
|
||||
vst1_u8(d, t0);
|
||||
|
||||
s += 8;
|
||||
d += 8;
|
||||
width -= 8;
|
||||
} while (width > 0);
|
||||
src += src_stride;
|
||||
dst += dst_stride;
|
||||
h -= 1;
|
||||
} while (h > 0);
|
||||
#endif
|
||||
}
|
||||
#if defined(__aarch64__)
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
void av1_convolve_y_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
|
||||
|
|
@ -505,9 +621,12 @@ void av1_convolve_y_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
|
|||
filter_params_y, subpel_y_q4 & SUBPEL_MASK);
|
||||
|
||||
if (w <= 4) {
|
||||
uint8x8_t d01, d23;
|
||||
int16x4_t s0, s1, s2, s3, s4, s5, s6, s7, s8, s9, s10, d0, d1, d2, d3;
|
||||
|
||||
uint8x8_t d01;
|
||||
int16x4_t s0, s1, s2, s3, s4, s5, s6, s7, d0;
|
||||
#if defined(__aarch64__)
|
||||
uint8x8_t d23;
|
||||
int16x4_t s8, s9, s10, d1, d2, d3;
|
||||
#endif
|
||||
s0 = vreinterpret_s16_u16(vget_low_u16(vmovl_u8(vld1_u8(src))));
|
||||
src += src_stride;
|
||||
s1 = vreinterpret_s16_u16(vget_low_u16(vmovl_u8(vld1_u8(src))));
|
||||
|
|
@ -526,6 +645,7 @@ void av1_convolve_y_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
|
|||
do {
|
||||
s7 = vreinterpret_s16_u16(vget_low_u16(vmovl_u8(vld1_u8(src))));
|
||||
src += src_stride;
|
||||
#if defined(__aarch64__)
|
||||
s8 = vreinterpret_s16_u16(vget_low_u16(vmovl_u8(vld1_u8(src))));
|
||||
src += src_stride;
|
||||
s9 = vreinterpret_s16_u16(vget_low_u16(vmovl_u8(vld1_u8(src))));
|
||||
|
|
@ -591,14 +711,41 @@ void av1_convolve_y_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
|
|||
s5 = s9;
|
||||
s6 = s10;
|
||||
h -= 4;
|
||||
#else
|
||||
__builtin_prefetch(dst + 0 * dst_stride);
|
||||
__builtin_prefetch(src + 0 * src_stride);
|
||||
|
||||
d0 = convolve8_4x4(s0, s1, s2, s3, s4, s5, s6, s7, y_filter);
|
||||
|
||||
d01 = vqrshrun_n_s16(vcombine_s16(d0, d0), FILTER_BITS);
|
||||
|
||||
if (w == 4) {
|
||||
vst1_lane_u32((uint32_t *)dst, vreinterpret_u32_u8(d01), 0);
|
||||
dst += dst_stride;
|
||||
} else if (w == 2) {
|
||||
vst1_lane_u16((uint16_t *)dst, vreinterpret_u16_u8(d01), 0);
|
||||
dst += dst_stride;
|
||||
}
|
||||
s0 = s1;
|
||||
s1 = s2;
|
||||
s2 = s3;
|
||||
s3 = s4;
|
||||
s4 = s5;
|
||||
s5 = s6;
|
||||
s6 = s7;
|
||||
h -= 1;
|
||||
#endif
|
||||
} while (h > 0);
|
||||
} else {
|
||||
int height;
|
||||
const uint8_t *s;
|
||||
uint8_t *d;
|
||||
uint8x8_t t0, t1, t2, t3;
|
||||
int16x8_t s0, s1, s2, s3, s4, s5, s6, s7, s8, s9, s10;
|
||||
|
||||
uint8x8_t t0;
|
||||
int16x8_t s0, s1, s2, s3, s4, s5, s6, s7;
|
||||
#if defined(__aarch64__)
|
||||
uint8x8_t t1, t2, t3;
|
||||
int16x8_t s8, s9, s10;
|
||||
#endif
|
||||
do {
|
||||
__builtin_prefetch(src + 0 * src_stride);
|
||||
__builtin_prefetch(src + 1 * src_stride);
|
||||
|
|
@ -628,6 +775,7 @@ void av1_convolve_y_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
|
|||
do {
|
||||
s7 = vreinterpretq_s16_u16(vmovl_u8(vld1_u8(s)));
|
||||
s += src_stride;
|
||||
#if defined(__aarch64__)
|
||||
s8 = vreinterpretq_s16_u16(vmovl_u8(vld1_u8(s)));
|
||||
s += src_stride;
|
||||
s9 = vreinterpretq_s16_u16(vmovl_u8(vld1_u8(s)));
|
||||
|
|
@ -670,6 +818,24 @@ void av1_convolve_y_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
|
|||
s5 = s9;
|
||||
s6 = s10;
|
||||
height -= 4;
|
||||
#else
|
||||
__builtin_prefetch(d);
|
||||
__builtin_prefetch(s);
|
||||
|
||||
t0 = convolve8_vert_8x4(s0, s1, s2, s3, s4, s5, s6, s7, y_filter);
|
||||
|
||||
vst1_u8(d, t0);
|
||||
d += dst_stride;
|
||||
|
||||
s0 = s1;
|
||||
s1 = s2;
|
||||
s2 = s3;
|
||||
s3 = s4;
|
||||
s4 = s5;
|
||||
s5 = s6;
|
||||
s6 = s7;
|
||||
height -= 1;
|
||||
#endif
|
||||
} while (height > 0);
|
||||
src += 8;
|
||||
dst += 8;
|
||||
|
|
@ -686,7 +852,10 @@ void av1_convolve_2d_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
|
|||
ConvolveParams *conv_params) {
|
||||
int im_dst_stride;
|
||||
int width, height;
|
||||
uint8x8_t t0, t1, t2, t3, t4, t5, t6, t7;
|
||||
uint8x8_t t0;
|
||||
#if defined(__aarch64__)
|
||||
uint8x8_t t1, t2, t3, t4, t5, t6, t7;
|
||||
#endif
|
||||
|
||||
DECLARE_ALIGNED(16, int16_t,
|
||||
im_block[(MAX_SB_SIZE + HORIZ_EXTRA_ROWS) * MAX_SB_SIZE]);
|
||||
|
|
@ -724,13 +893,18 @@ void av1_convolve_2d_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
|
|||
assert(conv_params->round_0 > 0);
|
||||
|
||||
if (w <= 4) {
|
||||
int16x4_t s0, s1, s2, s3, s4, s5, s6, s7, s8, s9, s10, d0, d1, d2, d3;
|
||||
int16x4_t s0, s1, s2, s3, s4, s5, s6, s7, d0;
|
||||
#if defined(__aarch64__)
|
||||
int16x4_t s8, s9, s10, d1, d2, d3;
|
||||
#endif
|
||||
|
||||
const int16x4_t horiz_const = vdup_n_s16((1 << (bd + FILTER_BITS - 2)));
|
||||
const int16x4_t shift_round_0 = vdup_n_s16(-(conv_params->round_0 - 1));
|
||||
|
||||
do {
|
||||
s = src_ptr;
|
||||
|
||||
#if defined(__aarch64__)
|
||||
__builtin_prefetch(s + 0 * src_stride);
|
||||
__builtin_prefetch(s + 1 * src_stride);
|
||||
__builtin_prefetch(s + 2 * src_stride);
|
||||
|
|
@ -789,16 +963,56 @@ void av1_convolve_2d_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
|
|||
src_ptr += 4 * src_stride;
|
||||
dst_ptr += 4 * im_dst_stride;
|
||||
height -= 4;
|
||||
#else
|
||||
int16x8_t tt0;
|
||||
|
||||
__builtin_prefetch(s);
|
||||
|
||||
t0 = vld1_u8(s); // a0 a1 a2 a3 a4 a5 a6 a7
|
||||
tt0 = vreinterpretq_s16_u16(vmovl_u8(t0));
|
||||
s0 = vget_low_s16(tt0);
|
||||
s4 = vget_high_s16(tt0);
|
||||
|
||||
__builtin_prefetch(dst_ptr);
|
||||
s += 8;
|
||||
|
||||
t0 = vld1_u8(s); // a8 a9 a10 a11 a12 a13 a14 a15
|
||||
s7 = vget_low_s16(vreinterpretq_s16_u16(vmovl_u8(t0)));
|
||||
|
||||
s1 = vext_s16(s0, s4, 1); // a1 a2 a3 a4
|
||||
s2 = vext_s16(s0, s4, 2); // a2 a3 a4 a5
|
||||
s3 = vext_s16(s0, s4, 3); // a3 a4 a5 a6
|
||||
s5 = vext_s16(s4, s7, 1); // a5 a6 a7 a8
|
||||
s6 = vext_s16(s4, s7, 2); // a6 a7 a8 a9
|
||||
s7 = vext_s16(s4, s7, 3); // a7 a8 a9 a10
|
||||
|
||||
d0 = convolve8_4x4_s16(s0, s1, s2, s3, s4, s5, s6, s7, x_filter_tmp,
|
||||
horiz_const, shift_round_0);
|
||||
|
||||
if (w == 4) {
|
||||
vst1_s16(dst_ptr, d0);
|
||||
dst_ptr += im_dst_stride;
|
||||
} else if (w == 2) {
|
||||
vst1_lane_u32((uint32_t *)dst_ptr, vreinterpret_u32_s16(d0), 0);
|
||||
dst_ptr += im_dst_stride;
|
||||
}
|
||||
|
||||
src_ptr += src_stride;
|
||||
height -= 1;
|
||||
#endif
|
||||
} while (height > 0);
|
||||
} else {
|
||||
int16_t *d_tmp;
|
||||
int16x8_t s0, s1, s2, s3, s4, s5, s6, s7, res0;
|
||||
#if defined(__aarch64__)
|
||||
int16x8_t s8, s9, s10, res1, res2, res3, res4, res5, res6, res7;
|
||||
int16x8_t s11, s12, s13, s14;
|
||||
int16x8_t s0, s1, s2, s3, s4, s5, s6, s7, s8, s9, s10;
|
||||
int16x8_t res0, res1, res2, res3, res4, res5, res6, res7;
|
||||
#endif
|
||||
|
||||
const int16x8_t horiz_const = vdupq_n_s16((1 << (bd + FILTER_BITS - 2)));
|
||||
const int16x8_t shift_round_0 = vdupq_n_s16(-(conv_params->round_0 - 1));
|
||||
|
||||
#if defined(__aarch64__)
|
||||
do {
|
||||
__builtin_prefetch(src_ptr + 0 * src_stride);
|
||||
__builtin_prefetch(src_ptr + 1 * src_stride);
|
||||
|
|
@ -886,6 +1100,45 @@ void av1_convolve_2d_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
|
|||
dst_ptr += 8 * im_dst_stride;
|
||||
height -= 8;
|
||||
} while (height > 0);
|
||||
#else
|
||||
do {
|
||||
t0 = vld1_u8(src_ptr);
|
||||
s0 = vreinterpretq_s16_u16(vmovl_u8(t0)); // a0 a1 a2 a3 a4 a5 a6 a7
|
||||
|
||||
width = w;
|
||||
s = src_ptr + 8;
|
||||
d_tmp = dst_ptr;
|
||||
|
||||
__builtin_prefetch(dst_ptr);
|
||||
|
||||
do {
|
||||
t0 = vld1_u8(s); // a8 a9 a10 a11 a12 a13 a14 a15
|
||||
s7 = vreinterpretq_s16_u16(vmovl_u8(t0));
|
||||
int16x8_t sum = s0;
|
||||
s0 = s7;
|
||||
|
||||
s1 = vextq_s16(sum, s7, 1); // a1 a2 a3 a4 a5 a6 a7 a8
|
||||
s2 = vextq_s16(sum, s7, 2); // a2 a3 a4 a5 a6 a7 a8 a9
|
||||
s3 = vextq_s16(sum, s7, 3); // a3 a4 a5 a6 a7 a8 a9 a10
|
||||
s4 = vextq_s16(sum, s7, 4); // a4 a5 a6 a7 a8 a9 a10 a11
|
||||
s5 = vextq_s16(sum, s7, 5); // a5 a6 a7 a8 a9 a10 a11 a12
|
||||
s6 = vextq_s16(sum, s7, 6); // a6 a7 a8 a9 a10 a11 a12 a13
|
||||
s7 = vextq_s16(sum, s7, 7); // a7 a8 a9 a10 a11 a12 a13 a14
|
||||
|
||||
res0 = convolve8_8x8_s16(sum, s1, s2, s3, s4, s5, s6, s7, x_filter_tmp,
|
||||
horiz_const, shift_round_0);
|
||||
|
||||
vst1q_s16(d_tmp, res0);
|
||||
|
||||
s += 8;
|
||||
d_tmp += 8;
|
||||
width -= 8;
|
||||
} while (width > 0);
|
||||
src_ptr += src_stride;
|
||||
dst_ptr += im_dst_stride;
|
||||
height -= 1;
|
||||
} while (height > 0);
|
||||
#endif
|
||||
}
|
||||
|
||||
// vertical
|
||||
|
|
@ -910,10 +1163,17 @@ void av1_convolve_2d_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
|
|||
width = w;
|
||||
|
||||
if (width <= 4) {
|
||||
int16x4_t s0, s1, s2, s3, s4, s5, s6, s7, s8, s9, s10;
|
||||
uint16x4_t d0, d1, d2, d3;
|
||||
uint16x8_t dd0, dd1;
|
||||
uint8x8_t d01, d23;
|
||||
int16x4_t s0, s1, s2, s3, s4, s5, s6, s7;
|
||||
uint16x4_t d0;
|
||||
uint16x8_t dd0;
|
||||
uint8x8_t d01;
|
||||
|
||||
#if defined(__aarch64__)
|
||||
int16x4_t s8, s9, s10;
|
||||
uint16x4_t d1, d2, d3;
|
||||
uint16x8_t dd1;
|
||||
uint8x8_t d23;
|
||||
#endif
|
||||
|
||||
d_u8 = dst_u8_ptr;
|
||||
v_s = v_src_ptr;
|
||||
|
|
@ -931,6 +1191,7 @@ void av1_convolve_2d_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
|
|||
v_s += (7 * im_stride);
|
||||
|
||||
do {
|
||||
#if defined(__aarch64__)
|
||||
load_s16_4x4(v_s, im_stride, &s7, &s8, &s9, &s10);
|
||||
v_s += (im_stride << 2);
|
||||
|
||||
|
|
@ -1008,11 +1269,48 @@ void av1_convolve_2d_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
|
|||
s5 = s9;
|
||||
s6 = s10;
|
||||
height -= 4;
|
||||
#else
|
||||
s7 = vld1_s16(v_s);
|
||||
v_s += im_stride;
|
||||
|
||||
__builtin_prefetch(d_u8 + 0 * dst_stride);
|
||||
|
||||
d0 = convolve8_vert_4x4_s32(s0, s1, s2, s3, s4, s5, s6, s7, y_filter,
|
||||
round_shift_vec, offset_const,
|
||||
sub_const_vec);
|
||||
|
||||
dd0 = vqrshlq_u16(vcombine_u16(d0, d0), vec_round_bits);
|
||||
d01 = vqmovn_u16(dd0);
|
||||
|
||||
if (w == 4) {
|
||||
vst1_lane_u32((uint32_t *)d_u8, vreinterpret_u32_u8(d01),
|
||||
0); // 00 01 02 03
|
||||
d_u8 += dst_stride;
|
||||
|
||||
} else if (w == 2) {
|
||||
vst1_lane_u16((uint16_t *)d_u8, vreinterpret_u16_u8(d01),
|
||||
0); // 00 01
|
||||
d_u8 += dst_stride;
|
||||
}
|
||||
|
||||
s0 = s1;
|
||||
s1 = s2;
|
||||
s2 = s3;
|
||||
s3 = s4;
|
||||
s4 = s5;
|
||||
s5 = s6;
|
||||
s6 = s7;
|
||||
height -= 1;
|
||||
#endif
|
||||
} while (height > 0);
|
||||
} else {
|
||||
// if width is a multiple of 8 & height is a multiple of 4
|
||||
int16x8_t s0, s1, s2, s3, s4, s5, s6, s7, s8, s9, s10;
|
||||
uint8x8_t res0, res1, res2, res3;
|
||||
int16x8_t s0, s1, s2, s3, s4, s5, s6, s7;
|
||||
uint8x8_t res0;
|
||||
#if defined(__aarch64__)
|
||||
int16x8_t s8, s9, s10;
|
||||
uint8x8_t res1, res2, res3;
|
||||
#endif
|
||||
|
||||
do {
|
||||
__builtin_prefetch(v_src_ptr + 0 * im_stride);
|
||||
|
|
@ -1032,6 +1330,7 @@ void av1_convolve_2d_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
|
|||
height = h;
|
||||
|
||||
do {
|
||||
#if defined(__aarch64__)
|
||||
load_s16_8x4(v_s, im_stride, &s7, &s8, &s9, &s10);
|
||||
v_s += (im_stride << 2);
|
||||
|
||||
|
|
@ -1076,6 +1375,28 @@ void av1_convolve_2d_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
|
|||
s5 = s9;
|
||||
s6 = s10;
|
||||
height -= 4;
|
||||
#else
|
||||
s7 = vld1q_s16(v_s);
|
||||
v_s += im_stride;
|
||||
|
||||
__builtin_prefetch(d_u8 + 0 * dst_stride);
|
||||
|
||||
res0 = convolve8_vert_8x4_s32(s0, s1, s2, s3, s4, s5, s6, s7,
|
||||
y_filter, round_shift_vec, offset_const,
|
||||
sub_const_vec, vec_round_bits);
|
||||
|
||||
vst1_u8(d_u8, res0);
|
||||
d_u8 += dst_stride;
|
||||
|
||||
s0 = s1;
|
||||
s1 = s2;
|
||||
s2 = s3;
|
||||
s3 = s4;
|
||||
s4 = s5;
|
||||
s5 = s6;
|
||||
s6 = s7;
|
||||
height -= 1;
|
||||
#endif
|
||||
} while (height > 0);
|
||||
v_src_ptr += 8;
|
||||
dst_u8_ptr += 8;
|
||||
|
|
|
|||
|
|
@ -8,8 +8,8 @@
|
|||
* be found in the AUTHORS file in the root of the source tree.
|
||||
*/
|
||||
|
||||
#ifndef AV1_COMMON_ARM_CONVOLVE_NEON_H_
|
||||
#define AV1_COMMON_ARM_CONVOLVE_NEON_H_
|
||||
#ifndef AOM_AV1_COMMON_ARM_CONVOLVE_NEON_H_
|
||||
#define AOM_AV1_COMMON_ARM_CONVOLVE_NEON_H_
|
||||
|
||||
#include <arm_neon.h>
|
||||
|
||||
|
|
@ -225,4 +225,4 @@ static INLINE uint16x4_t convolve8_4x4_s32(
|
|||
return res;
|
||||
}
|
||||
|
||||
#endif // AV1_COMMON_ARM_CONVOLVE_NEON_H_
|
||||
#endif // AOM_AV1_COMMON_ARM_CONVOLVE_NEON_H_
|
||||
|
|
|
|||
512
third_party/aom/av1/common/arm/jnt_convolve_neon.c
vendored
512
third_party/aom/av1/common/arm/jnt_convolve_neon.c
vendored
|
|
@ -22,12 +22,108 @@
|
|||
#include "av1/common/arm/mem_neon.h"
|
||||
#include "av1/common/arm/transpose_neon.h"
|
||||
|
||||
#if !defined(__aarch64__)
|
||||
static INLINE void compute_avg_4x1(uint16x4_t res0, uint16x4_t d0,
|
||||
const uint16_t fwd_offset,
|
||||
const uint16_t bck_offset,
|
||||
const int16x4_t sub_const_vec,
|
||||
const int16_t round_bits,
|
||||
const int use_jnt_comp_avg, uint8x8_t *t0) {
|
||||
int16x4_t tmp0;
|
||||
uint16x4_t tmp_u0;
|
||||
uint32x4_t sum0;
|
||||
int32x4_t dst0;
|
||||
int16x8_t tmp4;
|
||||
|
||||
if (use_jnt_comp_avg) {
|
||||
const int32x4_t round_bits_vec = vdupq_n_s32((int32_t)(-round_bits));
|
||||
|
||||
sum0 = vmull_n_u16(res0, fwd_offset);
|
||||
sum0 = vmlal_n_u16(sum0, d0, bck_offset);
|
||||
|
||||
sum0 = vshrq_n_u32(sum0, DIST_PRECISION_BITS);
|
||||
|
||||
dst0 = vsubq_s32(vreinterpretq_s32_u32(sum0), vmovl_s16(sub_const_vec));
|
||||
|
||||
dst0 = vqrshlq_s32(dst0, round_bits_vec);
|
||||
|
||||
tmp0 = vqmovn_s32(dst0);
|
||||
tmp4 = vcombine_s16(tmp0, tmp0);
|
||||
|
||||
*t0 = vqmovun_s16(tmp4);
|
||||
} else {
|
||||
const int16x4_t round_bits_vec = vdup_n_s16(-round_bits);
|
||||
tmp_u0 = vhadd_u16(res0, d0);
|
||||
|
||||
tmp0 = vsub_s16(vreinterpret_s16_u16(tmp_u0), sub_const_vec);
|
||||
|
||||
tmp0 = vqrshl_s16(tmp0, round_bits_vec);
|
||||
|
||||
tmp4 = vcombine_s16(tmp0, tmp0);
|
||||
|
||||
*t0 = vqmovun_s16(tmp4);
|
||||
}
|
||||
}
|
||||
|
||||
static INLINE void compute_avg_8x1(uint16x8_t res0, uint16x8_t d0,
|
||||
const uint16_t fwd_offset,
|
||||
const uint16_t bck_offset,
|
||||
const int16x4_t sub_const,
|
||||
const int16_t round_bits,
|
||||
const int use_jnt_comp_avg, uint8x8_t *t0) {
|
||||
int16x4_t tmp0, tmp2;
|
||||
int16x8_t f0;
|
||||
uint32x4_t sum0, sum2;
|
||||
int32x4_t dst0, dst2;
|
||||
|
||||
uint16x8_t tmp_u0;
|
||||
|
||||
if (use_jnt_comp_avg) {
|
||||
const int32x4_t sub_const_vec = vmovl_s16(sub_const);
|
||||
const int32x4_t round_bits_vec = vdupq_n_s32(-(int32_t)round_bits);
|
||||
|
||||
sum0 = vmull_n_u16(vget_low_u16(res0), fwd_offset);
|
||||
sum0 = vmlal_n_u16(sum0, vget_low_u16(d0), bck_offset);
|
||||
sum0 = vshrq_n_u32(sum0, DIST_PRECISION_BITS);
|
||||
|
||||
sum2 = vmull_n_u16(vget_high_u16(res0), fwd_offset);
|
||||
sum2 = vmlal_n_u16(sum2, vget_high_u16(d0), bck_offset);
|
||||
sum2 = vshrq_n_u32(sum2, DIST_PRECISION_BITS);
|
||||
|
||||
dst0 = vsubq_s32(vreinterpretq_s32_u32(sum0), sub_const_vec);
|
||||
dst2 = vsubq_s32(vreinterpretq_s32_u32(sum2), sub_const_vec);
|
||||
|
||||
dst0 = vqrshlq_s32(dst0, round_bits_vec);
|
||||
dst2 = vqrshlq_s32(dst2, round_bits_vec);
|
||||
|
||||
tmp0 = vqmovn_s32(dst0);
|
||||
tmp2 = vqmovn_s32(dst2);
|
||||
|
||||
f0 = vcombine_s16(tmp0, tmp2);
|
||||
|
||||
*t0 = vqmovun_s16(f0);
|
||||
|
||||
} else {
|
||||
const int16x8_t sub_const_vec = vcombine_s16(sub_const, sub_const);
|
||||
const int16x8_t round_bits_vec = vdupq_n_s16(-round_bits);
|
||||
|
||||
tmp_u0 = vhaddq_u16(res0, d0);
|
||||
|
||||
f0 = vsubq_s16(vreinterpretq_s16_u16(tmp_u0), sub_const_vec);
|
||||
|
||||
f0 = vqrshlq_s16(f0, round_bits_vec);
|
||||
|
||||
*t0 = vqmovun_s16(f0);
|
||||
}
|
||||
}
|
||||
#endif // !defined(__arch64__)
|
||||
|
||||
static INLINE void compute_avg_4x4(
|
||||
uint16x4_t res0, uint16x4_t res1, uint16x4_t res2, uint16x4_t res3,
|
||||
uint16x4_t d0, uint16x4_t d1, uint16x4_t d2, uint16x4_t d3,
|
||||
const uint16_t fwd_offset, const uint16_t bck_offset,
|
||||
const int16x4_t sub_const_vec, const int16_t round_bits,
|
||||
const int32_t use_jnt_comp_avg, uint8x8_t *t0, uint8x8_t *t1) {
|
||||
const int use_jnt_comp_avg, uint8x8_t *t0, uint8x8_t *t1) {
|
||||
int16x4_t tmp0, tmp1, tmp2, tmp3;
|
||||
uint16x4_t tmp_u0, tmp_u1, tmp_u2, tmp_u3;
|
||||
uint32x4_t sum0, sum1, sum2, sum3;
|
||||
|
|
@ -107,7 +203,7 @@ static INLINE void compute_avg_8x4(
|
|||
uint16x8_t d0, uint16x8_t d1, uint16x8_t d2, uint16x8_t d3,
|
||||
const uint16_t fwd_offset, const uint16_t bck_offset,
|
||||
const int16x4_t sub_const, const int16_t round_bits,
|
||||
const int32_t use_jnt_comp_avg, uint8x8_t *t0, uint8x8_t *t1, uint8x8_t *t2,
|
||||
const int use_jnt_comp_avg, uint8x8_t *t0, uint8x8_t *t1, uint8x8_t *t2,
|
||||
uint8x8_t *t3) {
|
||||
int16x4_t tmp0, tmp1, tmp2, tmp3, tmp4, tmp5, tmp6, tmp7;
|
||||
int16x8_t f0, f1, f2, f3;
|
||||
|
|
@ -231,7 +327,6 @@ static INLINE void jnt_convolve_2d_horiz_neon(
|
|||
int16_t *dst_ptr;
|
||||
int dst_stride;
|
||||
int width, height;
|
||||
uint8x8_t t0, t1, t2, t3, t4, t5, t6, t7;
|
||||
|
||||
dst_ptr = im_block;
|
||||
dst_stride = im_stride;
|
||||
|
|
@ -239,15 +334,22 @@ static INLINE void jnt_convolve_2d_horiz_neon(
|
|||
width = w;
|
||||
|
||||
if (w == 4) {
|
||||
int16x4_t s0, s1, s2, s3, s4, s5, s6, s7, s8, s9, s10, d0, d1, d2, d3;
|
||||
int16x8_t tt0, tt1, tt2, tt3;
|
||||
int16x4_t s0, s1, s2, s3, s4, s5, s6, s7, d0;
|
||||
int16x8_t tt0;
|
||||
uint8x8_t t0;
|
||||
|
||||
const int16x4_t horiz_const = vdup_n_s16((1 << (bd + FILTER_BITS - 2)));
|
||||
const int16x4_t shift_round_0 = vdup_n_s16(-(round_0));
|
||||
|
||||
#if defined(__aarch64__)
|
||||
int16x4_t s8, s9, s10, d1, d2, d3;
|
||||
int16x8_t tt1, tt2, tt3;
|
||||
uint8x8_t t1, t2, t3;
|
||||
#endif
|
||||
do {
|
||||
s = src;
|
||||
__builtin_prefetch(s + 0 * src_stride);
|
||||
#if defined(__aarch64__)
|
||||
__builtin_prefetch(s + 1 * src_stride);
|
||||
__builtin_prefetch(s + 2 * src_stride);
|
||||
__builtin_prefetch(s + 3 * src_stride);
|
||||
|
|
@ -301,17 +403,48 @@ static INLINE void jnt_convolve_2d_horiz_neon(
|
|||
src += 4 * src_stride;
|
||||
dst_ptr += 4 * dst_stride;
|
||||
height -= 4;
|
||||
#else
|
||||
t0 = vld1_u8(s); // a0 a1 a2 a3 a4 a5 a6 a7
|
||||
tt0 = vreinterpretq_s16_u16(vmovl_u8(t0)); // a0 a1 a2 a3 a4 a5 a6 a7
|
||||
s0 = vget_low_s16(tt0); // a0 a1 a2 a3
|
||||
s4 = vget_high_s16(tt0); // a4 a5 a6 a7
|
||||
__builtin_prefetch(dst_ptr);
|
||||
s += 8;
|
||||
t0 = vld1_u8(s); // a8 a9 a10 a11
|
||||
|
||||
// a8 a9 a10 a11
|
||||
s7 = vget_low_s16(vreinterpretq_s16_u16(vmovl_u8(t0)));
|
||||
|
||||
s1 = vext_s16(s0, s4, 1); // a1 a2 a3 a4
|
||||
s2 = vext_s16(s0, s4, 2); // a2 a3 a4 a5
|
||||
s3 = vext_s16(s0, s4, 3); // a3 a4 a5 a6
|
||||
s5 = vext_s16(s4, s7, 1); // a5 a6 a7 a8
|
||||
s6 = vext_s16(s4, s7, 2); // a6 a7 a8 a9
|
||||
s7 = vext_s16(s4, s7, 3); // a7 a8 a9 a10
|
||||
|
||||
d0 = convolve8_4x4_s16(s0, s1, s2, s3, s4, s5, s6, s7, x_filter_tmp,
|
||||
horiz_const, shift_round_0);
|
||||
|
||||
vst1_s16(dst_ptr, d0);
|
||||
|
||||
src += src_stride;
|
||||
dst_ptr += dst_stride;
|
||||
height -= 1;
|
||||
#endif
|
||||
} while (height > 0);
|
||||
} else {
|
||||
int16_t *d_tmp;
|
||||
int16x8_t s11, s12, s13, s14;
|
||||
int16x8_t s0, s1, s2, s3, s4, s5, s6, s7, s8, s9, s10;
|
||||
int16x8_t res0, res1, res2, res3, res4, res5, res6, res7;
|
||||
int16x8_t s0, s1, s2, s3, s4, s5, s6, s7;
|
||||
int16x8_t res0;
|
||||
uint8x8_t t0;
|
||||
|
||||
const int16x8_t horiz_const = vdupq_n_s16((1 << (bd + FILTER_BITS - 2)));
|
||||
const int16x8_t shift_round_0 = vdupq_n_s16(-(round_0));
|
||||
|
||||
do {
|
||||
#if defined(__aarch64__)
|
||||
uint8x8_t t1, t2, t3, t4, t5, t6, t7;
|
||||
int16x8_t s8, s9, s10, s11, s12, s13, s14;
|
||||
int16x8_t res1, res2, res3, res4, res5, res6, res7;
|
||||
__builtin_prefetch(src + 0 * src_stride);
|
||||
__builtin_prefetch(src + 1 * src_stride);
|
||||
__builtin_prefetch(src + 2 * src_stride);
|
||||
|
|
@ -390,6 +523,42 @@ static INLINE void jnt_convolve_2d_horiz_neon(
|
|||
src += 8 * src_stride;
|
||||
dst_ptr += 8 * dst_stride;
|
||||
height -= 8;
|
||||
#else
|
||||
int16x8_t temp_0;
|
||||
t0 = vld1_u8(src);
|
||||
s0 = vreinterpretq_s16_u16(vmovl_u8(t0)); // a0 a1 a2 a3 a4 a5 a6 a7
|
||||
|
||||
width = w;
|
||||
s = src + 8;
|
||||
d_tmp = dst_ptr;
|
||||
__builtin_prefetch(dst_ptr);
|
||||
|
||||
do {
|
||||
t0 = vld1_u8(s); // a8 a9 a10 a11 a12 a13 a14 a15
|
||||
s7 = vreinterpretq_s16_u16(vmovl_u8(t0));
|
||||
temp_0 = s0;
|
||||
s0 = s7;
|
||||
|
||||
s1 = vextq_s16(temp_0, s7, 1); // a1 a2 a3 a4 a5 a6 a7 a8
|
||||
s2 = vextq_s16(temp_0, s7, 2); // a2 a3 a4 a5 a6 a7 a8 a9
|
||||
s3 = vextq_s16(temp_0, s7, 3); // a3 a4 a5 a6 a7 a8 a9 a10
|
||||
s4 = vextq_s16(temp_0, s7, 4); // a4 a5 a6 a7 a8 a9 a10 a11
|
||||
s5 = vextq_s16(temp_0, s7, 5); // a5 a6 a7 a8 a9 a10 a11 a12
|
||||
s6 = vextq_s16(temp_0, s7, 6); // a6 a7 a8 a9 a10 a11 a12 a13
|
||||
s7 = vextq_s16(temp_0, s7, 7); // a7 a8 a9 a10 a11 a12 a13 a14
|
||||
|
||||
res0 = convolve8_8x8_s16(temp_0, s1, s2, s3, s4, s5, s6, s7,
|
||||
x_filter_tmp, horiz_const, shift_round_0);
|
||||
vst1q_s16(d_tmp, res0);
|
||||
|
||||
s += 8;
|
||||
d_tmp += 8;
|
||||
width -= 8;
|
||||
} while (width > 0);
|
||||
src += src_stride;
|
||||
dst_ptr += dst_stride;
|
||||
height -= 1;
|
||||
#endif
|
||||
} while (height > 0);
|
||||
}
|
||||
}
|
||||
|
|
@ -420,10 +589,15 @@ static INLINE void jnt_convolve_2d_vert_neon(
|
|||
const int do_average = conv_params->do_average;
|
||||
const int use_jnt_comp_avg = conv_params->use_jnt_comp_avg;
|
||||
|
||||
int16x4_t s0, s1, s2, s3, s4, s5, s6, s7, s8, s9, s10;
|
||||
uint16x4_t res4, res5, res6, res7;
|
||||
uint16x4_t d0, d1, d2, d3;
|
||||
uint8x8_t t0, t1;
|
||||
int16x4_t s0, s1, s2, s3, s4, s5, s6, s7;
|
||||
uint16x4_t res4, d0;
|
||||
uint8x8_t t0;
|
||||
|
||||
#if defined(__aarch64__)
|
||||
int16x4_t s8, s9, s10;
|
||||
uint16x4_t res5, res6, res7, d1, d2, d3;
|
||||
uint8x8_t t1;
|
||||
#endif
|
||||
|
||||
dst = conv_params->dst;
|
||||
src_ptr = im_block;
|
||||
|
|
@ -450,6 +624,7 @@ static INLINE void jnt_convolve_2d_vert_neon(
|
|||
s += (7 * im_stride);
|
||||
|
||||
do {
|
||||
#if defined(__aarch64__)
|
||||
load_s16_4x4(s, im_stride, &s7, &s8, &s9, &s10);
|
||||
s += (im_stride << 2);
|
||||
|
||||
|
|
@ -480,17 +655,13 @@ static INLINE void jnt_convolve_2d_vert_neon(
|
|||
bck_offset, sub_const_vec, round_bits, use_jnt_comp_avg,
|
||||
&t0, &t1);
|
||||
|
||||
vst1_lane_u32((uint32_t *)d_u8, vreinterpret_u32_u8(t0),
|
||||
0); // 00 01 02 03
|
||||
vst1_lane_u32((uint32_t *)d_u8, vreinterpret_u32_u8(t0), 0);
|
||||
d_u8 += dst8_stride;
|
||||
vst1_lane_u32((uint32_t *)d_u8, vreinterpret_u32_u8(t0),
|
||||
1); // 10 11 12 13
|
||||
vst1_lane_u32((uint32_t *)d_u8, vreinterpret_u32_u8(t0), 1);
|
||||
d_u8 += dst8_stride;
|
||||
vst1_lane_u32((uint32_t *)d_u8, vreinterpret_u32_u8(t1),
|
||||
0); // 20 21 22 23
|
||||
vst1_lane_u32((uint32_t *)d_u8, vreinterpret_u32_u8(t1), 0);
|
||||
d_u8 += dst8_stride;
|
||||
vst1_lane_u32((uint32_t *)d_u8, vreinterpret_u32_u8(t1),
|
||||
1); // 30 31 32 33
|
||||
vst1_lane_u32((uint32_t *)d_u8, vreinterpret_u32_u8(t1), 1);
|
||||
d_u8 += dst8_stride;
|
||||
|
||||
} else {
|
||||
|
|
@ -505,6 +676,39 @@ static INLINE void jnt_convolve_2d_vert_neon(
|
|||
s5 = s9;
|
||||
s6 = s10;
|
||||
height -= 4;
|
||||
#else
|
||||
s7 = vld1_s16(s);
|
||||
s += (im_stride);
|
||||
|
||||
__builtin_prefetch(d + 0 * dst_stride);
|
||||
__builtin_prefetch(d_u8 + 0 * dst8_stride);
|
||||
|
||||
d0 = convolve8_4x4_s32(s0, s1, s2, s3, s4, s5, s6, s7, y_filter,
|
||||
round_shift_vec, offset_const);
|
||||
|
||||
if (do_average) {
|
||||
res4 = vld1_u16(d);
|
||||
d += (dst_stride);
|
||||
|
||||
compute_avg_4x1(res4, d0, fwd_offset, bck_offset, sub_const_vec,
|
||||
round_bits, use_jnt_comp_avg, &t0);
|
||||
|
||||
vst1_lane_u32((uint32_t *)d_u8, vreinterpret_u32_u8(t0), 0);
|
||||
d_u8 += dst8_stride;
|
||||
|
||||
} else {
|
||||
vst1_u16(d, d0);
|
||||
d += (dst_stride);
|
||||
}
|
||||
s0 = s1;
|
||||
s1 = s2;
|
||||
s2 = s3;
|
||||
s3 = s4;
|
||||
s4 = s5;
|
||||
s5 = s6;
|
||||
s6 = s7;
|
||||
height--;
|
||||
#endif
|
||||
} while (height > 0);
|
||||
src_ptr += 4;
|
||||
dst_ptr += 4;
|
||||
|
|
@ -722,8 +926,10 @@ void av1_jnt_convolve_x_neon(const uint8_t *src, int src_stride, uint8_t *dst8,
|
|||
uint8_t *dst_u8_ptr;
|
||||
CONV_BUF_TYPE *d, *dst_ptr;
|
||||
int width, height;
|
||||
uint8x8_t t0, t1, t2, t3, t4, t5, t6, t7;
|
||||
|
||||
uint8x8_t t0;
|
||||
#if defined(__aarch64__)
|
||||
uint8x8_t t1, t2, t3, t4, t5, t6, t7;
|
||||
#endif
|
||||
s = src_ptr;
|
||||
dst_ptr = dst;
|
||||
dst_u8_ptr = dst8;
|
||||
|
|
@ -731,11 +937,18 @@ void av1_jnt_convolve_x_neon(const uint8_t *src, int src_stride, uint8_t *dst8,
|
|||
height = h;
|
||||
|
||||
if ((w == 4) || (h == 4)) {
|
||||
int16x4_t s0, s1, s2, s3, s4, s5, s6, s7, s8, s9, s10, d0, d1, d2, d3;
|
||||
int16x8_t tt0, tt1, tt2, tt3;
|
||||
uint16x4_t res4, res5, res6, res7;
|
||||
uint32x2_t tu0, tu1;
|
||||
int16x4_t s0, s1, s2, s3, s4, s5, s6, s7, d0;
|
||||
int16x8_t tt0;
|
||||
uint16x4_t res4;
|
||||
#if defined(__aarch64__)
|
||||
int16x4_t s8, s9, s10, d1, d2, d3;
|
||||
int16x8_t tt1, tt2, tt3;
|
||||
uint16x4_t res5, res6, res7;
|
||||
uint32x2_t tu0 = vdup_n_u32(0), tu1 = vdup_n_u32(0);
|
||||
int16x8_t u0, u1;
|
||||
#else
|
||||
int16x4_t temp_0;
|
||||
#endif
|
||||
const int16x4_t zero = vdup_n_s16(0);
|
||||
const int16x4_t round_offset_vec = vdup_n_s16(round_offset);
|
||||
const int16x4_t shift_round_0 = vdup_n_s16(-conv_params->round_0 + 1);
|
||||
|
|
@ -746,6 +959,7 @@ void av1_jnt_convolve_x_neon(const uint8_t *src, int src_stride, uint8_t *dst8,
|
|||
d_u8 = dst_u8_ptr;
|
||||
width = w;
|
||||
__builtin_prefetch(s + 0 * src_stride);
|
||||
#if defined(__aarch64__)
|
||||
__builtin_prefetch(s + 1 * src_stride);
|
||||
__builtin_prefetch(s + 2 * src_stride);
|
||||
__builtin_prefetch(s + 3 * src_stride);
|
||||
|
|
@ -854,15 +1068,66 @@ void av1_jnt_convolve_x_neon(const uint8_t *src, int src_stride, uint8_t *dst8,
|
|||
dst_ptr += (dst_stride << 2);
|
||||
dst_u8_ptr += (dst8_stride << 2);
|
||||
height -= 4;
|
||||
#else
|
||||
t0 = vld1_u8(s); // a0 a1 a2 a3 a4 a5 a6 a7
|
||||
tt0 = vreinterpretq_s16_u16(vmovl_u8(t0)); // a0 a1 a2 a3 a4 a5 a6 a7
|
||||
s0 = vget_low_s16(tt0); // a0 a1 a2 a3
|
||||
s4 = vget_high_s16(tt0); // a4 a5 a6 a7
|
||||
__builtin_prefetch(d);
|
||||
|
||||
s += 8;
|
||||
do {
|
||||
t0 = vld1_u8(s); // a8 a9 a10 a11
|
||||
|
||||
// a8 a9 a10 a11
|
||||
s7 = vget_low_s16(vreinterpretq_s16_u16(vmovl_u8(t0)));
|
||||
temp_0 = s7;
|
||||
s1 = vext_s16(s0, s4, 1); // a1 a2 a3 a4
|
||||
s2 = vext_s16(s0, s4, 2); // a2 a3 a4 a5
|
||||
s3 = vext_s16(s0, s4, 3); // a3 a4 a5 a6
|
||||
s5 = vext_s16(s4, s7, 1); // a5 a6 a7 a8
|
||||
s6 = vext_s16(s4, s7, 2); // a6 a7 a8 a9
|
||||
s7 = vext_s16(s4, s7, 3); // a7 a8 a9 a10
|
||||
|
||||
d0 = convolve8_4x4_s16(s0, s1, s2, s3, s4, s5, s6, s7, x_filter_tmp,
|
||||
zero, shift_round_0);
|
||||
d0 = vrshl_s16(d0, horiz_const);
|
||||
d0 = vadd_s16(d0, round_offset_vec);
|
||||
s0 = s4;
|
||||
s4 = temp_0;
|
||||
if (conv_params->do_average) {
|
||||
__builtin_prefetch(d);
|
||||
__builtin_prefetch(d_u8);
|
||||
|
||||
res4 = vld1_u16(d);
|
||||
|
||||
compute_avg_4x1(res4, vreinterpret_u16_s16(d0), fwd_offset,
|
||||
bck_offset, round_offset_vec, round_bits,
|
||||
use_jnt_comp_avg, &t0);
|
||||
|
||||
vst1_lane_u32((uint32_t *)d_u8, vreinterpret_u32_u8(t0),
|
||||
0); // 00 01 02 03
|
||||
} else {
|
||||
vst1_u16(d, vreinterpret_u16_s16(d0));
|
||||
}
|
||||
|
||||
s += 4;
|
||||
width -= 4;
|
||||
d += 4;
|
||||
d_u8 += 4;
|
||||
} while (width > 0);
|
||||
src_ptr += (src_stride);
|
||||
dst_ptr += (dst_stride);
|
||||
dst_u8_ptr += (dst8_stride);
|
||||
height--;
|
||||
#endif
|
||||
} while (height > 0);
|
||||
} else {
|
||||
CONV_BUF_TYPE *d_tmp;
|
||||
uint8_t *d_u8_tmp;
|
||||
int16x8_t s11, s12, s13, s14;
|
||||
int16x8_t s0, s1, s2, s3, s4, s5, s6, s7, s8, s9, s10;
|
||||
int16x8_t res0, res1, res2, res3, res4, res5, res6, res7;
|
||||
uint16x8_t res8, res9, res10, res11;
|
||||
|
||||
int16x8_t s0, s1, s2, s3, s4, s5, s6, s7;
|
||||
int16x8_t res0;
|
||||
uint16x8_t res8;
|
||||
const int16x8_t round_offset128 = vdupq_n_s16(round_offset);
|
||||
const int16x4_t round_offset64 = vdup_n_s16(round_offset);
|
||||
const int16x8_t shift_round_0 = vdupq_n_s16(-conv_params->round_0 + 1);
|
||||
|
|
@ -872,6 +1137,11 @@ void av1_jnt_convolve_x_neon(const uint8_t *src, int src_stride, uint8_t *dst8,
|
|||
d = dst_ptr = dst;
|
||||
d_u8 = dst_u8_ptr = dst8;
|
||||
do {
|
||||
#if defined(__aarch64__)
|
||||
int16x8_t s11, s12, s13, s14;
|
||||
int16x8_t s8, s9, s10;
|
||||
int16x8_t res1, res2, res3, res4, res5, res6, res7;
|
||||
uint16x8_t res9, res10, res11;
|
||||
__builtin_prefetch(src_ptr + 0 * src_stride);
|
||||
__builtin_prefetch(src_ptr + 1 * src_stride);
|
||||
__builtin_prefetch(src_ptr + 2 * src_stride);
|
||||
|
|
@ -1007,6 +1277,67 @@ void av1_jnt_convolve_x_neon(const uint8_t *src, int src_stride, uint8_t *dst8,
|
|||
dst_ptr += 8 * dst_stride;
|
||||
dst_u8_ptr += 8 * dst8_stride;
|
||||
height -= 8;
|
||||
#else
|
||||
int16x8_t temp_0;
|
||||
__builtin_prefetch(src_ptr);
|
||||
t0 = vld1_u8(src_ptr);
|
||||
s0 = vreinterpretq_s16_u16(vmovl_u8(t0)); // a0 a1 a2 a3 a4 a5 a6 a7
|
||||
|
||||
width = w;
|
||||
s = src_ptr + 8;
|
||||
d = dst_ptr;
|
||||
d_u8_tmp = dst_u8_ptr;
|
||||
|
||||
__builtin_prefetch(dst_ptr);
|
||||
|
||||
do {
|
||||
d_u8 = d_u8_tmp;
|
||||
d_tmp = d;
|
||||
|
||||
t0 = vld1_u8(s); // a8 a9 a10 a11 a12 a13 a14 a15
|
||||
s7 = vreinterpretq_s16_u16(vmovl_u8(t0));
|
||||
temp_0 = s0;
|
||||
s0 = s7;
|
||||
|
||||
s1 = vextq_s16(temp_0, s7, 1); // a1 a2 a3 a4 a5 a6 a7 a8
|
||||
s2 = vextq_s16(temp_0, s7, 2); // a2 a3 a4 a5 a6 a7 a8 a9
|
||||
s3 = vextq_s16(temp_0, s7, 3); // a3 a4 a5 a6 a7 a8 a9 a10
|
||||
s4 = vextq_s16(temp_0, s7, 4); // a4 a5 a6 a7 a8 a9 a10 a11
|
||||
s5 = vextq_s16(temp_0, s7, 5); // a5 a6 a7 a8 a9 a10 a11 a12
|
||||
s6 = vextq_s16(temp_0, s7, 6); // a6 a7 a8 a9 a10 a11 a12 a13
|
||||
s7 = vextq_s16(temp_0, s7, 7); // a7 a8 a9 a10 a11 a12 a13 a14
|
||||
|
||||
res0 = convolve8_8x8_s16(temp_0, s1, s2, s3, s4, s5, s6, s7,
|
||||
x_filter_tmp, zero, shift_round_0);
|
||||
|
||||
res0 = vrshlq_s16(res0, horiz_const);
|
||||
res0 = vaddq_s16(res0, round_offset128);
|
||||
|
||||
if (conv_params->do_average) {
|
||||
res8 = vld1q_u16(d_tmp);
|
||||
d_tmp += (dst_stride);
|
||||
|
||||
compute_avg_8x1(res8, vreinterpretq_u16_s16(res0), fwd_offset,
|
||||
bck_offset, round_offset64, round_bits,
|
||||
use_jnt_comp_avg, &t0);
|
||||
|
||||
vst1_u8(d_u8, t0);
|
||||
d_u8 += (dst8_stride);
|
||||
} else {
|
||||
vst1q_u16(d_tmp, vreinterpretq_u16_s16(res0));
|
||||
d_tmp += (dst_stride);
|
||||
}
|
||||
|
||||
s += 8;
|
||||
d += 8;
|
||||
width -= 8;
|
||||
d_u8_tmp += 8;
|
||||
} while (width > 0);
|
||||
src_ptr += src_stride;
|
||||
dst_ptr += dst_stride;
|
||||
dst_u8_ptr += dst8_stride;
|
||||
height--;
|
||||
#endif
|
||||
} while (height > 0);
|
||||
}
|
||||
}
|
||||
|
|
@ -1057,7 +1388,6 @@ void av1_jnt_convolve_y_neon(const uint8_t *src, int src_stride, uint8_t *dst8,
|
|||
uint8_t *dst_u8_ptr;
|
||||
CONV_BUF_TYPE *d, *dst_ptr;
|
||||
int width, height;
|
||||
uint8x8_t t0, t1, t2, t3, t4, t5, t6, t7;
|
||||
|
||||
s = src_ptr;
|
||||
dst_ptr = dst;
|
||||
|
|
@ -1070,11 +1400,18 @@ void av1_jnt_convolve_y_neon(const uint8_t *src, int src_stride, uint8_t *dst8,
|
|||
assert((conv_params->round_1 - 2) >= bits);
|
||||
|
||||
if ((w == 4) || (h == 4)) {
|
||||
int16x4_t s0, s1, s2, s3, s4, s5, s6, s7, s8, s9, s10, d0, d1, d2, d3;
|
||||
uint16x4_t res4, res5, res6, res7;
|
||||
uint32x2_t tu0, tu1, tu2, tu3;
|
||||
int16x4_t s0, s1, s2, s3, s4, s5, s6, s7, d0;
|
||||
uint16x4_t res4;
|
||||
uint32x2_t tu0 = vdup_n_u32(0), tu1 = vdup_n_u32(0), tu2 = vdup_n_u32(0),
|
||||
tu3 = vdup_n_u32(0);
|
||||
int16x8_t u0, u1, u2, u3;
|
||||
uint8x8_t t0;
|
||||
|
||||
#if defined(__aarch64__)
|
||||
int16x4_t s8, s9, s10, d1, d2, d3;
|
||||
uint16x4_t res5, res6, res7;
|
||||
uint8x8_t t1;
|
||||
#endif
|
||||
const int16x4_t round_offset64 = vdup_n_s16(round_offset);
|
||||
const int16x4_t shift_vec = vdup_n_s16(-shift_value);
|
||||
const int16x4_t zero = vdup_n_s16(0);
|
||||
|
|
@ -1111,6 +1448,7 @@ void av1_jnt_convolve_y_neon(const uint8_t *src, int src_stride, uint8_t *dst8,
|
|||
|
||||
s += (7 * src_stride);
|
||||
do {
|
||||
#if defined(__aarch64__)
|
||||
load_unaligned_u8_4x4(s, src_stride, &tu0, &tu1);
|
||||
|
||||
u0 = vreinterpretq_s16_u16(vmovl_u8(vreinterpret_u8_u32(tu0)));
|
||||
|
|
@ -1154,17 +1492,13 @@ void av1_jnt_convolve_y_neon(const uint8_t *src, int src_stride, uint8_t *dst8,
|
|||
round_offset64, round_bits, use_jnt_comp_avg, &t0,
|
||||
&t1);
|
||||
|
||||
vst1_lane_u32((uint32_t *)d_u8, vreinterpret_u32_u8(t0),
|
||||
0); // 00 01 02 03
|
||||
vst1_lane_u32((uint32_t *)d_u8, vreinterpret_u32_u8(t0), 0);
|
||||
d_u8 += dst8_stride;
|
||||
vst1_lane_u32((uint32_t *)d_u8, vreinterpret_u32_u8(t0),
|
||||
1); // 10 11 12 13
|
||||
vst1_lane_u32((uint32_t *)d_u8, vreinterpret_u32_u8(t0), 1);
|
||||
d_u8 += dst8_stride;
|
||||
vst1_lane_u32((uint32_t *)d_u8, vreinterpret_u32_u8(t1),
|
||||
0); // 20 21 22 23
|
||||
vst1_lane_u32((uint32_t *)d_u8, vreinterpret_u32_u8(t1), 0);
|
||||
d_u8 += dst8_stride;
|
||||
vst1_lane_u32((uint32_t *)d_u8, vreinterpret_u32_u8(t1),
|
||||
1); // 30 31 32 33
|
||||
vst1_lane_u32((uint32_t *)d_u8, vreinterpret_u32_u8(t1), 1);
|
||||
d_u8 += dst8_stride;
|
||||
} else {
|
||||
store_u16_4x4(d, dst_stride, vreinterpret_u16_s16(d0),
|
||||
|
|
@ -1183,6 +1517,44 @@ void av1_jnt_convolve_y_neon(const uint8_t *src, int src_stride, uint8_t *dst8,
|
|||
|
||||
s += (src_stride << 2);
|
||||
height -= 4;
|
||||
#else
|
||||
load_unaligned_u8_4x1(s, src_stride, &tu0);
|
||||
u0 = vreinterpretq_s16_u16(vmovl_u8(vreinterpret_u8_u32(tu0)));
|
||||
s7 = vget_low_s16(u0);
|
||||
|
||||
d0 = convolve8_4x4_s16(s0, s1, s2, s3, s4, s5, s6, s7, y_filter_tmp,
|
||||
zero, shift_vec);
|
||||
|
||||
d0 = vadd_s16(d0, round_offset64);
|
||||
|
||||
if (conv_params->do_average) {
|
||||
__builtin_prefetch(d);
|
||||
|
||||
res4 = vld1_u16(d);
|
||||
d += (dst_stride);
|
||||
|
||||
compute_avg_4x1(res4, vreinterpret_u16_s16(d0), fwd_offset,
|
||||
bck_offset, round_offset64, round_bits,
|
||||
use_jnt_comp_avg, &t0);
|
||||
|
||||
vst1_lane_u32((uint32_t *)d_u8, vreinterpret_u32_u8(t0), 0);
|
||||
d_u8 += dst8_stride;
|
||||
} else {
|
||||
vst1_u16(d, vreinterpret_u16_s16(d0));
|
||||
d += (dst_stride);
|
||||
}
|
||||
|
||||
s0 = s1;
|
||||
s1 = s2;
|
||||
s2 = s3;
|
||||
s3 = s4;
|
||||
s4 = s5;
|
||||
s5 = s6;
|
||||
s6 = s7;
|
||||
|
||||
s += (src_stride);
|
||||
height--;
|
||||
#endif
|
||||
} while (height > 0);
|
||||
src_ptr += 4;
|
||||
dst_ptr += 4;
|
||||
|
|
@ -1191,15 +1563,19 @@ void av1_jnt_convolve_y_neon(const uint8_t *src, int src_stride, uint8_t *dst8,
|
|||
} while (width > 0);
|
||||
} else {
|
||||
CONV_BUF_TYPE *d_tmp;
|
||||
int16x8_t s11, s12, s13, s14;
|
||||
int16x8_t s0, s1, s2, s3, s4, s5, s6, s7, s8, s9, s10;
|
||||
int16x8_t res0, res1, res2, res3, res4, res5, res6, res7;
|
||||
uint16x8_t res8, res9, res10, res11;
|
||||
int16x8_t s0, s1, s2, s3, s4, s5, s6, s7;
|
||||
int16x8_t res0;
|
||||
uint16x8_t res8;
|
||||
uint8x8_t t0, t1, t2, t3, t4, t5, t6, t7;
|
||||
const int16x8_t round_offset128 = vdupq_n_s16(round_offset);
|
||||
const int16x8_t shift_vec = vdupq_n_s16(-shift_value);
|
||||
const int16x4_t round_offset64 = vdup_n_s16(round_offset);
|
||||
const int16x8_t zero = vdupq_n_s16(0);
|
||||
|
||||
#if defined(__aarch64__)
|
||||
int16x8_t s8, s9, s10, s11, s12, s13, s14;
|
||||
int16x8_t res1, res2, res3, res4, res5, res6, res7;
|
||||
uint16x8_t res10, res11, res9;
|
||||
#endif
|
||||
dst_ptr = dst;
|
||||
dst_u8_ptr = dst8;
|
||||
do {
|
||||
|
|
@ -1227,6 +1603,7 @@ void av1_jnt_convolve_y_neon(const uint8_t *src, int src_stride, uint8_t *dst8,
|
|||
d_u8 = dst_u8_ptr;
|
||||
|
||||
do {
|
||||
#if defined(__aarch64__)
|
||||
load_u8_8x8(s, src_stride, &t0, &t1, &t2, &t3, &t4, &t5, &t6, &t7);
|
||||
|
||||
s7 = vreinterpretq_s16_u16(vmovl_u8(t0));
|
||||
|
|
@ -1316,6 +1693,43 @@ void av1_jnt_convolve_y_neon(const uint8_t *src, int src_stride, uint8_t *dst8,
|
|||
s6 = s14;
|
||||
s += (8 * src_stride);
|
||||
height -= 8;
|
||||
#else
|
||||
s7 = vreinterpretq_s16_u16(vmovl_u8(vld1_u8(s)));
|
||||
|
||||
__builtin_prefetch(dst_ptr);
|
||||
|
||||
res0 = convolve8_8x8_s16(s0, s1, s2, s3, s4, s5, s6, s7, y_filter_tmp,
|
||||
zero, shift_vec);
|
||||
res0 = vaddq_s16(res0, round_offset128);
|
||||
|
||||
s0 = s1;
|
||||
s1 = s2;
|
||||
s2 = s3;
|
||||
s3 = s4;
|
||||
s4 = s5;
|
||||
s5 = s6;
|
||||
s6 = s7;
|
||||
|
||||
if (conv_params->do_average) {
|
||||
__builtin_prefetch(d_tmp);
|
||||
|
||||
res8 = vld1q_u16(d_tmp);
|
||||
d_tmp += (dst_stride);
|
||||
|
||||
compute_avg_8x1(res8, vreinterpretq_u16_s16(res0), fwd_offset,
|
||||
bck_offset, round_offset64, round_bits,
|
||||
use_jnt_comp_avg, &t0);
|
||||
|
||||
vst1_u8(d_u8, t0);
|
||||
d_u8 += (dst8_stride);
|
||||
} else {
|
||||
vst1q_u16(d_tmp, vreinterpretq_u16_s16(res0));
|
||||
d_tmp += dst_stride;
|
||||
}
|
||||
|
||||
s += (src_stride);
|
||||
height--;
|
||||
#endif
|
||||
} while (height > 0);
|
||||
src_ptr += 8;
|
||||
dst_ptr += 8;
|
||||
|
|
|
|||
15
third_party/aom/av1/common/arm/mem_neon.h
vendored
15
third_party/aom/av1/common/arm/mem_neon.h
vendored
|
|
@ -8,8 +8,8 @@
|
|||
* be found in the AUTHORS file in the root of the source tree.
|
||||
*/
|
||||
|
||||
#ifndef AV1_COMMON_ARM_MEM_NEON_H_
|
||||
#define AV1_COMMON_ARM_MEM_NEON_H_
|
||||
#ifndef AOM_AV1_COMMON_ARM_MEM_NEON_H_
|
||||
#define AOM_AV1_COMMON_ARM_MEM_NEON_H_
|
||||
|
||||
#include <arm_neon.h>
|
||||
#include <string.h>
|
||||
|
|
@ -362,6 +362,15 @@ static INLINE void load_unaligned_u8_4x4(const uint8_t *buf, int stride,
|
|||
*tu1 = vset_lane_u32(a, *tu1, 1);
|
||||
}
|
||||
|
||||
static INLINE void load_unaligned_u8_4x1(const uint8_t *buf, int stride,
|
||||
uint32x2_t *tu0) {
|
||||
uint32_t a;
|
||||
|
||||
memcpy(&a, buf, 4);
|
||||
buf += stride;
|
||||
*tu0 = vset_lane_u32(a, *tu0, 0);
|
||||
}
|
||||
|
||||
static INLINE void load_unaligned_u8_4x2(const uint8_t *buf, int stride,
|
||||
uint32x2_t *tu0) {
|
||||
uint32_t a;
|
||||
|
|
@ -482,4 +491,4 @@ static INLINE void store_u32_4x4(uint32_t *s, int32_t p, uint32x4_t s1,
|
|||
vst1q_u32(s, s4);
|
||||
}
|
||||
|
||||
#endif // AV1_COMMON_ARM_MEM_NEON_H_
|
||||
#endif // AOM_AV1_COMMON_ARM_MEM_NEON_H_
|
||||
|
|
|
|||
18
third_party/aom/av1/common/arm/selfguided_neon.c
vendored
18
third_party/aom/av1/common/arm/selfguided_neon.c
vendored
|
|
@ -1007,10 +1007,11 @@ static INLINE void cross_sum_fast_odd_row_inp16(uint16_t *buf, int32x4_t *a0,
|
|||
vaddq_u32(vmovl_u16(vget_high_u16(xl)), vmovl_u16(vget_high_u16(x))));
|
||||
}
|
||||
|
||||
void final_filter_fast_internal(uint16_t *A, int32_t *B, const int buf_stride,
|
||||
int16_t *src, const int src_stride,
|
||||
int32_t *dst, const int dst_stride,
|
||||
const int width, const int height) {
|
||||
static void final_filter_fast_internal(uint16_t *A, int32_t *B,
|
||||
const int buf_stride, int16_t *src,
|
||||
const int src_stride, int32_t *dst,
|
||||
const int dst_stride, const int width,
|
||||
const int height) {
|
||||
int16x8_t s0;
|
||||
int32_t *B_tmp, *dst_ptr;
|
||||
uint16_t *A_tmp;
|
||||
|
|
@ -1340,10 +1341,10 @@ static INLINE void src_convert_hbd_copy(const uint16_t *src, int src_stride,
|
|||
}
|
||||
}
|
||||
|
||||
void av1_selfguided_restoration_neon(const uint8_t *dat8, int width, int height,
|
||||
int stride, int32_t *flt0, int32_t *flt1,
|
||||
int flt_stride, int sgr_params_idx,
|
||||
int bit_depth, int highbd) {
|
||||
int av1_selfguided_restoration_neon(const uint8_t *dat8, int width, int height,
|
||||
int stride, int32_t *flt0, int32_t *flt1,
|
||||
int flt_stride, int sgr_params_idx,
|
||||
int bit_depth, int highbd) {
|
||||
const sgr_params_type *const params = &sgr_params[sgr_params_idx];
|
||||
assert(!(params->r[0] == 0 && params->r[1] == 0));
|
||||
|
||||
|
|
@ -1376,6 +1377,7 @@ void av1_selfguided_restoration_neon(const uint8_t *dat8, int width, int height,
|
|||
if (params->r[1] > 0)
|
||||
restoration_internal(dgd16, width, height, dgd16_stride, flt1, flt_stride,
|
||||
bit_depth, sgr_params_idx, 1);
|
||||
return 0;
|
||||
}
|
||||
|
||||
void apply_selfguided_restoration_neon(const uint8_t *dat8, int width,
|
||||
|
|
|
|||
83
third_party/aom/av1/common/arm/transpose_neon.h
vendored
83
third_party/aom/av1/common/arm/transpose_neon.h
vendored
|
|
@ -8,8 +8,8 @@
|
|||
* be found in the AUTHORS file in the root of the source tree.
|
||||
*/
|
||||
|
||||
#ifndef AV1_COMMON_ARM_TRANSPOSE_NEON_H_
|
||||
#define AV1_COMMON_ARM_TRANSPOSE_NEON_H_
|
||||
#ifndef AOM_AV1_COMMON_ARM_TRANSPOSE_NEON_H_
|
||||
#define AOM_AV1_COMMON_ARM_TRANSPOSE_NEON_H_
|
||||
|
||||
#include <arm_neon.h>
|
||||
|
||||
|
|
@ -386,6 +386,83 @@ static INLINE void transpose_s16_8x8(int16x8_t *a0, int16x8_t *a1,
|
|||
vget_high_s16(vreinterpretq_s16_s32(c3.val[1])));
|
||||
}
|
||||
|
||||
static INLINE int16x8x2_t vpx_vtrnq_s64_to_s16(int32x4_t a0, int32x4_t a1) {
|
||||
int16x8x2_t b0;
|
||||
b0.val[0] = vcombine_s16(vreinterpret_s16_s32(vget_low_s32(a0)),
|
||||
vreinterpret_s16_s32(vget_low_s32(a1)));
|
||||
b0.val[1] = vcombine_s16(vreinterpret_s16_s32(vget_high_s32(a0)),
|
||||
vreinterpret_s16_s32(vget_high_s32(a1)));
|
||||
return b0;
|
||||
}
|
||||
|
||||
static INLINE void transpose_s16_8x8q(int16x8_t *a0, int16x8_t *out) {
|
||||
// Swap 16 bit elements. Goes from:
|
||||
// a0: 00 01 02 03 04 05 06 07
|
||||
// a1: 10 11 12 13 14 15 16 17
|
||||
// a2: 20 21 22 23 24 25 26 27
|
||||
// a3: 30 31 32 33 34 35 36 37
|
||||
// a4: 40 41 42 43 44 45 46 47
|
||||
// a5: 50 51 52 53 54 55 56 57
|
||||
// a6: 60 61 62 63 64 65 66 67
|
||||
// a7: 70 71 72 73 74 75 76 77
|
||||
// to:
|
||||
// b0.val[0]: 00 10 02 12 04 14 06 16
|
||||
// b0.val[1]: 01 11 03 13 05 15 07 17
|
||||
// b1.val[0]: 20 30 22 32 24 34 26 36
|
||||
// b1.val[1]: 21 31 23 33 25 35 27 37
|
||||
// b2.val[0]: 40 50 42 52 44 54 46 56
|
||||
// b2.val[1]: 41 51 43 53 45 55 47 57
|
||||
// b3.val[0]: 60 70 62 72 64 74 66 76
|
||||
// b3.val[1]: 61 71 63 73 65 75 67 77
|
||||
|
||||
const int16x8x2_t b0 = vtrnq_s16(*a0, *(a0 + 1));
|
||||
const int16x8x2_t b1 = vtrnq_s16(*(a0 + 2), *(a0 + 3));
|
||||
const int16x8x2_t b2 = vtrnq_s16(*(a0 + 4), *(a0 + 5));
|
||||
const int16x8x2_t b3 = vtrnq_s16(*(a0 + 6), *(a0 + 7));
|
||||
|
||||
// Swap 32 bit elements resulting in:
|
||||
// c0.val[0]: 00 10 20 30 04 14 24 34
|
||||
// c0.val[1]: 02 12 22 32 06 16 26 36
|
||||
// c1.val[0]: 01 11 21 31 05 15 25 35
|
||||
// c1.val[1]: 03 13 23 33 07 17 27 37
|
||||
// c2.val[0]: 40 50 60 70 44 54 64 74
|
||||
// c2.val[1]: 42 52 62 72 46 56 66 76
|
||||
// c3.val[0]: 41 51 61 71 45 55 65 75
|
||||
// c3.val[1]: 43 53 63 73 47 57 67 77
|
||||
|
||||
const int32x4x2_t c0 = vtrnq_s32(vreinterpretq_s32_s16(b0.val[0]),
|
||||
vreinterpretq_s32_s16(b1.val[0]));
|
||||
const int32x4x2_t c1 = vtrnq_s32(vreinterpretq_s32_s16(b0.val[1]),
|
||||
vreinterpretq_s32_s16(b1.val[1]));
|
||||
const int32x4x2_t c2 = vtrnq_s32(vreinterpretq_s32_s16(b2.val[0]),
|
||||
vreinterpretq_s32_s16(b3.val[0]));
|
||||
const int32x4x2_t c3 = vtrnq_s32(vreinterpretq_s32_s16(b2.val[1]),
|
||||
vreinterpretq_s32_s16(b3.val[1]));
|
||||
|
||||
// Swap 64 bit elements resulting in:
|
||||
// d0.val[0]: 00 10 20 30 40 50 60 70
|
||||
// d0.val[1]: 04 14 24 34 44 54 64 74
|
||||
// d1.val[0]: 01 11 21 31 41 51 61 71
|
||||
// d1.val[1]: 05 15 25 35 45 55 65 75
|
||||
// d2.val[0]: 02 12 22 32 42 52 62 72
|
||||
// d2.val[1]: 06 16 26 36 46 56 66 76
|
||||
// d3.val[0]: 03 13 23 33 43 53 63 73
|
||||
// d3.val[1]: 07 17 27 37 47 57 67 77
|
||||
const int16x8x2_t d0 = vpx_vtrnq_s64_to_s16(c0.val[0], c2.val[0]);
|
||||
const int16x8x2_t d1 = vpx_vtrnq_s64_to_s16(c1.val[0], c3.val[0]);
|
||||
const int16x8x2_t d2 = vpx_vtrnq_s64_to_s16(c0.val[1], c2.val[1]);
|
||||
const int16x8x2_t d3 = vpx_vtrnq_s64_to_s16(c1.val[1], c3.val[1]);
|
||||
|
||||
*out = d0.val[0];
|
||||
*(out + 1) = d1.val[0];
|
||||
*(out + 2) = d2.val[0];
|
||||
*(out + 3) = d3.val[0];
|
||||
*(out + 4) = d0.val[1];
|
||||
*(out + 5) = d1.val[1];
|
||||
*(out + 6) = d2.val[1];
|
||||
*(out + 7) = d3.val[1];
|
||||
}
|
||||
|
||||
static INLINE void transpose_s16_4x4d(int16x4_t *a0, int16x4_t *a1,
|
||||
int16x4_t *a2, int16x4_t *a3) {
|
||||
// Swap 16 bit elements. Goes from:
|
||||
|
|
@ -457,4 +534,4 @@ static INLINE void transpose_s32_4x4(int32x4_t *a0, int32x4_t *a1,
|
|||
*a3 = c1.val[1];
|
||||
}
|
||||
|
||||
#endif // AV1_COMMON_ARM_TRANSPOSE_NEON_H_
|
||||
#endif // AOM_AV1_COMMON_ARM_TRANSPOSE_NEON_H_
|
||||
|
|
|
|||
714
third_party/aom/av1/common/arm/warp_plane_neon.c
vendored
Normal file
714
third_party/aom/av1/common/arm/warp_plane_neon.c
vendored
Normal file
|
|
@ -0,0 +1,714 @@
|
|||
/*
|
||||
* Copyright (c) 2018, Alliance for Open Media. All rights reserved
|
||||
*
|
||||
* This source code is subject to the terms of the BSD 2 Clause License and
|
||||
* the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
|
||||
* was not distributed with this source code in the LICENSE file, you can
|
||||
* obtain it at www.aomedia.org/license/software. If the Alliance for Open
|
||||
* Media Patent License 1.0 was not distributed with this source code in the
|
||||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#include <assert.h>
|
||||
#include <arm_neon.h>
|
||||
#include <memory.h>
|
||||
#include <math.h>
|
||||
|
||||
#include "aom_dsp/aom_dsp_common.h"
|
||||
#include "aom_ports/mem.h"
|
||||
#include "config/av1_rtcd.h"
|
||||
#include "av1/common/warped_motion.h"
|
||||
#include "av1/common/scale.h"
|
||||
|
||||
/* This is a modified version of 'warped_filter' from warped_motion.c:
|
||||
* Each coefficient is stored in 8 bits instead of 16 bits
|
||||
* The coefficients are rearranged in the column order 0, 2, 4, 6, 1, 3, 5, 7
|
||||
|
||||
This is done in order to avoid overflow: Since the tap with the largest
|
||||
coefficient could be any of taps 2, 3, 4 or 5, we can't use the summation
|
||||
order ((0 + 1) + (4 + 5)) + ((2 + 3) + (6 + 7)) used in the regular
|
||||
convolve functions.
|
||||
|
||||
Instead, we use the summation order
|
||||
((0 + 2) + (4 + 6)) + ((1 + 3) + (5 + 7)).
|
||||
The rearrangement of coefficients in this table is so that we can get the
|
||||
coefficients into the correct order more quickly.
|
||||
*/
|
||||
/* clang-format off */
|
||||
DECLARE_ALIGNED(8, static const int8_t,
|
||||
filter_8bit_neon[WARPEDPIXEL_PREC_SHIFTS * 3 + 1][8]) = {
|
||||
#if WARPEDPIXEL_PREC_BITS == 6
|
||||
// [-1, 0)
|
||||
{ 0, 127, 0, 0, 0, 1, 0, 0}, { 0, 127, 0, 0, -1, 2, 0, 0},
|
||||
{ 1, 127, -1, 0, -3, 4, 0, 0}, { 1, 126, -2, 0, -4, 6, 1, 0},
|
||||
{ 1, 126, -3, 0, -5, 8, 1, 0}, { 1, 125, -4, 0, -6, 11, 1, 0},
|
||||
{ 1, 124, -4, 0, -7, 13, 1, 0}, { 2, 123, -5, 0, -8, 15, 1, 0},
|
||||
{ 2, 122, -6, 0, -9, 18, 1, 0}, { 2, 121, -6, 0, -10, 20, 1, 0},
|
||||
{ 2, 120, -7, 0, -11, 22, 2, 0}, { 2, 119, -8, 0, -12, 25, 2, 0},
|
||||
{ 3, 117, -8, 0, -13, 27, 2, 0}, { 3, 116, -9, 0, -13, 29, 2, 0},
|
||||
{ 3, 114, -10, 0, -14, 32, 3, 0}, { 3, 113, -10, 0, -15, 35, 2, 0},
|
||||
{ 3, 111, -11, 0, -15, 37, 3, 0}, { 3, 109, -11, 0, -16, 40, 3, 0},
|
||||
{ 3, 108, -12, 0, -16, 42, 3, 0}, { 4, 106, -13, 0, -17, 45, 3, 0},
|
||||
{ 4, 104, -13, 0, -17, 47, 3, 0}, { 4, 102, -14, 0, -17, 50, 3, 0},
|
||||
{ 4, 100, -14, 0, -17, 52, 3, 0}, { 4, 98, -15, 0, -18, 55, 4, 0},
|
||||
{ 4, 96, -15, 0, -18, 58, 3, 0}, { 4, 94, -16, 0, -18, 60, 4, 0},
|
||||
{ 4, 91, -16, 0, -18, 63, 4, 0}, { 4, 89, -16, 0, -18, 65, 4, 0},
|
||||
{ 4, 87, -17, 0, -18, 68, 4, 0}, { 4, 85, -17, 0, -18, 70, 4, 0},
|
||||
{ 4, 82, -17, 0, -18, 73, 4, 0}, { 4, 80, -17, 0, -18, 75, 4, 0},
|
||||
{ 4, 78, -18, 0, -18, 78, 4, 0}, { 4, 75, -18, 0, -17, 80, 4, 0},
|
||||
{ 4, 73, -18, 0, -17, 82, 4, 0}, { 4, 70, -18, 0, -17, 85, 4, 0},
|
||||
{ 4, 68, -18, 0, -17, 87, 4, 0}, { 4, 65, -18, 0, -16, 89, 4, 0},
|
||||
{ 4, 63, -18, 0, -16, 91, 4, 0}, { 4, 60, -18, 0, -16, 94, 4, 0},
|
||||
{ 3, 58, -18, 0, -15, 96, 4, 0}, { 4, 55, -18, 0, -15, 98, 4, 0},
|
||||
{ 3, 52, -17, 0, -14, 100, 4, 0}, { 3, 50, -17, 0, -14, 102, 4, 0},
|
||||
{ 3, 47, -17, 0, -13, 104, 4, 0}, { 3, 45, -17, 0, -13, 106, 4, 0},
|
||||
{ 3, 42, -16, 0, -12, 108, 3, 0}, { 3, 40, -16, 0, -11, 109, 3, 0},
|
||||
{ 3, 37, -15, 0, -11, 111, 3, 0}, { 2, 35, -15, 0, -10, 113, 3, 0},
|
||||
{ 3, 32, -14, 0, -10, 114, 3, 0}, { 2, 29, -13, 0, -9, 116, 3, 0},
|
||||
{ 2, 27, -13, 0, -8, 117, 3, 0}, { 2, 25, -12, 0, -8, 119, 2, 0},
|
||||
{ 2, 22, -11, 0, -7, 120, 2, 0}, { 1, 20, -10, 0, -6, 121, 2, 0},
|
||||
{ 1, 18, -9, 0, -6, 122, 2, 0}, { 1, 15, -8, 0, -5, 123, 2, 0},
|
||||
{ 1, 13, -7, 0, -4, 124, 1, 0}, { 1, 11, -6, 0, -4, 125, 1, 0},
|
||||
{ 1, 8, -5, 0, -3, 126, 1, 0}, { 1, 6, -4, 0, -2, 126, 1, 0},
|
||||
{ 0, 4, -3, 0, -1, 127, 1, 0}, { 0, 2, -1, 0, 0, 127, 0, 0},
|
||||
// [0, 1)
|
||||
{ 0, 0, 1, 0, 0, 127, 0, 0}, { 0, -1, 2, 0, 0, 127, 0, 0},
|
||||
{ 0, -3, 4, 1, 1, 127, -2, 0}, { 0, -5, 6, 1, 1, 127, -2, 0},
|
||||
{ 0, -6, 8, 1, 2, 126, -3, 0}, {-1, -7, 11, 2, 2, 126, -4, -1},
|
||||
{-1, -8, 13, 2, 3, 125, -5, -1}, {-1, -10, 16, 3, 3, 124, -6, -1},
|
||||
{-1, -11, 18, 3, 4, 123, -7, -1}, {-1, -12, 20, 3, 4, 122, -7, -1},
|
||||
{-1, -13, 23, 3, 4, 121, -8, -1}, {-2, -14, 25, 4, 5, 120, -9, -1},
|
||||
{-1, -15, 27, 4, 5, 119, -10, -1}, {-1, -16, 30, 4, 5, 118, -11, -1},
|
||||
{-2, -17, 33, 5, 6, 116, -12, -1}, {-2, -17, 35, 5, 6, 114, -12, -1},
|
||||
{-2, -18, 38, 5, 6, 113, -13, -1}, {-2, -19, 41, 6, 7, 111, -14, -2},
|
||||
{-2, -19, 43, 6, 7, 110, -15, -2}, {-2, -20, 46, 6, 7, 108, -15, -2},
|
||||
{-2, -20, 49, 6, 7, 106, -16, -2}, {-2, -21, 51, 7, 7, 104, -16, -2},
|
||||
{-2, -21, 54, 7, 7, 102, -17, -2}, {-2, -21, 56, 7, 8, 100, -18, -2},
|
||||
{-2, -22, 59, 7, 8, 98, -18, -2}, {-2, -22, 62, 7, 8, 96, -19, -2},
|
||||
{-2, -22, 64, 7, 8, 94, -19, -2}, {-2, -22, 67, 8, 8, 91, -20, -2},
|
||||
{-2, -22, 69, 8, 8, 89, -20, -2}, {-2, -22, 72, 8, 8, 87, -21, -2},
|
||||
{-2, -21, 74, 8, 8, 84, -21, -2}, {-2, -22, 77, 8, 8, 82, -21, -2},
|
||||
{-2, -21, 79, 8, 8, 79, -21, -2}, {-2, -21, 82, 8, 8, 77, -22, -2},
|
||||
{-2, -21, 84, 8, 8, 74, -21, -2}, {-2, -21, 87, 8, 8, 72, -22, -2},
|
||||
{-2, -20, 89, 8, 8, 69, -22, -2}, {-2, -20, 91, 8, 8, 67, -22, -2},
|
||||
{-2, -19, 94, 8, 7, 64, -22, -2}, {-2, -19, 96, 8, 7, 62, -22, -2},
|
||||
{-2, -18, 98, 8, 7, 59, -22, -2}, {-2, -18, 100, 8, 7, 56, -21, -2},
|
||||
{-2, -17, 102, 7, 7, 54, -21, -2}, {-2, -16, 104, 7, 7, 51, -21, -2},
|
||||
{-2, -16, 106, 7, 6, 49, -20, -2}, {-2, -15, 108, 7, 6, 46, -20, -2},
|
||||
{-2, -15, 110, 7, 6, 43, -19, -2}, {-2, -14, 111, 7, 6, 41, -19, -2},
|
||||
{-1, -13, 113, 6, 5, 38, -18, -2}, {-1, -12, 114, 6, 5, 35, -17, -2},
|
||||
{-1, -12, 116, 6, 5, 33, -17, -2}, {-1, -11, 118, 5, 4, 30, -16, -1},
|
||||
{-1, -10, 119, 5, 4, 27, -15, -1}, {-1, -9, 120, 5, 4, 25, -14, -2},
|
||||
{-1, -8, 121, 4, 3, 23, -13, -1}, {-1, -7, 122, 4, 3, 20, -12, -1},
|
||||
{-1, -7, 123, 4, 3, 18, -11, -1}, {-1, -6, 124, 3, 3, 16, -10, -1},
|
||||
{-1, -5, 125, 3, 2, 13, -8, -1}, {-1, -4, 126, 2, 2, 11, -7, -1},
|
||||
{ 0, -3, 126, 2, 1, 8, -6, 0}, { 0, -2, 127, 1, 1, 6, -5, 0},
|
||||
{ 0, -2, 127, 1, 1, 4, -3, 0}, { 0, 0, 127, 0, 0, 2, -1, 0},
|
||||
// [1, 2)
|
||||
{ 0, 0, 127, 0, 0, 1, 0, 0}, { 0, 0, 127, 0, 0, -1, 2, 0},
|
||||
{ 0, 1, 127, -1, 0, -3, 4, 0}, { 0, 1, 126, -2, 0, -4, 6, 1},
|
||||
{ 0, 1, 126, -3, 0, -5, 8, 1}, { 0, 1, 125, -4, 0, -6, 11, 1},
|
||||
{ 0, 1, 124, -4, 0, -7, 13, 1}, { 0, 2, 123, -5, 0, -8, 15, 1},
|
||||
{ 0, 2, 122, -6, 0, -9, 18, 1}, { 0, 2, 121, -6, 0, -10, 20, 1},
|
||||
{ 0, 2, 120, -7, 0, -11, 22, 2}, { 0, 2, 119, -8, 0, -12, 25, 2},
|
||||
{ 0, 3, 117, -8, 0, -13, 27, 2}, { 0, 3, 116, -9, 0, -13, 29, 2},
|
||||
{ 0, 3, 114, -10, 0, -14, 32, 3}, { 0, 3, 113, -10, 0, -15, 35, 2},
|
||||
{ 0, 3, 111, -11, 0, -15, 37, 3}, { 0, 3, 109, -11, 0, -16, 40, 3},
|
||||
{ 0, 3, 108, -12, 0, -16, 42, 3}, { 0, 4, 106, -13, 0, -17, 45, 3},
|
||||
{ 0, 4, 104, -13, 0, -17, 47, 3}, { 0, 4, 102, -14, 0, -17, 50, 3},
|
||||
{ 0, 4, 100, -14, 0, -17, 52, 3}, { 0, 4, 98, -15, 0, -18, 55, 4},
|
||||
{ 0, 4, 96, -15, 0, -18, 58, 3}, { 0, 4, 94, -16, 0, -18, 60, 4},
|
||||
{ 0, 4, 91, -16, 0, -18, 63, 4}, { 0, 4, 89, -16, 0, -18, 65, 4},
|
||||
{ 0, 4, 87, -17, 0, -18, 68, 4}, { 0, 4, 85, -17, 0, -18, 70, 4},
|
||||
{ 0, 4, 82, -17, 0, -18, 73, 4}, { 0, 4, 80, -17, 0, -18, 75, 4},
|
||||
{ 0, 4, 78, -18, 0, -18, 78, 4}, { 0, 4, 75, -18, 0, -17, 80, 4},
|
||||
{ 0, 4, 73, -18, 0, -17, 82, 4}, { 0, 4, 70, -18, 0, -17, 85, 4},
|
||||
{ 0, 4, 68, -18, 0, -17, 87, 4}, { 0, 4, 65, -18, 0, -16, 89, 4},
|
||||
{ 0, 4, 63, -18, 0, -16, 91, 4}, { 0, 4, 60, -18, 0, -16, 94, 4},
|
||||
{ 0, 3, 58, -18, 0, -15, 96, 4}, { 0, 4, 55, -18, 0, -15, 98, 4},
|
||||
{ 0, 3, 52, -17, 0, -14, 100, 4}, { 0, 3, 50, -17, 0, -14, 102, 4},
|
||||
{ 0, 3, 47, -17, 0, -13, 104, 4}, { 0, 3, 45, -17, 0, -13, 106, 4},
|
||||
{ 0, 3, 42, -16, 0, -12, 108, 3}, { 0, 3, 40, -16, 0, -11, 109, 3},
|
||||
{ 0, 3, 37, -15, 0, -11, 111, 3}, { 0, 2, 35, -15, 0, -10, 113, 3},
|
||||
{ 0, 3, 32, -14, 0, -10, 114, 3}, { 0, 2, 29, -13, 0, -9, 116, 3},
|
||||
{ 0, 2, 27, -13, 0, -8, 117, 3}, { 0, 2, 25, -12, 0, -8, 119, 2},
|
||||
{ 0, 2, 22, -11, 0, -7, 120, 2}, { 0, 1, 20, -10, 0, -6, 121, 2},
|
||||
{ 0, 1, 18, -9, 0, -6, 122, 2}, { 0, 1, 15, -8, 0, -5, 123, 2},
|
||||
{ 0, 1, 13, -7, 0, -4, 124, 1}, { 0, 1, 11, -6, 0, -4, 125, 1},
|
||||
{ 0, 1, 8, -5, 0, -3, 126, 1}, { 0, 1, 6, -4, 0, -2, 126, 1},
|
||||
{ 0, 0, 4, -3, 0, -1, 127, 1}, { 0, 0, 2, -1, 0, 0, 127, 0},
|
||||
// dummy (replicate row index 191)
|
||||
{ 0, 0, 2, -1, 0, 0, 127, 0},
|
||||
|
||||
#else
|
||||
// [-1, 0)
|
||||
{ 0, 127, 0, 0, 0, 1, 0, 0}, { 1, 127, -1, 0, -3, 4, 0, 0},
|
||||
{ 1, 126, -3, 0, -5, 8, 1, 0}, { 1, 124, -4, 0, -7, 13, 1, 0},
|
||||
{ 2, 122, -6, 0, -9, 18, 1, 0}, { 2, 120, -7, 0, -11, 22, 2, 0},
|
||||
{ 3, 117, -8, 0, -13, 27, 2, 0}, { 3, 114, -10, 0, -14, 32, 3, 0},
|
||||
{ 3, 111, -11, 0, -15, 37, 3, 0}, { 3, 108, -12, 0, -16, 42, 3, 0},
|
||||
{ 4, 104, -13, 0, -17, 47, 3, 0}, { 4, 100, -14, 0, -17, 52, 3, 0},
|
||||
{ 4, 96, -15, 0, -18, 58, 3, 0}, { 4, 91, -16, 0, -18, 63, 4, 0},
|
||||
{ 4, 87, -17, 0, -18, 68, 4, 0}, { 4, 82, -17, 0, -18, 73, 4, 0},
|
||||
{ 4, 78, -18, 0, -18, 78, 4, 0}, { 4, 73, -18, 0, -17, 82, 4, 0},
|
||||
{ 4, 68, -18, 0, -17, 87, 4, 0}, { 4, 63, -18, 0, -16, 91, 4, 0},
|
||||
{ 3, 58, -18, 0, -15, 96, 4, 0}, { 3, 52, -17, 0, -14, 100, 4, 0},
|
||||
{ 3, 47, -17, 0, -13, 104, 4, 0}, { 3, 42, -16, 0, -12, 108, 3, 0},
|
||||
{ 3, 37, -15, 0, -11, 111, 3, 0}, { 3, 32, -14, 0, -10, 114, 3, 0},
|
||||
{ 2, 27, -13, 0, -8, 117, 3, 0}, { 2, 22, -11, 0, -7, 120, 2, 0},
|
||||
{ 1, 18, -9, 0, -6, 122, 2, 0}, { 1, 13, -7, 0, -4, 124, 1, 0},
|
||||
{ 1, 8, -5, 0, -3, 126, 1, 0}, { 0, 4, -3, 0, -1, 127, 1, 0},
|
||||
// [0, 1)
|
||||
{ 0, 0, 1, 0, 0, 127, 0, 0}, { 0, -3, 4, 1, 1, 127, -2, 0},
|
||||
{ 0, -6, 8, 1, 2, 126, -3, 0}, {-1, -8, 13, 2, 3, 125, -5, -1},
|
||||
{-1, -11, 18, 3, 4, 123, -7, -1}, {-1, -13, 23, 3, 4, 121, -8, -1},
|
||||
{-1, -15, 27, 4, 5, 119, -10, -1}, {-2, -17, 33, 5, 6, 116, -12, -1},
|
||||
{-2, -18, 38, 5, 6, 113, -13, -1}, {-2, -19, 43, 6, 7, 110, -15, -2},
|
||||
{-2, -20, 49, 6, 7, 106, -16, -2}, {-2, -21, 54, 7, 7, 102, -17, -2},
|
||||
{-2, -22, 59, 7, 8, 98, -18, -2}, {-2, -22, 64, 7, 8, 94, -19, -2},
|
||||
{-2, -22, 69, 8, 8, 89, -20, -2}, {-2, -21, 74, 8, 8, 84, -21, -2},
|
||||
{-2, -21, 79, 8, 8, 79, -21, -2}, {-2, -21, 84, 8, 8, 74, -21, -2},
|
||||
{-2, -20, 89, 8, 8, 69, -22, -2}, {-2, -19, 94, 8, 7, 64, -22, -2},
|
||||
{-2, -18, 98, 8, 7, 59, -22, -2}, {-2, -17, 102, 7, 7, 54, -21, -2},
|
||||
{-2, -16, 106, 7, 6, 49, -20, -2}, {-2, -15, 110, 7, 6, 43, -19, -2},
|
||||
{-1, -13, 113, 6, 5, 38, -18, -2}, {-1, -12, 116, 6, 5, 33, -17, -2},
|
||||
{-1, -10, 119, 5, 4, 27, -15, -1}, {-1, -8, 121, 4, 3, 23, -13, -1},
|
||||
{-1, -7, 123, 4, 3, 18, -11, -1}, {-1, -5, 125, 3, 2, 13, -8, -1},
|
||||
{ 0, -3, 126, 2, 1, 8, -6, 0}, { 0, -2, 127, 1, 1, 4, -3, 0},
|
||||
// [1, 2)
|
||||
{ 0, 0, 127, 0, 0, 1, 0, 0}, { 0, 1, 127, -1, 0, -3, 4, 0},
|
||||
{ 0, 1, 126, -3, 0, -5, 8, 1}, { 0, 1, 124, -4, 0, -7, 13, 1},
|
||||
{ 0, 2, 122, -6, 0, -9, 18, 1}, { 0, 2, 120, -7, 0, -11, 22, 2},
|
||||
{ 0, 3, 117, -8, 0, -13, 27, 2}, { 0, 3, 114, -10, 0, -14, 32, 3},
|
||||
{ 0, 3, 111, -11, 0, -15, 37, 3}, { 0, 3, 108, -12, 0, -16, 42, 3},
|
||||
{ 0, 4, 104, -13, 0, -17, 47, 3}, { 0, 4, 100, -14, 0, -17, 52, 3},
|
||||
{ 0, 4, 96, -15, 0, -18, 58, 3}, { 0, 4, 91, -16, 0, -18, 63, 4},
|
||||
{ 0, 4, 87, -17, 0, -18, 68, 4}, { 0, 4, 82, -17, 0, -18, 73, 4},
|
||||
{ 0, 4, 78, -18, 0, -18, 78, 4}, { 0, 4, 73, -18, 0, -17, 82, 4},
|
||||
{ 0, 4, 68, -18, 0, -17, 87, 4}, { 0, 4, 63, -18, 0, -16, 91, 4},
|
||||
{ 0, 3, 58, -18, 0, -15, 96, 4}, { 0, 3, 52, -17, 0, -14, 100, 4},
|
||||
{ 0, 3, 47, -17, 0, -13, 104, 4}, { 0, 3, 42, -16, 0, -12, 108, 3},
|
||||
{ 0, 3, 37, -15, 0, -11, 111, 3}, { 0, 3, 32, -14, 0, -10, 114, 3},
|
||||
{ 0, 2, 27, -13, 0, -8, 117, 3}, { 0, 2, 22, -11, 0, -7, 120, 2},
|
||||
{ 0, 1, 18, -9, 0, -6, 122, 2}, { 0, 1, 13, -7, 0, -4, 124, 1},
|
||||
{ 0, 1, 8, -5, 0, -3, 126, 1}, { 0, 0, 4, -3, 0, -1, 127, 1},
|
||||
// dummy (replicate row index 95)
|
||||
{ 0, 0, 4, -3, 0, -1, 127, 1},
|
||||
#endif // WARPEDPIXEL_PREC_BITS == 6
|
||||
};
|
||||
/* clang-format on */
|
||||
|
||||
static INLINE void convolve(int32x2x2_t x0, int32x2x2_t x1, uint8x8_t src_0,
|
||||
uint8x8_t src_1, int16x4_t *res) {
|
||||
int16x8_t coeff_0, coeff_1;
|
||||
int16x8_t pix_0, pix_1;
|
||||
|
||||
coeff_0 = vcombine_s16(vreinterpret_s16_s32(x0.val[0]),
|
||||
vreinterpret_s16_s32(x1.val[0]));
|
||||
coeff_1 = vcombine_s16(vreinterpret_s16_s32(x0.val[1]),
|
||||
vreinterpret_s16_s32(x1.val[1]));
|
||||
|
||||
pix_0 = vreinterpretq_s16_u16(vmovl_u8(src_0));
|
||||
pix_0 = vmulq_s16(coeff_0, pix_0);
|
||||
|
||||
pix_1 = vreinterpretq_s16_u16(vmovl_u8(src_1));
|
||||
pix_0 = vmlaq_s16(pix_0, coeff_1, pix_1);
|
||||
|
||||
*res = vpadd_s16(vget_low_s16(pix_0), vget_high_s16(pix_0));
|
||||
}
|
||||
|
||||
static INLINE void horizontal_filter_neon(uint8x16_t src_1, uint8x16_t src_2,
|
||||
uint8x16_t src_3, uint8x16_t src_4,
|
||||
int16x8_t *tmp_dst, int sx, int alpha,
|
||||
int k, const int offset_bits_horiz,
|
||||
const int reduce_bits_horiz) {
|
||||
const uint8x16_t mask = { 255, 0, 255, 0, 255, 0, 255, 0,
|
||||
255, 0, 255, 0, 255, 0, 255, 0 };
|
||||
const int32x4_t add_const = vdupq_n_s32((int32_t)(1 << offset_bits_horiz));
|
||||
const int16x8_t shift = vdupq_n_s16(-(int16_t)reduce_bits_horiz);
|
||||
|
||||
int16x8_t f0, f1, f2, f3, f4, f5, f6, f7;
|
||||
int32x2x2_t b0, b1;
|
||||
uint8x8_t src_1_low, src_2_low, src_3_low, src_4_low, src_5_low, src_6_low;
|
||||
int32x4_t tmp_res_low, tmp_res_high;
|
||||
uint16x8_t res;
|
||||
int16x4_t res_0246_even, res_0246_odd, res_1357_even, res_1357_odd;
|
||||
|
||||
uint8x16_t tmp_0 = vandq_u8(src_1, mask);
|
||||
uint8x16_t tmp_1 = vandq_u8(src_2, mask);
|
||||
uint8x16_t tmp_2 = vandq_u8(src_3, mask);
|
||||
uint8x16_t tmp_3 = vandq_u8(src_4, mask);
|
||||
|
||||
tmp_2 = vextq_u8(tmp_0, tmp_0, 1);
|
||||
tmp_3 = vextq_u8(tmp_1, tmp_1, 1);
|
||||
|
||||
src_1 = vaddq_u8(tmp_0, tmp_2);
|
||||
src_2 = vaddq_u8(tmp_1, tmp_3);
|
||||
|
||||
src_1_low = vget_low_u8(src_1);
|
||||
src_2_low = vget_low_u8(src_2);
|
||||
src_3_low = vget_low_u8(vextq_u8(src_1, src_1, 4));
|
||||
src_4_low = vget_low_u8(vextq_u8(src_2, src_2, 4));
|
||||
src_5_low = vget_low_u8(vextq_u8(src_1, src_1, 2));
|
||||
src_6_low = vget_low_u8(vextq_u8(src_1, src_1, 6));
|
||||
|
||||
// Loading the 8 filter taps
|
||||
f0 = vmovl_s8(
|
||||
vld1_s8(filter_8bit_neon[(sx + 0 * alpha) >> WARPEDDIFF_PREC_BITS]));
|
||||
f1 = vmovl_s8(
|
||||
vld1_s8(filter_8bit_neon[(sx + 1 * alpha) >> WARPEDDIFF_PREC_BITS]));
|
||||
f2 = vmovl_s8(
|
||||
vld1_s8(filter_8bit_neon[(sx + 2 * alpha) >> WARPEDDIFF_PREC_BITS]));
|
||||
f3 = vmovl_s8(
|
||||
vld1_s8(filter_8bit_neon[(sx + 3 * alpha) >> WARPEDDIFF_PREC_BITS]));
|
||||
f4 = vmovl_s8(
|
||||
vld1_s8(filter_8bit_neon[(sx + 4 * alpha) >> WARPEDDIFF_PREC_BITS]));
|
||||
f5 = vmovl_s8(
|
||||
vld1_s8(filter_8bit_neon[(sx + 5 * alpha) >> WARPEDDIFF_PREC_BITS]));
|
||||
f6 = vmovl_s8(
|
||||
vld1_s8(filter_8bit_neon[(sx + 6 * alpha) >> WARPEDDIFF_PREC_BITS]));
|
||||
f7 = vmovl_s8(
|
||||
vld1_s8(filter_8bit_neon[(sx + 7 * alpha) >> WARPEDDIFF_PREC_BITS]));
|
||||
|
||||
b0 = vtrn_s32(vreinterpret_s32_s16(vget_low_s16(f0)),
|
||||
vreinterpret_s32_s16(vget_low_s16(f2)));
|
||||
b1 = vtrn_s32(vreinterpret_s32_s16(vget_low_s16(f4)),
|
||||
vreinterpret_s32_s16(vget_low_s16(f6)));
|
||||
convolve(b0, b1, src_1_low, src_3_low, &res_0246_even);
|
||||
|
||||
b0 = vtrn_s32(vreinterpret_s32_s16(vget_low_s16(f1)),
|
||||
vreinterpret_s32_s16(vget_low_s16(f3)));
|
||||
b1 = vtrn_s32(vreinterpret_s32_s16(vget_low_s16(f5)),
|
||||
vreinterpret_s32_s16(vget_low_s16(f7)));
|
||||
convolve(b0, b1, src_2_low, src_4_low, &res_0246_odd);
|
||||
|
||||
b0 = vtrn_s32(vreinterpret_s32_s16(vget_high_s16(f0)),
|
||||
vreinterpret_s32_s16(vget_high_s16(f2)));
|
||||
b1 = vtrn_s32(vreinterpret_s32_s16(vget_high_s16(f4)),
|
||||
vreinterpret_s32_s16(vget_high_s16(f6)));
|
||||
convolve(b0, b1, src_2_low, src_4_low, &res_1357_even);
|
||||
|
||||
b0 = vtrn_s32(vreinterpret_s32_s16(vget_high_s16(f1)),
|
||||
vreinterpret_s32_s16(vget_high_s16(f3)));
|
||||
b1 = vtrn_s32(vreinterpret_s32_s16(vget_high_s16(f5)),
|
||||
vreinterpret_s32_s16(vget_high_s16(f7)));
|
||||
convolve(b0, b1, src_5_low, src_6_low, &res_1357_odd);
|
||||
|
||||
tmp_res_low = vaddl_s16(res_0246_even, res_1357_even);
|
||||
tmp_res_high = vaddl_s16(res_0246_odd, res_1357_odd);
|
||||
|
||||
tmp_res_low = vaddq_s32(tmp_res_low, add_const);
|
||||
tmp_res_high = vaddq_s32(tmp_res_high, add_const);
|
||||
|
||||
res = vcombine_u16(vqmovun_s32(tmp_res_low), vqmovun_s32(tmp_res_high));
|
||||
res = vqrshlq_u16(res, shift);
|
||||
|
||||
tmp_dst[k + 7] = vreinterpretq_s16_u16(res);
|
||||
}
|
||||
|
||||
static INLINE void vertical_filter_neon(const int16x8_t *src,
|
||||
int32x4_t *res_low, int32x4_t *res_high,
|
||||
int sy, int gamma) {
|
||||
int16x4_t src_0, src_1, fltr_0, fltr_1;
|
||||
int32x4_t res_0, res_1;
|
||||
int32x2_t res_0_im, res_1_im;
|
||||
int32x4_t res_even, res_odd, im_res_0, im_res_1;
|
||||
|
||||
int16x8_t f0, f1, f2, f3, f4, f5, f6, f7;
|
||||
int16x8x2_t b0, b1, b2, b3;
|
||||
int32x4x2_t c0, c1, c2, c3;
|
||||
int32x4x2_t d0, d1, d2, d3;
|
||||
|
||||
b0 = vtrnq_s16(src[0], src[1]);
|
||||
b1 = vtrnq_s16(src[2], src[3]);
|
||||
b2 = vtrnq_s16(src[4], src[5]);
|
||||
b3 = vtrnq_s16(src[6], src[7]);
|
||||
|
||||
c0 = vtrnq_s32(vreinterpretq_s32_s16(b0.val[0]),
|
||||
vreinterpretq_s32_s16(b0.val[1]));
|
||||
c1 = vtrnq_s32(vreinterpretq_s32_s16(b1.val[0]),
|
||||
vreinterpretq_s32_s16(b1.val[1]));
|
||||
c2 = vtrnq_s32(vreinterpretq_s32_s16(b2.val[0]),
|
||||
vreinterpretq_s32_s16(b2.val[1]));
|
||||
c3 = vtrnq_s32(vreinterpretq_s32_s16(b3.val[0]),
|
||||
vreinterpretq_s32_s16(b3.val[1]));
|
||||
|
||||
f0 = vld1q_s16(
|
||||
(int16_t *)(warped_filter + ((sy + 0 * gamma) >> WARPEDDIFF_PREC_BITS)));
|
||||
f1 = vld1q_s16(
|
||||
(int16_t *)(warped_filter + ((sy + 1 * gamma) >> WARPEDDIFF_PREC_BITS)));
|
||||
f2 = vld1q_s16(
|
||||
(int16_t *)(warped_filter + ((sy + 2 * gamma) >> WARPEDDIFF_PREC_BITS)));
|
||||
f3 = vld1q_s16(
|
||||
(int16_t *)(warped_filter + ((sy + 3 * gamma) >> WARPEDDIFF_PREC_BITS)));
|
||||
f4 = vld1q_s16(
|
||||
(int16_t *)(warped_filter + ((sy + 4 * gamma) >> WARPEDDIFF_PREC_BITS)));
|
||||
f5 = vld1q_s16(
|
||||
(int16_t *)(warped_filter + ((sy + 5 * gamma) >> WARPEDDIFF_PREC_BITS)));
|
||||
f6 = vld1q_s16(
|
||||
(int16_t *)(warped_filter + ((sy + 6 * gamma) >> WARPEDDIFF_PREC_BITS)));
|
||||
f7 = vld1q_s16(
|
||||
(int16_t *)(warped_filter + ((sy + 7 * gamma) >> WARPEDDIFF_PREC_BITS)));
|
||||
|
||||
d0 = vtrnq_s32(vreinterpretq_s32_s16(f0), vreinterpretq_s32_s16(f2));
|
||||
d1 = vtrnq_s32(vreinterpretq_s32_s16(f4), vreinterpretq_s32_s16(f6));
|
||||
d2 = vtrnq_s32(vreinterpretq_s32_s16(f1), vreinterpretq_s32_s16(f3));
|
||||
d3 = vtrnq_s32(vreinterpretq_s32_s16(f5), vreinterpretq_s32_s16(f7));
|
||||
|
||||
// row:0,1 even_col:0,2
|
||||
src_0 = vget_low_s16(vreinterpretq_s16_s32(c0.val[0]));
|
||||
fltr_0 = vget_low_s16(vreinterpretq_s16_s32(d0.val[0]));
|
||||
res_0 = vmull_s16(src_0, fltr_0);
|
||||
|
||||
// row:0,1,2,3 even_col:0,2
|
||||
src_0 = vget_low_s16(vreinterpretq_s16_s32(c1.val[0]));
|
||||
fltr_0 = vget_low_s16(vreinterpretq_s16_s32(d0.val[1]));
|
||||
res_0 = vmlal_s16(res_0, src_0, fltr_0);
|
||||
res_0_im = vpadd_s32(vget_low_s32(res_0), vget_high_s32(res_0));
|
||||
|
||||
// row:0,1 even_col:4,6
|
||||
src_1 = vget_low_s16(vreinterpretq_s16_s32(c0.val[1]));
|
||||
fltr_1 = vget_low_s16(vreinterpretq_s16_s32(d1.val[0]));
|
||||
res_1 = vmull_s16(src_1, fltr_1);
|
||||
|
||||
// row:0,1,2,3 even_col:4,6
|
||||
src_1 = vget_low_s16(vreinterpretq_s16_s32(c1.val[1]));
|
||||
fltr_1 = vget_low_s16(vreinterpretq_s16_s32(d1.val[1]));
|
||||
res_1 = vmlal_s16(res_1, src_1, fltr_1);
|
||||
res_1_im = vpadd_s32(vget_low_s32(res_1), vget_high_s32(res_1));
|
||||
|
||||
// row:0,1,2,3 even_col:0,2,4,6
|
||||
im_res_0 = vcombine_s32(res_0_im, res_1_im);
|
||||
|
||||
// row:4,5 even_col:0,2
|
||||
src_0 = vget_low_s16(vreinterpretq_s16_s32(c2.val[0]));
|
||||
fltr_0 = vget_high_s16(vreinterpretq_s16_s32(d0.val[0]));
|
||||
res_0 = vmull_s16(src_0, fltr_0);
|
||||
|
||||
// row:4,5,6,7 even_col:0,2
|
||||
src_0 = vget_low_s16(vreinterpretq_s16_s32(c3.val[0]));
|
||||
fltr_0 = vget_high_s16(vreinterpretq_s16_s32(d0.val[1]));
|
||||
res_0 = vmlal_s16(res_0, src_0, fltr_0);
|
||||
res_0_im = vpadd_s32(vget_low_s32(res_0), vget_high_s32(res_0));
|
||||
|
||||
// row:4,5 even_col:4,6
|
||||
src_1 = vget_low_s16(vreinterpretq_s16_s32(c2.val[1]));
|
||||
fltr_1 = vget_high_s16(vreinterpretq_s16_s32(d1.val[0]));
|
||||
res_1 = vmull_s16(src_1, fltr_1);
|
||||
|
||||
// row:4,5,6,7 even_col:4,6
|
||||
src_1 = vget_low_s16(vreinterpretq_s16_s32(c3.val[1]));
|
||||
fltr_1 = vget_high_s16(vreinterpretq_s16_s32(d1.val[1]));
|
||||
res_1 = vmlal_s16(res_1, src_1, fltr_1);
|
||||
res_1_im = vpadd_s32(vget_low_s32(res_1), vget_high_s32(res_1));
|
||||
|
||||
// row:4,5,6,7 even_col:0,2,4,6
|
||||
im_res_1 = vcombine_s32(res_0_im, res_1_im);
|
||||
|
||||
// row:0-7 even_col:0,2,4,6
|
||||
res_even = vaddq_s32(im_res_0, im_res_1);
|
||||
|
||||
// row:0,1 odd_col:1,3
|
||||
src_0 = vget_high_s16(vreinterpretq_s16_s32(c0.val[0]));
|
||||
fltr_0 = vget_low_s16(vreinterpretq_s16_s32(d2.val[0]));
|
||||
res_0 = vmull_s16(src_0, fltr_0);
|
||||
|
||||
// row:0,1,2,3 odd_col:1,3
|
||||
src_0 = vget_high_s16(vreinterpretq_s16_s32(c1.val[0]));
|
||||
fltr_0 = vget_low_s16(vreinterpretq_s16_s32(d2.val[1]));
|
||||
res_0 = vmlal_s16(res_0, src_0, fltr_0);
|
||||
res_0_im = vpadd_s32(vget_low_s32(res_0), vget_high_s32(res_0));
|
||||
|
||||
// row:0,1 odd_col:5,7
|
||||
src_1 = vget_high_s16(vreinterpretq_s16_s32(c0.val[1]));
|
||||
fltr_1 = vget_low_s16(vreinterpretq_s16_s32(d3.val[0]));
|
||||
res_1 = vmull_s16(src_1, fltr_1);
|
||||
|
||||
// row:0,1,2,3 odd_col:5,7
|
||||
src_1 = vget_high_s16(vreinterpretq_s16_s32(c1.val[1]));
|
||||
fltr_1 = vget_low_s16(vreinterpretq_s16_s32(d3.val[1]));
|
||||
res_1 = vmlal_s16(res_1, src_1, fltr_1);
|
||||
res_1_im = vpadd_s32(vget_low_s32(res_1), vget_high_s32(res_1));
|
||||
|
||||
// row:0,1,2,3 odd_col:1,3,5,7
|
||||
im_res_0 = vcombine_s32(res_0_im, res_1_im);
|
||||
|
||||
// row:4,5 odd_col:1,3
|
||||
src_0 = vget_high_s16(vreinterpretq_s16_s32(c2.val[0]));
|
||||
fltr_0 = vget_high_s16(vreinterpretq_s16_s32(d2.val[0]));
|
||||
res_0 = vmull_s16(src_0, fltr_0);
|
||||
|
||||
// row:4,5,6,7 odd_col:1,3
|
||||
src_0 = vget_high_s16(vreinterpretq_s16_s32(c3.val[0]));
|
||||
fltr_0 = vget_high_s16(vreinterpretq_s16_s32(d2.val[1]));
|
||||
res_0 = vmlal_s16(res_0, src_0, fltr_0);
|
||||
res_0_im = vpadd_s32(vget_low_s32(res_0), vget_high_s32(res_0));
|
||||
|
||||
// row:4,5 odd_col:5,7
|
||||
src_1 = vget_high_s16(vreinterpretq_s16_s32(c2.val[1]));
|
||||
fltr_1 = vget_high_s16(vreinterpretq_s16_s32(d3.val[0]));
|
||||
res_1 = vmull_s16(src_1, fltr_1);
|
||||
|
||||
// row:4,5,6,7 odd_col:5,7
|
||||
src_1 = vget_high_s16(vreinterpretq_s16_s32(c3.val[1]));
|
||||
fltr_1 = vget_high_s16(vreinterpretq_s16_s32(d3.val[1]));
|
||||
res_1 = vmlal_s16(res_1, src_1, fltr_1);
|
||||
res_1_im = vpadd_s32(vget_low_s32(res_1), vget_high_s32(res_1));
|
||||
|
||||
// row:4,5,6,7 odd_col:1,3,5,7
|
||||
im_res_1 = vcombine_s32(res_0_im, res_1_im);
|
||||
|
||||
// row:0-7 odd_col:1,3,5,7
|
||||
res_odd = vaddq_s32(im_res_0, im_res_1);
|
||||
|
||||
// reordering as 0 1 2 3 | 4 5 6 7
|
||||
c0 = vtrnq_s32(res_even, res_odd);
|
||||
|
||||
// Final store
|
||||
*res_low = vcombine_s32(vget_low_s32(c0.val[0]), vget_low_s32(c0.val[1]));
|
||||
*res_high = vcombine_s32(vget_high_s32(c0.val[0]), vget_high_s32(c0.val[1]));
|
||||
}
|
||||
|
||||
void av1_warp_affine_neon(const int32_t *mat, const uint8_t *ref, int width,
|
||||
int height, int stride, uint8_t *pred, int p_col,
|
||||
int p_row, int p_width, int p_height, int p_stride,
|
||||
int subsampling_x, int subsampling_y,
|
||||
ConvolveParams *conv_params, int16_t alpha,
|
||||
int16_t beta, int16_t gamma, int16_t delta) {
|
||||
int16x8_t tmp[15];
|
||||
const int bd = 8;
|
||||
const int w0 = conv_params->fwd_offset;
|
||||
const int w1 = conv_params->bck_offset;
|
||||
const int32x4_t fwd = vdupq_n_s32((int32_t)w0);
|
||||
const int32x4_t bwd = vdupq_n_s32((int32_t)w1);
|
||||
const int16x8_t sub_constant = vdupq_n_s16((1 << (bd - 1)) + (1 << bd));
|
||||
|
||||
int limit = 0;
|
||||
uint8x16_t vec_dup, mask_val;
|
||||
int32x4_t res_lo, res_hi;
|
||||
int16x8_t result_final;
|
||||
uint8x16_t src_1, src_2, src_3, src_4;
|
||||
uint8x16_t indx_vec = {
|
||||
0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15
|
||||
};
|
||||
uint8x16_t cmp_vec;
|
||||
|
||||
const int reduce_bits_horiz = conv_params->round_0;
|
||||
const int reduce_bits_vert = conv_params->is_compound
|
||||
? conv_params->round_1
|
||||
: 2 * FILTER_BITS - reduce_bits_horiz;
|
||||
const int32x4_t shift_vert = vdupq_n_s32(-(int32_t)reduce_bits_vert);
|
||||
const int offset_bits_horiz = bd + FILTER_BITS - 1;
|
||||
|
||||
assert(IMPLIES(conv_params->is_compound, conv_params->dst != NULL));
|
||||
|
||||
const int offset_bits_vert = bd + 2 * FILTER_BITS - reduce_bits_horiz;
|
||||
int32x4_t add_const_vert = vdupq_n_s32((int32_t)(1 << offset_bits_vert));
|
||||
const int round_bits =
|
||||
2 * FILTER_BITS - conv_params->round_0 - conv_params->round_1;
|
||||
const int16x4_t round_bits_vec = vdup_n_s16(-(int16_t)round_bits);
|
||||
const int offset_bits = bd + 2 * FILTER_BITS - conv_params->round_0;
|
||||
const int16x4_t res_sub_const =
|
||||
vdup_n_s16(-((1 << (offset_bits - conv_params->round_1)) +
|
||||
(1 << (offset_bits - conv_params->round_1 - 1))));
|
||||
int k;
|
||||
|
||||
assert(IMPLIES(conv_params->do_average, conv_params->is_compound));
|
||||
|
||||
for (int i = 0; i < p_height; i += 8) {
|
||||
for (int j = 0; j < p_width; j += 8) {
|
||||
const int32_t src_x = (p_col + j + 4) << subsampling_x;
|
||||
const int32_t src_y = (p_row + i + 4) << subsampling_y;
|
||||
const int32_t dst_x = mat[2] * src_x + mat[3] * src_y + mat[0];
|
||||
const int32_t dst_y = mat[4] * src_x + mat[5] * src_y + mat[1];
|
||||
const int32_t x4 = dst_x >> subsampling_x;
|
||||
const int32_t y4 = dst_y >> subsampling_y;
|
||||
|
||||
int32_t ix4 = x4 >> WARPEDMODEL_PREC_BITS;
|
||||
int32_t sx4 = x4 & ((1 << WARPEDMODEL_PREC_BITS) - 1);
|
||||
int32_t iy4 = y4 >> WARPEDMODEL_PREC_BITS;
|
||||
int32_t sy4 = y4 & ((1 << WARPEDMODEL_PREC_BITS) - 1);
|
||||
|
||||
sx4 += alpha * (-4) + beta * (-4) + (1 << (WARPEDDIFF_PREC_BITS - 1)) +
|
||||
(WARPEDPIXEL_PREC_SHIFTS << WARPEDDIFF_PREC_BITS);
|
||||
sy4 += gamma * (-4) + delta * (-4) + (1 << (WARPEDDIFF_PREC_BITS - 1)) +
|
||||
(WARPEDPIXEL_PREC_SHIFTS << WARPEDDIFF_PREC_BITS);
|
||||
|
||||
sx4 &= ~((1 << WARP_PARAM_REDUCE_BITS) - 1);
|
||||
sy4 &= ~((1 << WARP_PARAM_REDUCE_BITS) - 1);
|
||||
// horizontal
|
||||
if (ix4 <= -7) {
|
||||
for (k = -7; k < AOMMIN(8, p_height - i); ++k) {
|
||||
int iy = iy4 + k;
|
||||
if (iy < 0)
|
||||
iy = 0;
|
||||
else if (iy > height - 1)
|
||||
iy = height - 1;
|
||||
int16_t dup_val =
|
||||
(1 << (bd + FILTER_BITS - reduce_bits_horiz - 1)) +
|
||||
ref[iy * stride] * (1 << (FILTER_BITS - reduce_bits_horiz));
|
||||
|
||||
tmp[k + 7] = vdupq_n_s16(dup_val);
|
||||
}
|
||||
} else if (ix4 >= width + 6) {
|
||||
for (k = -7; k < AOMMIN(8, p_height - i); ++k) {
|
||||
int iy = iy4 + k;
|
||||
if (iy < 0)
|
||||
iy = 0;
|
||||
else if (iy > height - 1)
|
||||
iy = height - 1;
|
||||
int16_t dup_val = (1 << (bd + FILTER_BITS - reduce_bits_horiz - 1)) +
|
||||
ref[iy * stride + (width - 1)] *
|
||||
(1 << (FILTER_BITS - reduce_bits_horiz));
|
||||
tmp[k + 7] = vdupq_n_s16(dup_val);
|
||||
}
|
||||
} else if (((ix4 - 7) < 0) || ((ix4 + 9) > width)) {
|
||||
const int out_of_boundary_left = -(ix4 - 6);
|
||||
const int out_of_boundary_right = (ix4 + 8) - width;
|
||||
|
||||
for (k = -7; k < AOMMIN(8, p_height - i); ++k) {
|
||||
int iy = iy4 + k;
|
||||
if (iy < 0)
|
||||
iy = 0;
|
||||
else if (iy > height - 1)
|
||||
iy = height - 1;
|
||||
int sx = sx4 + beta * (k + 4);
|
||||
|
||||
const uint8_t *src = ref + iy * stride + ix4 - 7;
|
||||
src_1 = vld1q_u8(src);
|
||||
|
||||
if (out_of_boundary_left >= 0) {
|
||||
limit = out_of_boundary_left + 1;
|
||||
cmp_vec = vdupq_n_u8(out_of_boundary_left);
|
||||
vec_dup = vdupq_n_u8(*(src + limit));
|
||||
mask_val = vcleq_u8(indx_vec, cmp_vec);
|
||||
src_1 = vbslq_u8(mask_val, vec_dup, src_1);
|
||||
}
|
||||
if (out_of_boundary_right >= 0) {
|
||||
limit = 15 - (out_of_boundary_right + 1);
|
||||
cmp_vec = vdupq_n_u8(15 - out_of_boundary_right);
|
||||
vec_dup = vdupq_n_u8(*(src + limit));
|
||||
mask_val = vcgeq_u8(indx_vec, cmp_vec);
|
||||
src_1 = vbslq_u8(mask_val, vec_dup, src_1);
|
||||
}
|
||||
src_2 = vextq_u8(src_1, src_1, 1);
|
||||
src_3 = vextq_u8(src_2, src_2, 1);
|
||||
src_4 = vextq_u8(src_3, src_3, 1);
|
||||
|
||||
horizontal_filter_neon(src_1, src_2, src_3, src_4, tmp, sx, alpha, k,
|
||||
offset_bits_horiz, reduce_bits_horiz);
|
||||
}
|
||||
} else {
|
||||
for (k = -7; k < AOMMIN(8, p_height - i); ++k) {
|
||||
int iy = iy4 + k;
|
||||
if (iy < 0)
|
||||
iy = 0;
|
||||
else if (iy > height - 1)
|
||||
iy = height - 1;
|
||||
int sx = sx4 + beta * (k + 4);
|
||||
|
||||
const uint8_t *src = ref + iy * stride + ix4 - 7;
|
||||
src_1 = vld1q_u8(src);
|
||||
src_2 = vextq_u8(src_1, src_1, 1);
|
||||
src_3 = vextq_u8(src_2, src_2, 1);
|
||||
src_4 = vextq_u8(src_3, src_3, 1);
|
||||
|
||||
horizontal_filter_neon(src_1, src_2, src_3, src_4, tmp, sx, alpha, k,
|
||||
offset_bits_horiz, reduce_bits_horiz);
|
||||
}
|
||||
}
|
||||
|
||||
// vertical
|
||||
for (k = -4; k < AOMMIN(4, p_height - i - 4); ++k) {
|
||||
int sy = sy4 + delta * (k + 4);
|
||||
|
||||
const int16x8_t *v_src = tmp + (k + 4);
|
||||
|
||||
vertical_filter_neon(v_src, &res_lo, &res_hi, sy, gamma);
|
||||
|
||||
res_lo = vaddq_s32(res_lo, add_const_vert);
|
||||
res_hi = vaddq_s32(res_hi, add_const_vert);
|
||||
|
||||
if (conv_params->is_compound) {
|
||||
uint16_t *const p =
|
||||
(uint16_t *)&conv_params
|
||||
->dst[(i + k + 4) * conv_params->dst_stride + j];
|
||||
|
||||
res_lo = vrshlq_s32(res_lo, shift_vert);
|
||||
if (conv_params->do_average) {
|
||||
uint8_t *const dst8 = &pred[(i + k + 4) * p_stride + j];
|
||||
uint16x4_t tmp16_lo = vld1_u16(p);
|
||||
int32x4_t tmp32_lo = vreinterpretq_s32_u32(vmovl_u16(tmp16_lo));
|
||||
int16x4_t tmp16_low;
|
||||
if (conv_params->use_jnt_comp_avg) {
|
||||
res_lo = vmulq_s32(res_lo, bwd);
|
||||
tmp32_lo = vmulq_s32(tmp32_lo, fwd);
|
||||
tmp32_lo = vaddq_s32(tmp32_lo, res_lo);
|
||||
tmp16_low = vshrn_n_s32(tmp32_lo, DIST_PRECISION_BITS);
|
||||
} else {
|
||||
tmp32_lo = vaddq_s32(tmp32_lo, res_lo);
|
||||
tmp16_low = vshrn_n_s32(tmp32_lo, 1);
|
||||
}
|
||||
int16x4_t res_low = vadd_s16(tmp16_low, res_sub_const);
|
||||
res_low = vqrshl_s16(res_low, round_bits_vec);
|
||||
int16x8_t final_res_low = vcombine_s16(res_low, res_low);
|
||||
uint8x8_t res_8_low = vqmovun_s16(final_res_low);
|
||||
|
||||
vst1_lane_u32((uint32_t *)dst8, vreinterpret_u32_u8(res_8_low), 0);
|
||||
} else {
|
||||
uint16x4_t res_u16_low = vqmovun_s32(res_lo);
|
||||
vst1_u16(p, res_u16_low);
|
||||
}
|
||||
if (p_width > 4) {
|
||||
uint16_t *const p4 =
|
||||
(uint16_t *)&conv_params
|
||||
->dst[(i + k + 4) * conv_params->dst_stride + j + 4];
|
||||
|
||||
res_hi = vrshlq_s32(res_hi, shift_vert);
|
||||
if (conv_params->do_average) {
|
||||
uint8_t *const dst8_4 = &pred[(i + k + 4) * p_stride + j + 4];
|
||||
|
||||
uint16x4_t tmp16_hi = vld1_u16(p4);
|
||||
int32x4_t tmp32_hi = vreinterpretq_s32_u32(vmovl_u16(tmp16_hi));
|
||||
int16x4_t tmp16_high;
|
||||
if (conv_params->use_jnt_comp_avg) {
|
||||
res_hi = vmulq_s32(res_hi, bwd);
|
||||
tmp32_hi = vmulq_s32(tmp32_hi, fwd);
|
||||
tmp32_hi = vaddq_s32(tmp32_hi, res_hi);
|
||||
tmp16_high = vshrn_n_s32(tmp32_hi, DIST_PRECISION_BITS);
|
||||
} else {
|
||||
tmp32_hi = vaddq_s32(tmp32_hi, res_hi);
|
||||
tmp16_high = vshrn_n_s32(tmp32_hi, 1);
|
||||
}
|
||||
int16x4_t res_high = vadd_s16(tmp16_high, res_sub_const);
|
||||
res_high = vqrshl_s16(res_high, round_bits_vec);
|
||||
int16x8_t final_res_high = vcombine_s16(res_high, res_high);
|
||||
uint8x8_t res_8_high = vqmovun_s16(final_res_high);
|
||||
|
||||
vst1_lane_u32((uint32_t *)dst8_4, vreinterpret_u32_u8(res_8_high),
|
||||
0);
|
||||
} else {
|
||||
uint16x4_t res_u16_high = vqmovun_s32(res_hi);
|
||||
vst1_u16(p4, res_u16_high);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
res_lo = vrshlq_s32(res_lo, shift_vert);
|
||||
res_hi = vrshlq_s32(res_hi, shift_vert);
|
||||
|
||||
result_final = vcombine_s16(vmovn_s32(res_lo), vmovn_s32(res_hi));
|
||||
result_final = vsubq_s16(result_final, sub_constant);
|
||||
|
||||
uint8_t *const p = (uint8_t *)&pred[(i + k + 4) * p_stride + j];
|
||||
uint8x8_t val = vqmovun_s16(result_final);
|
||||
|
||||
if (p_width == 4) {
|
||||
vst1_lane_u32((uint32_t *)p, vreinterpret_u32_u8(val), 0);
|
||||
} else {
|
||||
vst1_u8(p, val);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
|
@ -26,7 +26,6 @@
|
|||
Apply horizontal filter and store in a temporary buffer. When applying
|
||||
vertical filter, overwrite the original pixel values.
|
||||
*/
|
||||
|
||||
void av1_wiener_convolve_add_src_neon(const uint8_t *src, ptrdiff_t src_stride,
|
||||
uint8_t *dst, ptrdiff_t dst_stride,
|
||||
const int16_t *filter_x, int x_step_q4,
|
||||
|
|
@ -78,8 +77,10 @@ void av1_wiener_convolve_add_src_neon(const uint8_t *src, ptrdiff_t src_stride,
|
|||
/* if height is a multiple of 8 */
|
||||
if (!(h & 7)) {
|
||||
int16x8_t res0, res1, res2, res3;
|
||||
uint16x8_t res4, res5, res6, res7, res8, res9, res10, res11;
|
||||
uint16x8_t res4;
|
||||
uint8x8_t t0, t1, t2, t3, t4, t5, t6, t7;
|
||||
#if defined(__aarch64__)
|
||||
uint16x8_t res5, res6, res7, res8, res9, res10, res11;
|
||||
uint8x8_t t8, t9, t10, t11, t12, t13, t14;
|
||||
|
||||
do {
|
||||
|
|
@ -190,16 +191,64 @@ void av1_wiener_convolve_add_src_neon(const uint8_t *src, ptrdiff_t src_stride,
|
|||
dst_ptr += 8 * MAX_SB_SIZE;
|
||||
height -= 8;
|
||||
} while (height > 0);
|
||||
#else
|
||||
uint8x8_t temp_0;
|
||||
|
||||
do {
|
||||
const uint8_t *s;
|
||||
|
||||
__builtin_prefetch(src_ptr);
|
||||
|
||||
t0 = vld1_u8(src_ptr); // a0 a1 a2 a3 a4 a5 a6 a7
|
||||
s = src_ptr + 8;
|
||||
d_tmp = dst_ptr;
|
||||
width = w;
|
||||
|
||||
__builtin_prefetch(dst_ptr);
|
||||
|
||||
do {
|
||||
t7 = vld1_u8(s); // a8 a9 a10 a11 a12 a13 a14 a15
|
||||
temp_0 = t0;
|
||||
t0 = t7;
|
||||
|
||||
t1 = vext_u8(temp_0, t7, 1); // a1 a2 a3 a4 a5 a6 a7 a8
|
||||
t2 = vext_u8(temp_0, t7, 2); // a2 a3 a4 a5 a6 a7 a8 a9
|
||||
t3 = vext_u8(temp_0, t7, 3); // a3 a4 a5 a6 a7 a8 a9 a10
|
||||
t4 = vext_u8(temp_0, t7, 4); // a4 a5 a6 a7 a8 a9 a10 a11
|
||||
t5 = vext_u8(temp_0, t7, 5); // a5 a6 a7 a8 a9 a10 a11 a12
|
||||
t6 = vext_u8(temp_0, t7, 6); // a6 a7 a8 a9 a10 a11 a12 a13
|
||||
t7 = vext_u8(temp_0, t7, 7); // a7 a8 a9 a10 a11 a12 a13 a14
|
||||
|
||||
res0 = vreinterpretq_s16_u16(vaddl_u8(temp_0, t6));
|
||||
res1 = vreinterpretq_s16_u16(vaddl_u8(t1, t5));
|
||||
res2 = vreinterpretq_s16_u16(vaddl_u8(t2, t4));
|
||||
res3 = vreinterpretq_s16_u16(vmovl_u8(t3));
|
||||
res4 = wiener_convolve8_horiz_8x8(res0, res1, res2, res3, filter_x_tmp,
|
||||
bd, conv_params->round_0);
|
||||
|
||||
vst1q_u16(d_tmp, res4);
|
||||
|
||||
s += 8;
|
||||
d_tmp += 8;
|
||||
width -= 8;
|
||||
} while (width > 0);
|
||||
src_ptr += src_stride;
|
||||
dst_ptr += MAX_SB_SIZE;
|
||||
height--;
|
||||
} while (height > 0);
|
||||
#endif
|
||||
} else {
|
||||
/*if height is a multiple of 4*/
|
||||
int16x8_t tt0, tt1, tt2, tt3;
|
||||
const uint8_t *s;
|
||||
uint16x4_t res0, res1, res2, res3, res4, res5, res6, res7;
|
||||
uint16x8_t d0, d1, d2, d3;
|
||||
int16x4_t s0, s1, s2, s3, s4, s5, s6, s7, s8, s9, s10;
|
||||
int16x4_t s11, s12, s13, s14;
|
||||
int16x8_t tt0, tt1, tt2, tt3;
|
||||
uint16x8_t d0;
|
||||
uint8x8_t t0, t1, t2, t3;
|
||||
|
||||
#if defined(__aarch64__)
|
||||
uint16x4_t res0, res1, res2, res3, res4, res5, res6, res7;
|
||||
uint16x8_t d1, d2, d3;
|
||||
int16x4_t s0, s1, s2, s3, s4, s5, s6, s7, s8, s9, s10;
|
||||
int16x4_t s11, s12, s13, s14;
|
||||
do {
|
||||
__builtin_prefetch(src_ptr + 0 * src_stride);
|
||||
__builtin_prefetch(src_ptr + 1 * src_stride);
|
||||
|
|
@ -292,11 +341,61 @@ void av1_wiener_convolve_add_src_neon(const uint8_t *src, ptrdiff_t src_stride,
|
|||
dst_ptr += 4 * MAX_SB_SIZE;
|
||||
height -= 4;
|
||||
} while (height > 0);
|
||||
#else
|
||||
uint8x8_t temp_0, t4, t5, t6, t7;
|
||||
|
||||
do {
|
||||
__builtin_prefetch(src_ptr);
|
||||
|
||||
t0 = vld1_u8(src_ptr); // a0 a1 a2 a3 a4 a5 a6 a7
|
||||
|
||||
__builtin_prefetch(dst_ptr);
|
||||
|
||||
s = src_ptr + 8;
|
||||
d_tmp = dst_ptr;
|
||||
width = w;
|
||||
|
||||
do {
|
||||
t7 = vld1_u8(s); // a8 a9 a10 a11 a12 a13 a14 a15
|
||||
temp_0 = t0;
|
||||
t0 = t7;
|
||||
|
||||
t1 = vext_u8(temp_0, t7, 1); // a1 a2 a3 a4 a5 a6 a7 a8
|
||||
t2 = vext_u8(temp_0, t7, 2); // a2 a3 a4 a5 a6 a7 a8 a9
|
||||
t3 = vext_u8(temp_0, t7, 3); // a3 a4 a5 a6 a7 a8 a9 a10
|
||||
t4 = vext_u8(temp_0, t7, 4); // a4 a5 a6 a7 a8 a9 a10 a11
|
||||
t5 = vext_u8(temp_0, t7, 5); // a5 a6 a7 a8 a9 a10 a11 a12
|
||||
t6 = vext_u8(temp_0, t7, 6); // a6 a7 a8 a9 a10 a11 a12 a13
|
||||
t7 = vext_u8(temp_0, t7, 7); // a7 a8 a9 a10 a11 a12 a13 a14
|
||||
|
||||
tt0 = vreinterpretq_s16_u16(vaddl_u8(temp_0, t6));
|
||||
tt1 = vreinterpretq_s16_u16(vaddl_u8(t1, t5));
|
||||
tt2 = vreinterpretq_s16_u16(vaddl_u8(t2, t4));
|
||||
tt3 = vreinterpretq_s16_u16(vmovl_u8(t3));
|
||||
d0 = wiener_convolve8_horiz_8x8(tt0, tt1, tt2, tt3, filter_x_tmp, bd,
|
||||
conv_params->round_0);
|
||||
|
||||
vst1q_u16(d_tmp, d0);
|
||||
|
||||
s += 8;
|
||||
d_tmp += 8;
|
||||
width -= 8;
|
||||
} while (width > 0);
|
||||
|
||||
src_ptr += src_stride;
|
||||
dst_ptr += MAX_SB_SIZE;
|
||||
height -= 1;
|
||||
} while (height > 0);
|
||||
#endif
|
||||
}
|
||||
|
||||
{
|
||||
int16x8_t s0, s1, s2, s3, s4, s5, s6, s7, s8, s9, s10;
|
||||
uint8x8_t t0, t1, t2, t3;
|
||||
int16x8_t s0, s1, s2, s3, s4, s5, s6, s7;
|
||||
uint8x8_t t0;
|
||||
#if defined(__aarch64__)
|
||||
int16x8_t s8, s9, s10;
|
||||
uint8x8_t t1, t2, t3;
|
||||
#endif
|
||||
int16_t *src_tmp_ptr, *s;
|
||||
uint8_t *dst_tmp_ptr;
|
||||
height = h;
|
||||
|
|
@ -324,6 +423,7 @@ void av1_wiener_convolve_add_src_neon(const uint8_t *src, ptrdiff_t src_stride,
|
|||
d = dst_tmp_ptr;
|
||||
height = h;
|
||||
|
||||
#if defined(__aarch64__)
|
||||
do {
|
||||
__builtin_prefetch(dst_tmp_ptr + 0 * dst_stride);
|
||||
__builtin_prefetch(dst_tmp_ptr + 1 * dst_stride);
|
||||
|
|
@ -397,5 +497,34 @@ void av1_wiener_convolve_add_src_neon(const uint8_t *src, ptrdiff_t src_stride,
|
|||
|
||||
w -= 8;
|
||||
} while (w > 0);
|
||||
#else
|
||||
do {
|
||||
__builtin_prefetch(dst_tmp_ptr + 0 * dst_stride);
|
||||
|
||||
s7 = vld1q_s16(s);
|
||||
s += src_stride;
|
||||
|
||||
t0 = wiener_convolve8_vert_4x8(s0, s1, s2, s3, s4, s5, s6, filter_y_tmp,
|
||||
bd, conv_params->round_1);
|
||||
|
||||
vst1_u8(d, t0);
|
||||
d += dst_stride;
|
||||
|
||||
s0 = s1;
|
||||
s1 = s2;
|
||||
s2 = s3;
|
||||
s3 = s4;
|
||||
s4 = s5;
|
||||
s5 = s6;
|
||||
s6 = s7;
|
||||
height -= 1;
|
||||
} while (height > 0);
|
||||
|
||||
src_tmp_ptr += 8;
|
||||
dst_tmp_ptr += 8;
|
||||
|
||||
w -= 8;
|
||||
} while (w > 0);
|
||||
#endif
|
||||
}
|
||||
}
|
||||
|
|
|
|||
140
third_party/aom/av1/common/av1_inv_txfm1d.c
vendored
140
third_party/aom/av1/common/av1_inv_txfm1d.c
vendored
|
|
@ -11,56 +11,7 @@
|
|||
|
||||
#include <stdlib.h>
|
||||
#include "av1/common/av1_inv_txfm1d.h"
|
||||
|
||||
static void range_check_buf(int32_t stage, const int32_t *input,
|
||||
const int32_t *buf, int32_t size, int8_t bit) {
|
||||
#if CONFIG_COEFFICIENT_RANGE_CHECKING
|
||||
const int64_t max_value = (1LL << (bit - 1)) - 1;
|
||||
const int64_t min_value = -(1LL << (bit - 1));
|
||||
|
||||
int in_range = 1;
|
||||
|
||||
for (int i = 0; i < size; ++i) {
|
||||
if (buf[i] < min_value || buf[i] > max_value) {
|
||||
in_range = 0;
|
||||
}
|
||||
}
|
||||
|
||||
if (!in_range) {
|
||||
fprintf(stderr, "Error: coeffs contain out-of-range values\n");
|
||||
fprintf(stderr, "size: %d\n", size);
|
||||
fprintf(stderr, "stage: %d\n", stage);
|
||||
fprintf(stderr, "allowed range: [%" PRId64 ";%" PRId64 "]\n", min_value,
|
||||
max_value);
|
||||
|
||||
fprintf(stderr, "coeffs: ");
|
||||
|
||||
fprintf(stderr, "[");
|
||||
for (int j = 0; j < size; j++) {
|
||||
if (j > 0) fprintf(stderr, ", ");
|
||||
fprintf(stderr, "%d", input[j]);
|
||||
}
|
||||
fprintf(stderr, "]\n");
|
||||
|
||||
fprintf(stderr, " buf: ");
|
||||
|
||||
fprintf(stderr, "[");
|
||||
for (int j = 0; j < size; j++) {
|
||||
if (j > 0) fprintf(stderr, ", ");
|
||||
fprintf(stderr, "%d", buf[j]);
|
||||
}
|
||||
fprintf(stderr, "]\n\n");
|
||||
}
|
||||
|
||||
assert(in_range);
|
||||
#else
|
||||
(void)stage;
|
||||
(void)input;
|
||||
(void)buf;
|
||||
(void)size;
|
||||
(void)bit;
|
||||
#endif
|
||||
}
|
||||
#include "av1/common/av1_txfm.h"
|
||||
|
||||
// TODO(angiebird): Make 1-d txfm functions static
|
||||
//
|
||||
|
|
@ -84,7 +35,7 @@ void av1_idct4_new(const int32_t *input, int32_t *output, int8_t cos_bit,
|
|||
bf1[1] = input[2];
|
||||
bf1[2] = input[1];
|
||||
bf1[3] = input[3];
|
||||
range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
|
||||
// stage 2
|
||||
stage++;
|
||||
|
|
@ -94,7 +45,7 @@ void av1_idct4_new(const int32_t *input, int32_t *output, int8_t cos_bit,
|
|||
bf1[1] = half_btf(cospi[32], bf0[0], -cospi[32], bf0[1], cos_bit);
|
||||
bf1[2] = half_btf(cospi[48], bf0[2], -cospi[16], bf0[3], cos_bit);
|
||||
bf1[3] = half_btf(cospi[16], bf0[2], cospi[48], bf0[3], cos_bit);
|
||||
range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
|
||||
// stage 3
|
||||
stage++;
|
||||
|
|
@ -129,7 +80,7 @@ void av1_idct8_new(const int32_t *input, int32_t *output, int8_t cos_bit,
|
|||
bf1[5] = input[5];
|
||||
bf1[6] = input[3];
|
||||
bf1[7] = input[7];
|
||||
range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
|
||||
// stage 2
|
||||
stage++;
|
||||
|
|
@ -143,7 +94,7 @@ void av1_idct8_new(const int32_t *input, int32_t *output, int8_t cos_bit,
|
|||
bf1[5] = half_btf(cospi[24], bf0[5], -cospi[40], bf0[6], cos_bit);
|
||||
bf1[6] = half_btf(cospi[40], bf0[5], cospi[24], bf0[6], cos_bit);
|
||||
bf1[7] = half_btf(cospi[8], bf0[4], cospi[56], bf0[7], cos_bit);
|
||||
range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
|
||||
// stage 3
|
||||
stage++;
|
||||
|
|
@ -157,7 +108,7 @@ void av1_idct8_new(const int32_t *input, int32_t *output, int8_t cos_bit,
|
|||
bf1[5] = clamp_value(bf0[4] - bf0[5], stage_range[stage]);
|
||||
bf1[6] = clamp_value(-bf0[6] + bf0[7], stage_range[stage]);
|
||||
bf1[7] = clamp_value(bf0[6] + bf0[7], stage_range[stage]);
|
||||
range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
|
||||
// stage 4
|
||||
stage++;
|
||||
|
|
@ -171,7 +122,7 @@ void av1_idct8_new(const int32_t *input, int32_t *output, int8_t cos_bit,
|
|||
bf1[5] = half_btf(-cospi[32], bf0[5], cospi[32], bf0[6], cos_bit);
|
||||
bf1[6] = half_btf(cospi[32], bf0[5], cospi[32], bf0[6], cos_bit);
|
||||
bf1[7] = bf0[7];
|
||||
range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
|
||||
// stage 5
|
||||
stage++;
|
||||
|
|
@ -218,7 +169,7 @@ void av1_idct16_new(const int32_t *input, int32_t *output, int8_t cos_bit,
|
|||
bf1[13] = input[11];
|
||||
bf1[14] = input[7];
|
||||
bf1[15] = input[15];
|
||||
range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
|
||||
// stage 2
|
||||
stage++;
|
||||
|
|
@ -240,7 +191,7 @@ void av1_idct16_new(const int32_t *input, int32_t *output, int8_t cos_bit,
|
|||
bf1[13] = half_btf(cospi[20], bf0[10], cospi[44], bf0[13], cos_bit);
|
||||
bf1[14] = half_btf(cospi[36], bf0[9], cospi[28], bf0[14], cos_bit);
|
||||
bf1[15] = half_btf(cospi[4], bf0[8], cospi[60], bf0[15], cos_bit);
|
||||
range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
|
||||
// stage 3
|
||||
stage++;
|
||||
|
|
@ -262,7 +213,7 @@ void av1_idct16_new(const int32_t *input, int32_t *output, int8_t cos_bit,
|
|||
bf1[13] = clamp_value(bf0[12] - bf0[13], stage_range[stage]);
|
||||
bf1[14] = clamp_value(-bf0[14] + bf0[15], stage_range[stage]);
|
||||
bf1[15] = clamp_value(bf0[14] + bf0[15], stage_range[stage]);
|
||||
range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
|
||||
// stage 4
|
||||
stage++;
|
||||
|
|
@ -284,7 +235,7 @@ void av1_idct16_new(const int32_t *input, int32_t *output, int8_t cos_bit,
|
|||
bf1[13] = half_btf(-cospi[16], bf0[10], cospi[48], bf0[13], cos_bit);
|
||||
bf1[14] = half_btf(cospi[48], bf0[9], cospi[16], bf0[14], cos_bit);
|
||||
bf1[15] = bf0[15];
|
||||
range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
|
||||
// stage 5
|
||||
stage++;
|
||||
|
|
@ -306,7 +257,7 @@ void av1_idct16_new(const int32_t *input, int32_t *output, int8_t cos_bit,
|
|||
bf1[13] = clamp_value(-bf0[13] + bf0[14], stage_range[stage]);
|
||||
bf1[14] = clamp_value(bf0[13] + bf0[14], stage_range[stage]);
|
||||
bf1[15] = clamp_value(bf0[12] + bf0[15], stage_range[stage]);
|
||||
range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
|
||||
// stage 6
|
||||
stage++;
|
||||
|
|
@ -328,7 +279,7 @@ void av1_idct16_new(const int32_t *input, int32_t *output, int8_t cos_bit,
|
|||
bf1[13] = half_btf(cospi[32], bf0[10], cospi[32], bf0[13], cos_bit);
|
||||
bf1[14] = bf0[14];
|
||||
bf1[15] = bf0[15];
|
||||
range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
|
||||
// stage 7
|
||||
stage++;
|
||||
|
|
@ -399,7 +350,7 @@ void av1_idct32_new(const int32_t *input, int32_t *output, int8_t cos_bit,
|
|||
bf1[29] = input[23];
|
||||
bf1[30] = input[15];
|
||||
bf1[31] = input[31];
|
||||
range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
|
||||
// stage 2
|
||||
stage++;
|
||||
|
|
@ -437,7 +388,7 @@ void av1_idct32_new(const int32_t *input, int32_t *output, int8_t cos_bit,
|
|||
bf1[29] = half_btf(cospi[18], bf0[18], cospi[46], bf0[29], cos_bit);
|
||||
bf1[30] = half_btf(cospi[34], bf0[17], cospi[30], bf0[30], cos_bit);
|
||||
bf1[31] = half_btf(cospi[2], bf0[16], cospi[62], bf0[31], cos_bit);
|
||||
range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
|
||||
// stage 3
|
||||
stage++;
|
||||
|
|
@ -475,7 +426,7 @@ void av1_idct32_new(const int32_t *input, int32_t *output, int8_t cos_bit,
|
|||
bf1[29] = clamp_value(bf0[28] - bf0[29], stage_range[stage]);
|
||||
bf1[30] = clamp_value(-bf0[30] + bf0[31], stage_range[stage]);
|
||||
bf1[31] = clamp_value(bf0[30] + bf0[31], stage_range[stage]);
|
||||
range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
|
||||
// stage 4
|
||||
stage++;
|
||||
|
|
@ -513,7 +464,7 @@ void av1_idct32_new(const int32_t *input, int32_t *output, int8_t cos_bit,
|
|||
bf1[29] = half_btf(-cospi[8], bf0[18], cospi[56], bf0[29], cos_bit);
|
||||
bf1[30] = half_btf(cospi[56], bf0[17], cospi[8], bf0[30], cos_bit);
|
||||
bf1[31] = bf0[31];
|
||||
range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
|
||||
// stage 5
|
||||
stage++;
|
||||
|
|
@ -551,7 +502,7 @@ void av1_idct32_new(const int32_t *input, int32_t *output, int8_t cos_bit,
|
|||
bf1[29] = clamp_value(-bf0[29] + bf0[30], stage_range[stage]);
|
||||
bf1[30] = clamp_value(bf0[29] + bf0[30], stage_range[stage]);
|
||||
bf1[31] = clamp_value(bf0[28] + bf0[31], stage_range[stage]);
|
||||
range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
|
||||
// stage 6
|
||||
stage++;
|
||||
|
|
@ -589,7 +540,7 @@ void av1_idct32_new(const int32_t *input, int32_t *output, int8_t cos_bit,
|
|||
bf1[29] = half_btf(cospi[48], bf0[18], cospi[16], bf0[29], cos_bit);
|
||||
bf1[30] = bf0[30];
|
||||
bf1[31] = bf0[31];
|
||||
range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
|
||||
// stage 7
|
||||
stage++;
|
||||
|
|
@ -627,7 +578,7 @@ void av1_idct32_new(const int32_t *input, int32_t *output, int8_t cos_bit,
|
|||
bf1[29] = clamp_value(bf0[26] + bf0[29], stage_range[stage]);
|
||||
bf1[30] = clamp_value(bf0[25] + bf0[30], stage_range[stage]);
|
||||
bf1[31] = clamp_value(bf0[24] + bf0[31], stage_range[stage]);
|
||||
range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
|
||||
// stage 8
|
||||
stage++;
|
||||
|
|
@ -665,7 +616,7 @@ void av1_idct32_new(const int32_t *input, int32_t *output, int8_t cos_bit,
|
|||
bf1[29] = bf0[29];
|
||||
bf1[30] = bf0[30];
|
||||
bf1[31] = bf0[31];
|
||||
range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
|
||||
// stage 9
|
||||
stage++;
|
||||
|
|
@ -760,7 +711,6 @@ void av1_iadst4_new(const int32_t *input, int32_t *output, int8_t cos_bit,
|
|||
output[1] = round_shift(x1, bit);
|
||||
output[2] = round_shift(x2, bit);
|
||||
output[3] = round_shift(x3, bit);
|
||||
range_check_buf(6, input, output, 4, stage_range[6]);
|
||||
}
|
||||
|
||||
void av1_iadst8_new(const int32_t *input, int32_t *output, int8_t cos_bit,
|
||||
|
|
@ -786,7 +736,7 @@ void av1_iadst8_new(const int32_t *input, int32_t *output, int8_t cos_bit,
|
|||
bf1[5] = input[4];
|
||||
bf1[6] = input[1];
|
||||
bf1[7] = input[6];
|
||||
range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
|
||||
// stage 2
|
||||
stage++;
|
||||
|
|
@ -800,7 +750,7 @@ void av1_iadst8_new(const int32_t *input, int32_t *output, int8_t cos_bit,
|
|||
bf1[5] = half_btf(cospi[28], bf0[4], -cospi[36], bf0[5], cos_bit);
|
||||
bf1[6] = half_btf(cospi[52], bf0[6], cospi[12], bf0[7], cos_bit);
|
||||
bf1[7] = half_btf(cospi[12], bf0[6], -cospi[52], bf0[7], cos_bit);
|
||||
range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
|
||||
// stage 3
|
||||
stage++;
|
||||
|
|
@ -814,7 +764,7 @@ void av1_iadst8_new(const int32_t *input, int32_t *output, int8_t cos_bit,
|
|||
bf1[5] = clamp_value(bf0[1] - bf0[5], stage_range[stage]);
|
||||
bf1[6] = clamp_value(bf0[2] - bf0[6], stage_range[stage]);
|
||||
bf1[7] = clamp_value(bf0[3] - bf0[7], stage_range[stage]);
|
||||
range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
|
||||
// stage 4
|
||||
stage++;
|
||||
|
|
@ -828,7 +778,7 @@ void av1_iadst8_new(const int32_t *input, int32_t *output, int8_t cos_bit,
|
|||
bf1[5] = half_btf(cospi[48], bf0[4], -cospi[16], bf0[5], cos_bit);
|
||||
bf1[6] = half_btf(-cospi[48], bf0[6], cospi[16], bf0[7], cos_bit);
|
||||
bf1[7] = half_btf(cospi[16], bf0[6], cospi[48], bf0[7], cos_bit);
|
||||
range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
|
||||
// stage 5
|
||||
stage++;
|
||||
|
|
@ -842,7 +792,7 @@ void av1_iadst8_new(const int32_t *input, int32_t *output, int8_t cos_bit,
|
|||
bf1[5] = clamp_value(bf0[5] + bf0[7], stage_range[stage]);
|
||||
bf1[6] = clamp_value(bf0[4] - bf0[6], stage_range[stage]);
|
||||
bf1[7] = clamp_value(bf0[5] - bf0[7], stage_range[stage]);
|
||||
range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
|
||||
// stage 6
|
||||
stage++;
|
||||
|
|
@ -856,7 +806,7 @@ void av1_iadst8_new(const int32_t *input, int32_t *output, int8_t cos_bit,
|
|||
bf1[5] = bf0[5];
|
||||
bf1[6] = half_btf(cospi[32], bf0[6], cospi[32], bf0[7], cos_bit);
|
||||
bf1[7] = half_btf(cospi[32], bf0[6], -cospi[32], bf0[7], cos_bit);
|
||||
range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
|
||||
// stage 7
|
||||
stage++;
|
||||
|
|
@ -903,7 +853,7 @@ void av1_iadst16_new(const int32_t *input, int32_t *output, int8_t cos_bit,
|
|||
bf1[13] = input[12];
|
||||
bf1[14] = input[1];
|
||||
bf1[15] = input[14];
|
||||
range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
|
||||
// stage 2
|
||||
stage++;
|
||||
|
|
@ -925,7 +875,7 @@ void av1_iadst16_new(const int32_t *input, int32_t *output, int8_t cos_bit,
|
|||
bf1[13] = half_btf(cospi[14], bf0[12], -cospi[50], bf0[13], cos_bit);
|
||||
bf1[14] = half_btf(cospi[58], bf0[14], cospi[6], bf0[15], cos_bit);
|
||||
bf1[15] = half_btf(cospi[6], bf0[14], -cospi[58], bf0[15], cos_bit);
|
||||
range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
|
||||
// stage 3
|
||||
stage++;
|
||||
|
|
@ -947,7 +897,7 @@ void av1_iadst16_new(const int32_t *input, int32_t *output, int8_t cos_bit,
|
|||
bf1[13] = clamp_value(bf0[5] - bf0[13], stage_range[stage]);
|
||||
bf1[14] = clamp_value(bf0[6] - bf0[14], stage_range[stage]);
|
||||
bf1[15] = clamp_value(bf0[7] - bf0[15], stage_range[stage]);
|
||||
range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
|
||||
// stage 4
|
||||
stage++;
|
||||
|
|
@ -969,7 +919,7 @@ void av1_iadst16_new(const int32_t *input, int32_t *output, int8_t cos_bit,
|
|||
bf1[13] = half_btf(cospi[8], bf0[12], cospi[56], bf0[13], cos_bit);
|
||||
bf1[14] = half_btf(-cospi[24], bf0[14], cospi[40], bf0[15], cos_bit);
|
||||
bf1[15] = half_btf(cospi[40], bf0[14], cospi[24], bf0[15], cos_bit);
|
||||
range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
|
||||
// stage 5
|
||||
stage++;
|
||||
|
|
@ -991,7 +941,7 @@ void av1_iadst16_new(const int32_t *input, int32_t *output, int8_t cos_bit,
|
|||
bf1[13] = clamp_value(bf0[9] - bf0[13], stage_range[stage]);
|
||||
bf1[14] = clamp_value(bf0[10] - bf0[14], stage_range[stage]);
|
||||
bf1[15] = clamp_value(bf0[11] - bf0[15], stage_range[stage]);
|
||||
range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
|
||||
// stage 6
|
||||
stage++;
|
||||
|
|
@ -1013,7 +963,7 @@ void av1_iadst16_new(const int32_t *input, int32_t *output, int8_t cos_bit,
|
|||
bf1[13] = half_btf(cospi[48], bf0[12], -cospi[16], bf0[13], cos_bit);
|
||||
bf1[14] = half_btf(-cospi[48], bf0[14], cospi[16], bf0[15], cos_bit);
|
||||
bf1[15] = half_btf(cospi[16], bf0[14], cospi[48], bf0[15], cos_bit);
|
||||
range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
|
||||
// stage 7
|
||||
stage++;
|
||||
|
|
@ -1035,7 +985,7 @@ void av1_iadst16_new(const int32_t *input, int32_t *output, int8_t cos_bit,
|
|||
bf1[13] = clamp_value(bf0[13] + bf0[15], stage_range[stage]);
|
||||
bf1[14] = clamp_value(bf0[12] - bf0[14], stage_range[stage]);
|
||||
bf1[15] = clamp_value(bf0[13] - bf0[15], stage_range[stage]);
|
||||
range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
|
||||
// stage 8
|
||||
stage++;
|
||||
|
|
@ -1057,7 +1007,7 @@ void av1_iadst16_new(const int32_t *input, int32_t *output, int8_t cos_bit,
|
|||
bf1[13] = bf0[13];
|
||||
bf1[14] = half_btf(cospi[32], bf0[14], cospi[32], bf0[15], cos_bit);
|
||||
bf1[15] = half_btf(cospi[32], bf0[14], -cospi[32], bf0[15], cos_bit);
|
||||
range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
|
||||
// stage 9
|
||||
stage++;
|
||||
|
|
@ -1193,7 +1143,7 @@ void av1_idct64_new(const int32_t *input, int32_t *output, int8_t cos_bit,
|
|||
bf1[61] = input[47];
|
||||
bf1[62] = input[31];
|
||||
bf1[63] = input[63];
|
||||
range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
|
||||
// stage 2
|
||||
stage++;
|
||||
|
|
@ -1263,7 +1213,7 @@ void av1_idct64_new(const int32_t *input, int32_t *output, int8_t cos_bit,
|
|||
bf1[61] = half_btf(cospi[17], bf0[34], cospi[47], bf0[61], cos_bit);
|
||||
bf1[62] = half_btf(cospi[33], bf0[33], cospi[31], bf0[62], cos_bit);
|
||||
bf1[63] = half_btf(cospi[1], bf0[32], cospi[63], bf0[63], cos_bit);
|
||||
range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
|
||||
// stage 3
|
||||
stage++;
|
||||
|
|
@ -1333,7 +1283,7 @@ void av1_idct64_new(const int32_t *input, int32_t *output, int8_t cos_bit,
|
|||
bf1[61] = clamp_value(bf0[60] - bf0[61], stage_range[stage]);
|
||||
bf1[62] = clamp_value(-bf0[62] + bf0[63], stage_range[stage]);
|
||||
bf1[63] = clamp_value(bf0[62] + bf0[63], stage_range[stage]);
|
||||
range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
|
||||
// stage 4
|
||||
stage++;
|
||||
|
|
@ -1403,7 +1353,7 @@ void av1_idct64_new(const int32_t *input, int32_t *output, int8_t cos_bit,
|
|||
bf1[61] = half_btf(-cospi[4], bf0[34], cospi[60], bf0[61], cos_bit);
|
||||
bf1[62] = half_btf(cospi[60], bf0[33], cospi[4], bf0[62], cos_bit);
|
||||
bf1[63] = bf0[63];
|
||||
range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
|
||||
// stage 5
|
||||
stage++;
|
||||
|
|
@ -1473,7 +1423,7 @@ void av1_idct64_new(const int32_t *input, int32_t *output, int8_t cos_bit,
|
|||
bf1[61] = clamp_value(-bf0[61] + bf0[62], stage_range[stage]);
|
||||
bf1[62] = clamp_value(bf0[61] + bf0[62], stage_range[stage]);
|
||||
bf1[63] = clamp_value(bf0[60] + bf0[63], stage_range[stage]);
|
||||
range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
|
||||
// stage 6
|
||||
stage++;
|
||||
|
|
@ -1543,7 +1493,7 @@ void av1_idct64_new(const int32_t *input, int32_t *output, int8_t cos_bit,
|
|||
bf1[61] = half_btf(cospi[56], bf0[34], cospi[8], bf0[61], cos_bit);
|
||||
bf1[62] = bf0[62];
|
||||
bf1[63] = bf0[63];
|
||||
range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
|
||||
// stage 7
|
||||
stage++;
|
||||
|
|
@ -1613,7 +1563,7 @@ void av1_idct64_new(const int32_t *input, int32_t *output, int8_t cos_bit,
|
|||
bf1[61] = clamp_value(bf0[58] + bf0[61], stage_range[stage]);
|
||||
bf1[62] = clamp_value(bf0[57] + bf0[62], stage_range[stage]);
|
||||
bf1[63] = clamp_value(bf0[56] + bf0[63], stage_range[stage]);
|
||||
range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
|
||||
// stage 8
|
||||
stage++;
|
||||
|
|
@ -1683,7 +1633,7 @@ void av1_idct64_new(const int32_t *input, int32_t *output, int8_t cos_bit,
|
|||
bf1[61] = bf0[61];
|
||||
bf1[62] = bf0[62];
|
||||
bf1[63] = bf0[63];
|
||||
range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
|
||||
// stage 9
|
||||
stage++;
|
||||
|
|
@ -1753,7 +1703,7 @@ void av1_idct64_new(const int32_t *input, int32_t *output, int8_t cos_bit,
|
|||
bf1[61] = clamp_value(bf0[50] + bf0[61], stage_range[stage]);
|
||||
bf1[62] = clamp_value(bf0[49] + bf0[62], stage_range[stage]);
|
||||
bf1[63] = clamp_value(bf0[48] + bf0[63], stage_range[stage]);
|
||||
range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
|
||||
// stage 10
|
||||
stage++;
|
||||
|
|
@ -1823,7 +1773,7 @@ void av1_idct64_new(const int32_t *input, int32_t *output, int8_t cos_bit,
|
|||
bf1[61] = bf0[61];
|
||||
bf1[62] = bf0[62];
|
||||
bf1[63] = bf0[63];
|
||||
range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
|
||||
|
||||
// stage 11
|
||||
stage++;
|
||||
|
|
|
|||
6
third_party/aom/av1/common/av1_inv_txfm1d.h
vendored
6
third_party/aom/av1/common/av1_inv_txfm1d.h
vendored
|
|
@ -9,8 +9,8 @@
|
|||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#ifndef AV1_INV_TXFM1D_H_
|
||||
#define AV1_INV_TXFM1D_H_
|
||||
#ifndef AOM_AV1_COMMON_AV1_INV_TXFM1D_H_
|
||||
#define AOM_AV1_COMMON_AV1_INV_TXFM1D_H_
|
||||
|
||||
#include "av1/common/av1_txfm.h"
|
||||
|
||||
|
|
@ -58,4 +58,4 @@ void av1_iidentity32_c(const int32_t *input, int32_t *output, int8_t cos_bit,
|
|||
}
|
||||
#endif
|
||||
|
||||
#endif // AV1_INV_TXFM1D_H_
|
||||
#endif // AOM_AV1_COMMON_AV1_INV_TXFM1D_H_
|
||||
|
|
|
|||
|
|
@ -9,8 +9,8 @@
|
|||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#ifndef AV1_INV_TXFM2D_CFG_H_
|
||||
#define AV1_INV_TXFM2D_CFG_H_
|
||||
#ifndef AOM_AV1_COMMON_AV1_INV_TXFM1D_CFG_H_
|
||||
#define AOM_AV1_COMMON_AV1_INV_TXFM1D_CFG_H_
|
||||
#include "av1/common/av1_inv_txfm1d.h"
|
||||
|
||||
// sum of fwd_shift_##
|
||||
|
|
@ -44,4 +44,4 @@ extern const int8_t *inv_txfm_shift_ls[TX_SIZES_ALL];
|
|||
extern const int8_t inv_cos_bit_col[5 /*row*/][5 /*col*/];
|
||||
extern const int8_t inv_cos_bit_row[5 /*row*/][5 /*col*/];
|
||||
|
||||
#endif // AV1_INV_TXFM2D_CFG_H_
|
||||
#endif // AOM_AV1_COMMON_AV1_INV_TXFM1D_CFG_H_
|
||||
|
|
|
|||
943
third_party/aom/av1/common/av1_loopfilter.c
vendored
943
third_party/aom/av1/common/av1_loopfilter.c
vendored
File diff suppressed because it is too large
Load diff
120
third_party/aom/av1/common/av1_loopfilter.h
vendored
120
third_party/aom/av1/common/av1_loopfilter.h
vendored
|
|
@ -9,8 +9,8 @@
|
|||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#ifndef AV1_COMMON_LOOPFILTER_H_
|
||||
#define AV1_COMMON_LOOPFILTER_H_
|
||||
#ifndef AOM_AV1_COMMON_AV1_LOOPFILTER_H_
|
||||
#define AOM_AV1_COMMON_AV1_LOOPFILTER_H_
|
||||
|
||||
#include "config/aom_config.h"
|
||||
|
||||
|
|
@ -60,51 +60,20 @@ typedef struct {
|
|||
uint8_t lfl_y_hor[MI_SIZE_64X64][MI_SIZE_64X64];
|
||||
uint8_t lfl_y_ver[MI_SIZE_64X64][MI_SIZE_64X64];
|
||||
|
||||
// U plane vertical edge and horizontal edge filter level
|
||||
uint8_t lfl_u_hor[MI_SIZE_64X64][MI_SIZE_64X64];
|
||||
uint8_t lfl_u_ver[MI_SIZE_64X64][MI_SIZE_64X64];
|
||||
// U plane filter level
|
||||
uint8_t lfl_u[MI_SIZE_64X64][MI_SIZE_64X64];
|
||||
|
||||
// V plane vertical edge and horizontal edge filter level
|
||||
uint8_t lfl_v_hor[MI_SIZE_64X64][MI_SIZE_64X64];
|
||||
uint8_t lfl_v_ver[MI_SIZE_64X64][MI_SIZE_64X64];
|
||||
// V plane filter level
|
||||
uint8_t lfl_v[MI_SIZE_64X64][MI_SIZE_64X64];
|
||||
|
||||
// other info
|
||||
FilterMask skip;
|
||||
FilterMask is_vert_border;
|
||||
FilterMask is_horz_border;
|
||||
// Y or UV planes, 5 tx sizes: 4x4, 8x8, 16x16, 32x32, 64x64
|
||||
FilterMask tx_size_ver[2][5];
|
||||
FilterMask tx_size_hor[2][5];
|
||||
} LoopFilterMask;
|
||||
|
||||
// To determine whether to apply loop filtering at one transform block edge,
|
||||
// we need information of the neighboring transform block. Specifically,
|
||||
// in determining a vertical edge, we need the information of the tx block
|
||||
// to its left. For a horizontal edge, we need info of the tx block above it.
|
||||
// Thus, we need to record info of right column and bottom row of tx blocks.
|
||||
// We record the information of the neighboring superblock, when bitmask
|
||||
// building for a superblock is finished. And it will be used for next
|
||||
// superblock bitmask building.
|
||||
// Information includes:
|
||||
// ------------------------------------------------------------
|
||||
// MI_SIZE_64X64
|
||||
// Y tx_size above |--------------|
|
||||
// Y tx_size left |--------------|
|
||||
// UV tx_size above |--------------|
|
||||
// UV tx_size left |--------------|
|
||||
// Y level above |--------------|
|
||||
// Y level left |--------------|
|
||||
// U level above |--------------|
|
||||
// U level left |--------------|
|
||||
// V level above |--------------|
|
||||
// V level left |--------------|
|
||||
// skip |--------------|
|
||||
// ------------------------------------------------------------
|
||||
typedef struct {
|
||||
TX_SIZE tx_size_y_above[MI_SIZE_64X64];
|
||||
TX_SIZE tx_size_y_left[MI_SIZE_64X64];
|
||||
TX_SIZE tx_size_uv_above[MI_SIZE_64X64];
|
||||
TX_SIZE tx_size_uv_left[MI_SIZE_64X64];
|
||||
uint8_t y_level_above[MI_SIZE_64X64];
|
||||
uint8_t y_level_left[MI_SIZE_64X64];
|
||||
uint8_t u_level_above[MI_SIZE_64X64];
|
||||
uint8_t u_level_left[MI_SIZE_64X64];
|
||||
uint8_t v_level_above[MI_SIZE_64X64];
|
||||
uint8_t v_level_left[MI_SIZE_64X64];
|
||||
uint8_t skip[MI_SIZE_64X64];
|
||||
} LpfSuperblockInfo;
|
||||
#endif // LOOP_FILTER_BITMASK
|
||||
|
||||
struct loopfilter {
|
||||
|
|
@ -130,7 +99,6 @@ struct loopfilter {
|
|||
LoopFilterMask *lfm;
|
||||
size_t lfm_num;
|
||||
int lfm_stride;
|
||||
LpfSuperblockInfo neighbor_sb_lpf_info;
|
||||
#endif // LOOP_FILTER_BITMASK
|
||||
};
|
||||
|
||||
|
|
@ -157,9 +125,15 @@ void av1_loop_filter_init(struct AV1Common *cm);
|
|||
void av1_loop_filter_frame_init(struct AV1Common *cm, int plane_start,
|
||||
int plane_end);
|
||||
|
||||
#if LOOP_FILTER_BITMASK
|
||||
void av1_loop_filter_frame(YV12_BUFFER_CONFIG *frame, struct AV1Common *cm,
|
||||
struct macroblockd *mbd, int is_decoding,
|
||||
int plane_start, int plane_end, int partial_frame);
|
||||
#else
|
||||
void av1_loop_filter_frame(YV12_BUFFER_CONFIG *frame, struct AV1Common *cm,
|
||||
struct macroblockd *mbd, int plane_start,
|
||||
int plane_end, int partial_frame);
|
||||
#endif
|
||||
|
||||
void av1_filter_block_plane_vert(const struct AV1Common *const cm,
|
||||
const MACROBLOCKD *const xd, const int plane,
|
||||
|
|
@ -180,6 +154,9 @@ typedef struct LoopFilterWorkerData {
|
|||
MACROBLOCKD *xd;
|
||||
} LFWorkerData;
|
||||
|
||||
uint8_t get_filter_level(const struct AV1Common *cm,
|
||||
const loop_filter_info_n *lfi_n, const int dir_idx,
|
||||
int plane, const MB_MODE_INFO *mbmi);
|
||||
#if LOOP_FILTER_BITMASK
|
||||
void av1_setup_bitmask(struct AV1Common *const cm, int mi_row, int mi_col,
|
||||
int plane, int subsampling_x, int subsampling_y,
|
||||
|
|
@ -192,10 +169,59 @@ void av1_filter_block_plane_ver(struct AV1Common *const cm,
|
|||
void av1_filter_block_plane_hor(struct AV1Common *const cm,
|
||||
struct macroblockd_plane *const plane, int pl,
|
||||
int mi_row, int mi_col);
|
||||
LoopFilterMask *get_loop_filter_mask(const struct AV1Common *const cm,
|
||||
int mi_row, int mi_col);
|
||||
int get_index_shift(int mi_col, int mi_row, int *index);
|
||||
|
||||
static const FilterMask left_txform_mask[TX_SIZES] = {
|
||||
{ { 0x0000000000000001ULL, // TX_4X4,
|
||||
0x0000000000000000ULL, 0x0000000000000000ULL, 0x0000000000000000ULL } },
|
||||
|
||||
{ { 0x0000000000010001ULL, // TX_8X8,
|
||||
0x0000000000000000ULL, 0x0000000000000000ULL, 0x0000000000000000ULL } },
|
||||
|
||||
{ { 0x0001000100010001ULL, // TX_16X16,
|
||||
0x0000000000000000ULL, 0x0000000000000000ULL, 0x0000000000000000ULL } },
|
||||
|
||||
{ { 0x0001000100010001ULL, // TX_32X32,
|
||||
0x0001000100010001ULL, 0x0000000000000000ULL, 0x0000000000000000ULL } },
|
||||
|
||||
{ { 0x0001000100010001ULL, // TX_64X64,
|
||||
0x0001000100010001ULL, 0x0001000100010001ULL, 0x0001000100010001ULL } },
|
||||
};
|
||||
|
||||
static const uint64_t above_txform_mask[2][TX_SIZES] = {
|
||||
{
|
||||
0x0000000000000001ULL, // TX_4X4
|
||||
0x0000000000000003ULL, // TX_8X8
|
||||
0x000000000000000fULL, // TX_16X16
|
||||
0x00000000000000ffULL, // TX_32X32
|
||||
0x000000000000ffffULL, // TX_64X64
|
||||
},
|
||||
{
|
||||
0x0000000000000001ULL, // TX_4X4
|
||||
0x0000000000000005ULL, // TX_8X8
|
||||
0x0000000000000055ULL, // TX_16X16
|
||||
0x0000000000005555ULL, // TX_32X32
|
||||
0x0000000055555555ULL, // TX_64X64
|
||||
},
|
||||
};
|
||||
|
||||
extern const int mask_id_table_tx_4x4[BLOCK_SIZES_ALL];
|
||||
|
||||
extern const int mask_id_table_tx_8x8[BLOCK_SIZES_ALL];
|
||||
|
||||
extern const int mask_id_table_tx_16x16[BLOCK_SIZES_ALL];
|
||||
|
||||
extern const int mask_id_table_tx_32x32[BLOCK_SIZES_ALL];
|
||||
|
||||
extern const FilterMask left_mask_univariant_reordered[67];
|
||||
|
||||
extern const FilterMask above_mask_univariant_reordered[67];
|
||||
#endif
|
||||
|
||||
#ifdef __cplusplus
|
||||
} // extern "C"
|
||||
#endif
|
||||
|
||||
#endif // AV1_COMMON_LOOPFILTER_H_
|
||||
#endif // AOM_AV1_COMMON_AV1_LOOPFILTER_H_
|
||||
|
|
|
|||
46
third_party/aom/av1/common/av1_rtcd_defs.pl
vendored
46
third_party/aom/av1/common/av1_rtcd_defs.pl
vendored
|
|
@ -76,12 +76,12 @@ specialize qw/av1_wiener_convolve_add_src sse2 avx2 neon/;
|
|||
specialize qw/av1_highbd_wiener_convolve_add_src ssse3/;
|
||||
specialize qw/av1_highbd_wiener_convolve_add_src avx2/;
|
||||
|
||||
|
||||
# directional intra predictor functions
|
||||
add_proto qw/void av1_dr_prediction_z1/, "uint8_t *dst, ptrdiff_t stride, int bw, int bh, const uint8_t *above, const uint8_t *left, int upsample_above, int dx, int dy";
|
||||
add_proto qw/void av1_dr_prediction_z2/, "uint8_t *dst, ptrdiff_t stride, int bw, int bh, const uint8_t *above, const uint8_t *left, int upsample_above, int upsample_left, int dx, int dy";
|
||||
add_proto qw/void av1_dr_prediction_z3/, "uint8_t *dst, ptrdiff_t stride, int bw, int bh, const uint8_t *above, const uint8_t *left, int upsample_left, int dx, int dy";
|
||||
|
||||
|
||||
# FILTER_INTRA predictor functions
|
||||
add_proto qw/void av1_filter_intra_predictor/, "uint8_t *dst, ptrdiff_t stride, TX_SIZE tx_size, const uint8_t *above, const uint8_t *left, int mode";
|
||||
specialize qw/av1_filter_intra_predictor sse4_1/;
|
||||
|
|
@ -108,6 +108,22 @@ specialize qw/av1_highbd_convolve8_vert/, "$sse2_x86_64";
|
|||
add_proto qw/void av1_inv_txfm_add/, "const tran_low_t *dqcoeff, uint8_t *dst, int stride, const TxfmParam *txfm_param";
|
||||
specialize qw/av1_inv_txfm_add ssse3 avx2 neon/;
|
||||
|
||||
add_proto qw/void av1_highbd_inv_txfm_add/, "const tran_low_t *dqcoeff, uint8_t *dst, int stride, const TxfmParam *txfm_param";
|
||||
specialize qw/av1_highbd_inv_txfm_add sse4_1 avx2/;
|
||||
|
||||
add_proto qw/void av1_highbd_inv_txfm_add_4x4/, "const tran_low_t *dqcoeff, uint8_t *dst, int stride, const TxfmParam *txfm_param";
|
||||
specialize qw/av1_highbd_inv_txfm_add_4x4 sse4_1/;
|
||||
add_proto qw/void av1_highbd_inv_txfm_add_8x8/, "const tran_low_t *dqcoeff, uint8_t *dst, int stride, const TxfmParam *txfm_param";
|
||||
specialize qw/av1_highbd_inv_txfm_add_8x8 sse4_1/;
|
||||
add_proto qw/void av1_highbd_inv_txfm_add_16x8/, "const tran_low_t *dqcoeff, uint8_t *dst, int stride, const TxfmParam *txfm_param";
|
||||
specialize qw/av1_highbd_inv_txfm_add_16x8 sse4_1/;
|
||||
add_proto qw/void av1_highbd_inv_txfm_add_8x16/, "const tran_low_t *dqcoeff, uint8_t *dst, int stride, const TxfmParam *txfm_param";
|
||||
specialize qw/av1_highbd_inv_txfm_add_8x16 sse4_1/;
|
||||
add_proto qw/void av1_highbd_inv_txfm_add_16x16/, "const tran_low_t *dqcoeff, uint8_t *dst, int stride, const TxfmParam *txfm_param";
|
||||
specialize qw/av1_highbd_inv_txfm_add_16x16 sse4_1/;
|
||||
add_proto qw/void av1_highbd_inv_txfm_add_32x32/, "const tran_low_t *dqcoeff, uint8_t *dst, int stride, const TxfmParam *txfm_param";
|
||||
specialize qw/av1_highbd_inv_txfm_add_32x32 sse4_1 avx2/;
|
||||
|
||||
add_proto qw/void av1_highbd_iwht4x4_1_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, int bd";
|
||||
add_proto qw/void av1_highbd_iwht4x4_16_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, int bd";
|
||||
|
||||
|
|
@ -122,9 +138,7 @@ specialize qw/av1_inv_txfm2d_add_4x4 sse4_1/;
|
|||
add_proto qw/void av1_inv_txfm2d_add_8x8/, "const int32_t *input, uint16_t *output, int stride, TX_TYPE tx_type, int bd";
|
||||
specialize qw/av1_inv_txfm2d_add_8x8 sse4_1/;
|
||||
add_proto qw/void av1_inv_txfm2d_add_16x16/, "const int32_t *input, uint16_t *output, int stride, TX_TYPE tx_type, int bd";
|
||||
specialize qw/av1_inv_txfm2d_add_16x16 sse4_1/;
|
||||
add_proto qw/void av1_inv_txfm2d_add_32x32/, "const int32_t *input, uint16_t *output, int stride, TX_TYPE tx_type, int bd";
|
||||
specialize qw/av1_inv_txfm2d_add_32x32 avx2/;
|
||||
|
||||
add_proto qw/void av1_inv_txfm2d_add_64x64/, "const int32_t *input, uint16_t *output, int stride, TX_TYPE tx_type, int bd";
|
||||
add_proto qw/void av1_inv_txfm2d_add_32x64/, "const int32_t *input, uint16_t *output, int stride, TX_TYPE tx_type, int bd";
|
||||
|
|
@ -132,8 +146,6 @@ add_proto qw/void av1_inv_txfm2d_add_64x32/, "const int32_t *input, uint16_t *ou
|
|||
add_proto qw/void av1_inv_txfm2d_add_16x64/, "const int32_t *input, uint16_t *output, int stride, TX_TYPE tx_type, int bd";
|
||||
add_proto qw/void av1_inv_txfm2d_add_64x16/, "const int32_t *input, uint16_t *output, int stride, TX_TYPE tx_type, int bd";
|
||||
|
||||
specialize qw/av1_inv_txfm2d_add_64x64 sse4_1/;
|
||||
|
||||
add_proto qw/void av1_inv_txfm2d_add_4x16/, "const int32_t *input, uint16_t *output, int stride, TX_TYPE tx_type, int bd";
|
||||
add_proto qw/void av1_inv_txfm2d_add_16x4/, "const int32_t *input, uint16_t *output, int stride, TX_TYPE tx_type, int bd";
|
||||
add_proto qw/void av1_inv_txfm2d_add_8x32/, "const int32_t *input, uint16_t *output, int stride, TX_TYPE tx_type, int bd";
|
||||
|
|
@ -146,13 +158,13 @@ add_proto qw/void av1_highbd_dr_prediction_z3/, "uint16_t *dst, ptrdiff_t stride
|
|||
|
||||
# build compound seg mask functions
|
||||
add_proto qw/void av1_build_compound_diffwtd_mask/, "uint8_t *mask, DIFFWTD_MASK_TYPE mask_type, const uint8_t *src0, int src0_stride, const uint8_t *src1, int src1_stride, int h, int w";
|
||||
specialize qw/av1_build_compound_diffwtd_mask sse4_1/;
|
||||
specialize qw/av1_build_compound_diffwtd_mask sse4_1 avx2/;
|
||||
|
||||
add_proto qw/void av1_build_compound_diffwtd_mask_highbd/, "uint8_t *mask, DIFFWTD_MASK_TYPE mask_type, const uint8_t *src0, int src0_stride, const uint8_t *src1, int src1_stride, int h, int w, int bd";
|
||||
specialize qw/av1_build_compound_diffwtd_mask_highbd ssse3 avx2/;
|
||||
|
||||
add_proto qw/void av1_build_compound_diffwtd_mask_d16/, "uint8_t *mask, DIFFWTD_MASK_TYPE mask_type, const CONV_BUF_TYPE *src0, int src0_stride, const CONV_BUF_TYPE *src1, int src1_stride, int h, int w, ConvolveParams *conv_params, int bd";
|
||||
specialize qw/av1_build_compound_diffwtd_mask_d16 sse4_1 neon/;
|
||||
specialize qw/av1_build_compound_diffwtd_mask_d16 sse4_1 avx2 neon/;
|
||||
|
||||
#
|
||||
# Encoder functions below this point.
|
||||
|
|
@ -186,7 +198,9 @@ if (aom_config("CONFIG_AV1_ENCODER") eq "yes") {
|
|||
add_proto qw/void av1_fwd_txfm2d_4x8/, "const int16_t *input, int32_t *output, int stride, TX_TYPE tx_type, int bd";
|
||||
add_proto qw/void av1_fwd_txfm2d_8x4/, "const int16_t *input, int32_t *output, int stride, TX_TYPE tx_type, int bd";
|
||||
add_proto qw/void av1_fwd_txfm2d_8x16/, "const int16_t *input, int32_t *output, int stride, TX_TYPE tx_type, int bd";
|
||||
specialize qw/av1_fwd_txfm2d_8x16 sse4_1/;
|
||||
add_proto qw/void av1_fwd_txfm2d_16x8/, "const int16_t *input, int32_t *output, int stride, TX_TYPE tx_type, int bd";
|
||||
specialize qw/av1_fwd_txfm2d_16x8 sse4_1/;
|
||||
add_proto qw/void av1_fwd_txfm2d_16x32/, "const int16_t *input, int32_t *output, int stride, TX_TYPE tx_type, int bd";
|
||||
add_proto qw/void av1_fwd_txfm2d_32x16/, "const int16_t *input, int32_t *output, int stride, TX_TYPE tx_type, int bd";
|
||||
add_proto qw/void av1_fwd_txfm2d_4x16/, "const int16_t *input, int32_t *output, int stride, TX_TYPE tx_type, int bd";
|
||||
|
|
@ -203,6 +217,7 @@ if (aom_config("CONFIG_AV1_ENCODER") eq "yes") {
|
|||
specialize qw/av1_fwd_txfm2d_32x32 sse4_1/;
|
||||
|
||||
add_proto qw/void av1_fwd_txfm2d_64x64/, "const int16_t *input, int32_t *output, int stride, TX_TYPE tx_type, int bd";
|
||||
specialize qw/av1_fwd_txfm2d_64x64 sse4_1/;
|
||||
add_proto qw/void av1_fwd_txfm2d_32x64/, "const int16_t *input, int32_t *output, int stride, TX_TYPE tx_type, int bd";
|
||||
add_proto qw/void av1_fwd_txfm2d_64x32/, "const int16_t *input, int32_t *output, int stride, TX_TYPE tx_type, int bd";
|
||||
add_proto qw/void av1_fwd_txfm2d_16x64/, "const int16_t *input, int32_t *output, int stride, TX_TYPE tx_type, int bd";
|
||||
|
|
@ -218,7 +233,7 @@ if (aom_config("CONFIG_AV1_ENCODER") eq "yes") {
|
|||
add_proto qw/void av1_temporal_filter_apply/, "uint8_t *frame1, unsigned int stride, uint8_t *frame2, unsigned int block_width, unsigned int block_height, int strength, int filter_weight, unsigned int *accumulator, uint16_t *count";
|
||||
specialize qw/av1_temporal_filter_apply sse2 msa/;
|
||||
|
||||
add_proto qw/void av1_quantize_b/, "const tran_low_t *coeff_ptr, intptr_t n_coeffs, int skip_block, const int16_t *zbin_ptr, const int16_t *round_ptr, const int16_t *quant_ptr, const int16_t *quant_shift_ptr, tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr, const int16_t *dequant_ptr, uint16_t *eob_ptr, const int16_t *scan, const int16_t *iscan, const qm_val_t * qm_ptr, const qm_val_t * iqm_ptr, int log_scale";
|
||||
add_proto qw/void av1_quantize_b/, "const tran_low_t *coeff_ptr, intptr_t n_coeffs, const int16_t *zbin_ptr, const int16_t *round_ptr, const int16_t *quant_ptr, const int16_t *quant_shift_ptr, tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr, const int16_t *dequant_ptr, uint16_t *eob_ptr, const int16_t *scan, const int16_t *iscan, const qm_val_t * qm_ptr, const qm_val_t * iqm_ptr, int log_scale";
|
||||
|
||||
# ENCODEMB INVOKE
|
||||
|
||||
|
|
@ -238,7 +253,7 @@ if (aom_config("CONFIG_AV1_ENCODER") eq "yes") {
|
|||
add_proto qw/void av1_get_nz_map_contexts/, "const uint8_t *const levels, const int16_t *const scan, const uint16_t eob, const TX_SIZE tx_size, const TX_CLASS tx_class, int8_t *const coeff_contexts";
|
||||
specialize qw/av1_get_nz_map_contexts sse2/;
|
||||
add_proto qw/void av1_txb_init_levels/, "const tran_low_t *const coeff, const int width, const int height, uint8_t *const levels";
|
||||
specialize qw/av1_txb_init_levels sse4_1/;
|
||||
specialize qw/av1_txb_init_levels sse4_1 avx2/;
|
||||
|
||||
add_proto qw/uint64_t av1_wedge_sse_from_residuals/, "const int16_t *r1, const int16_t *d, const uint8_t *m, int N";
|
||||
specialize qw/av1_wedge_sse_from_residuals sse2 avx2/;
|
||||
|
|
@ -251,6 +266,11 @@ if (aom_config("CONFIG_AV1_ENCODER") eq "yes") {
|
|||
add_proto qw/uint32_t av1_get_crc32c_value/, "void *crc_calculator, uint8_t *p, int length";
|
||||
specialize qw/av1_get_crc32c_value sse4_2/;
|
||||
|
||||
add_proto qw/void av1_compute_stats/, "int wiener_win, const uint8_t *dgd8, const uint8_t *src8, int h_start, int h_end, int v_start, int v_end, int dgd_stride, int src_stride, double *M, double *H";
|
||||
specialize qw/av1_compute_stats sse4_1 avx2/;
|
||||
|
||||
add_proto qw/int64_t av1_lowbd_pixel_proj_error/, " const uint8_t *src8, int width, int height, int src_stride, const uint8_t *dat8, int dat_stride, int32_t *flt0, int flt0_stride, int32_t *flt1, int flt1_stride, int xq[2], const sgr_params_type *params";
|
||||
specialize qw/av1_lowbd_pixel_proj_error sse4_1 avx2/;
|
||||
}
|
||||
# end encoder functions
|
||||
|
||||
|
|
@ -275,7 +295,7 @@ if ($opts{config} !~ /libs-x86-win32-vs.*/) {
|
|||
# WARPED_MOTION / GLOBAL_MOTION functions
|
||||
|
||||
add_proto qw/void av1_warp_affine/, "const int32_t *mat, const uint8_t *ref, int width, int height, int stride, uint8_t *pred, int p_col, int p_row, int p_width, int p_height, int p_stride, int subsampling_x, int subsampling_y, ConvolveParams *conv_params, int16_t alpha, int16_t beta, int16_t gamma, int16_t delta";
|
||||
specialize qw/av1_warp_affine sse4_1/;
|
||||
specialize qw/av1_warp_affine sse4_1 neon/;
|
||||
|
||||
add_proto qw/void av1_highbd_warp_affine/, "const int32_t *mat, const uint16_t *ref, int width, int height, int stride, uint16_t *pred, int p_col, int p_row, int p_width, int p_height, int p_stride, int subsampling_x, int subsampling_y, int bd, ConvolveParams *conv_params, int16_t alpha, int16_t beta, int16_t gamma, int16_t delta";
|
||||
specialize qw/av1_highbd_warp_affine sse4_1/;
|
||||
|
|
@ -290,9 +310,9 @@ if (aom_config("CONFIG_AV1_ENCODER") eq "yes") {
|
|||
add_proto qw/void apply_selfguided_restoration/, "const uint8_t *dat, int width, int height, int stride, int eps, const int *xqd, uint8_t *dst, int dst_stride, int32_t *tmpbuf, int bit_depth, int highbd";
|
||||
specialize qw/apply_selfguided_restoration sse4_1 avx2 neon/;
|
||||
|
||||
add_proto qw/void av1_selfguided_restoration/, "const uint8_t *dgd8, int width, int height,
|
||||
int dgd_stride, int32_t *flt0, int32_t *flt1, int flt_stride,
|
||||
int sgr_params_idx, int bit_depth, int highbd";
|
||||
add_proto qw/int av1_selfguided_restoration/, "const uint8_t *dgd8, int width, int height,
|
||||
int dgd_stride, int32_t *flt0, int32_t *flt1, int flt_stride,
|
||||
int sgr_params_idx, int bit_depth, int highbd";
|
||||
specialize qw/av1_selfguided_restoration sse4_1 avx2 neon/;
|
||||
|
||||
# CONVOLVE_ROUND/COMPOUND_ROUND functions
|
||||
|
|
|
|||
50
third_party/aom/av1/common/av1_txfm.c
vendored
50
third_party/aom/av1/common/av1_txfm.c
vendored
|
|
@ -108,3 +108,53 @@ const int8_t av1_txfm_stage_num_list[TXFM_TYPES] = {
|
|||
1, // TXFM_TYPE_IDENTITY16
|
||||
1, // TXFM_TYPE_IDENTITY32
|
||||
};
|
||||
|
||||
void av1_range_check_buf(int32_t stage, const int32_t *input,
|
||||
const int32_t *buf, int32_t size, int8_t bit) {
|
||||
#if CONFIG_COEFFICIENT_RANGE_CHECKING
|
||||
const int64_t max_value = (1LL << (bit - 1)) - 1;
|
||||
const int64_t min_value = -(1LL << (bit - 1));
|
||||
|
||||
int in_range = 1;
|
||||
|
||||
for (int i = 0; i < size; ++i) {
|
||||
if (buf[i] < min_value || buf[i] > max_value) {
|
||||
in_range = 0;
|
||||
}
|
||||
}
|
||||
|
||||
if (!in_range) {
|
||||
fprintf(stderr, "Error: coeffs contain out-of-range values\n");
|
||||
fprintf(stderr, "size: %d\n", size);
|
||||
fprintf(stderr, "stage: %d\n", stage);
|
||||
fprintf(stderr, "allowed range: [%" PRId64 ";%" PRId64 "]\n", min_value,
|
||||
max_value);
|
||||
|
||||
fprintf(stderr, "coeffs: ");
|
||||
|
||||
fprintf(stderr, "[");
|
||||
for (int j = 0; j < size; j++) {
|
||||
if (j > 0) fprintf(stderr, ", ");
|
||||
fprintf(stderr, "%d", input[j]);
|
||||
}
|
||||
fprintf(stderr, "]\n");
|
||||
|
||||
fprintf(stderr, " buf: ");
|
||||
|
||||
fprintf(stderr, "[");
|
||||
for (int j = 0; j < size; j++) {
|
||||
if (j > 0) fprintf(stderr, ", ");
|
||||
fprintf(stderr, "%d", buf[j]);
|
||||
}
|
||||
fprintf(stderr, "]\n\n");
|
||||
}
|
||||
|
||||
assert(in_range);
|
||||
#else
|
||||
(void)stage;
|
||||
(void)input;
|
||||
(void)buf;
|
||||
(void)size;
|
||||
(void)bit;
|
||||
#endif
|
||||
}
|
||||
|
|
|
|||
32
third_party/aom/av1/common/av1_txfm.h
vendored
32
third_party/aom/av1/common/av1_txfm.h
vendored
|
|
@ -9,8 +9,8 @@
|
|||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#ifndef AV1_TXFM_H_
|
||||
#define AV1_TXFM_H_
|
||||
#ifndef AOM_AV1_COMMON_AV1_TXFM_H_
|
||||
#define AOM_AV1_COMMON_AV1_TXFM_H_
|
||||
|
||||
#include <assert.h>
|
||||
#include <math.h>
|
||||
|
|
@ -39,7 +39,7 @@ extern const int32_t av1_sinpi_arr_data[7][5];
|
|||
static const int cos_bit_min = 10;
|
||||
static const int cos_bit_max = 16;
|
||||
|
||||
static const int NewSqrt2Bits = 12;
|
||||
#define NewSqrt2Bits ((int32_t)12)
|
||||
// 2^12 * sqrt(2)
|
||||
static const int32_t NewSqrt2 = 5793;
|
||||
// 2^12 / sqrt(2)
|
||||
|
|
@ -64,7 +64,7 @@ static INLINE int32_t range_check_value(int32_t value, int8_t bit) {
|
|||
#endif // CONFIG_COEFFICIENT_RANGE_CHECKING
|
||||
#if DO_RANGE_CHECK_CLAMP
|
||||
bit = AOMMIN(bit, 31);
|
||||
return clamp(value, (1 << (bit - 1)) - 1, -(1 << (bit - 1)));
|
||||
return clamp(value, -(1 << (bit - 1)), (1 << (bit - 1)) - 1);
|
||||
#endif // DO_RANGE_CHECK_CLAMP
|
||||
(void)bit;
|
||||
return value;
|
||||
|
|
@ -78,10 +78,25 @@ static INLINE int32_t round_shift(int64_t value, int bit) {
|
|||
static INLINE int32_t half_btf(int32_t w0, int32_t in0, int32_t w1, int32_t in1,
|
||||
int bit) {
|
||||
int64_t result_64 = (int64_t)(w0 * in0) + (int64_t)(w1 * in1);
|
||||
int64_t intermediate = result_64 + (1LL << (bit - 1));
|
||||
// NOTE(david.barker): The value 'result_64' may not necessarily fit
|
||||
// into 32 bits. However, the result of this function is nominally
|
||||
// ROUND_POWER_OF_TWO_64(result_64, bit)
|
||||
// and that is required to fit into stage_range[stage] many bits
|
||||
// (checked by range_check_buf()).
|
||||
//
|
||||
// Here we've unpacked that rounding operation, and it can be shown
|
||||
// that the value of 'intermediate' here *does* fit into 32 bits
|
||||
// for any conformant bitstream.
|
||||
// The upshot is that, if you do all this calculation using
|
||||
// wrapping 32-bit arithmetic instead of (non-wrapping) 64-bit arithmetic,
|
||||
// then you'll still get the correct result.
|
||||
// To provide a check on this logic, we assert that 'intermediate'
|
||||
// would fit into an int32 if range checking is enabled.
|
||||
#if CONFIG_COEFFICIENT_RANGE_CHECKING
|
||||
assert(result_64 >= INT32_MIN && result_64 <= INT32_MAX);
|
||||
assert(intermediate >= INT32_MIN && intermediate <= INT32_MAX);
|
||||
#endif
|
||||
return round_shift(result_64, bit);
|
||||
return (int32_t)(intermediate >> bit);
|
||||
}
|
||||
|
||||
static INLINE uint16_t highbd_clip_pixel_add(uint16_t dest, tran_high_t trans,
|
||||
|
|
@ -206,9 +221,12 @@ static INLINE int get_txw_idx(TX_SIZE tx_size) {
|
|||
static INLINE int get_txh_idx(TX_SIZE tx_size) {
|
||||
return tx_size_high_log2[tx_size] - tx_size_high_log2[0];
|
||||
}
|
||||
|
||||
void av1_range_check_buf(int32_t stage, const int32_t *input,
|
||||
const int32_t *buf, int32_t size, int8_t bit);
|
||||
#define MAX_TXWH_IDX 5
|
||||
#ifdef __cplusplus
|
||||
}
|
||||
#endif // __cplusplus
|
||||
|
||||
#endif // AV1_TXFM_H_
|
||||
#endif // AOM_AV1_COMMON_AV1_TXFM_H_
|
||||
|
|
|
|||
64
third_party/aom/av1/common/blockd.c
vendored
64
third_party/aom/av1/common/blockd.c
vendored
|
|
@ -28,66 +28,6 @@ PREDICTION_MODE av1_above_block_mode(const MB_MODE_INFO *above_mi) {
|
|||
return above_mi->mode;
|
||||
}
|
||||
|
||||
void av1_foreach_transformed_block_in_plane(
|
||||
const MACROBLOCKD *const xd, BLOCK_SIZE bsize, int plane,
|
||||
foreach_transformed_block_visitor visit, void *arg) {
|
||||
const struct macroblockd_plane *const pd = &xd->plane[plane];
|
||||
// block and transform sizes, in number of 4x4 blocks log 2 ("*_b")
|
||||
// 4x4=0, 8x8=2, 16x16=4, 32x32=6, 64x64=8
|
||||
// transform size varies per plane, look it up in a common way.
|
||||
const TX_SIZE tx_size = av1_get_tx_size(plane, xd);
|
||||
const BLOCK_SIZE plane_bsize =
|
||||
get_plane_block_size(bsize, pd->subsampling_x, pd->subsampling_y);
|
||||
const uint8_t txw_unit = tx_size_wide_unit[tx_size];
|
||||
const uint8_t txh_unit = tx_size_high_unit[tx_size];
|
||||
const int step = txw_unit * txh_unit;
|
||||
int i = 0, r, c;
|
||||
|
||||
// If mb_to_right_edge is < 0 we are in a situation in which
|
||||
// the current block size extends into the UMV and we won't
|
||||
// visit the sub blocks that are wholly within the UMV.
|
||||
const int max_blocks_wide = max_block_wide(xd, plane_bsize, plane);
|
||||
const int max_blocks_high = max_block_high(xd, plane_bsize, plane);
|
||||
|
||||
int blk_row, blk_col;
|
||||
|
||||
const BLOCK_SIZE max_unit_bsize =
|
||||
get_plane_block_size(BLOCK_64X64, pd->subsampling_x, pd->subsampling_y);
|
||||
int mu_blocks_wide = block_size_wide[max_unit_bsize] >> tx_size_wide_log2[0];
|
||||
int mu_blocks_high = block_size_high[max_unit_bsize] >> tx_size_high_log2[0];
|
||||
mu_blocks_wide = AOMMIN(max_blocks_wide, mu_blocks_wide);
|
||||
mu_blocks_high = AOMMIN(max_blocks_high, mu_blocks_high);
|
||||
|
||||
// Keep track of the row and column of the blocks we use so that we know
|
||||
// if we are in the unrestricted motion border.
|
||||
for (r = 0; r < max_blocks_high; r += mu_blocks_high) {
|
||||
const int unit_height = AOMMIN(mu_blocks_high + r, max_blocks_high);
|
||||
// Skip visiting the sub blocks that are wholly within the UMV.
|
||||
for (c = 0; c < max_blocks_wide; c += mu_blocks_wide) {
|
||||
const int unit_width = AOMMIN(mu_blocks_wide + c, max_blocks_wide);
|
||||
for (blk_row = r; blk_row < unit_height; blk_row += txh_unit) {
|
||||
for (blk_col = c; blk_col < unit_width; blk_col += txw_unit) {
|
||||
visit(plane, i, blk_row, blk_col, plane_bsize, tx_size, arg);
|
||||
i += step;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void av1_foreach_transformed_block(const MACROBLOCKD *const xd,
|
||||
BLOCK_SIZE bsize, int mi_row, int mi_col,
|
||||
foreach_transformed_block_visitor visit,
|
||||
void *arg, const int num_planes) {
|
||||
for (int plane = 0; plane < num_planes; ++plane) {
|
||||
if (!is_chroma_reference(mi_row, mi_col, bsize,
|
||||
xd->plane[plane].subsampling_x,
|
||||
xd->plane[plane].subsampling_y))
|
||||
continue;
|
||||
av1_foreach_transformed_block_in_plane(xd, bsize, plane, visit, arg);
|
||||
}
|
||||
}
|
||||
|
||||
void av1_set_contexts(const MACROBLOCKD *xd, struct macroblockd_plane *pd,
|
||||
int plane, BLOCK_SIZE plane_bsize, TX_SIZE tx_size,
|
||||
int has_eob, int aoff, int loff) {
|
||||
|
|
@ -159,6 +99,10 @@ void av1_setup_block_planes(MACROBLOCKD *xd, int ss_x, int ss_y,
|
|||
xd->plane[i].subsampling_x = i ? ss_x : 0;
|
||||
xd->plane[i].subsampling_y = i ? ss_y : 0;
|
||||
}
|
||||
for (i = num_planes; i < MAX_MB_PLANE; i++) {
|
||||
xd->plane[i].subsampling_x = 1;
|
||||
xd->plane[i].subsampling_y = 1;
|
||||
}
|
||||
}
|
||||
|
||||
const int16_t dr_intra_derivative[90] = {
|
||||
|
|
|
|||
53
third_party/aom/av1/common/blockd.h
vendored
53
third_party/aom/av1/common/blockd.h
vendored
|
|
@ -9,8 +9,8 @@
|
|||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#ifndef AV1_COMMON_BLOCKD_H_
|
||||
#define AV1_COMMON_BLOCKD_H_
|
||||
#ifndef AOM_AV1_COMMON_BLOCKD_H_
|
||||
#define AOM_AV1_COMMON_BLOCKD_H_
|
||||
|
||||
#include "config/aom_config.h"
|
||||
|
||||
|
|
@ -38,13 +38,13 @@ extern "C" {
|
|||
#define MAX_DIFFWTD_MASK_BITS 1
|
||||
|
||||
// DIFFWTD_MASK_TYPES should not surpass 1 << MAX_DIFFWTD_MASK_BITS
|
||||
typedef enum {
|
||||
typedef enum ATTRIBUTE_PACKED {
|
||||
DIFFWTD_38 = 0,
|
||||
DIFFWTD_38_INV,
|
||||
DIFFWTD_MASK_TYPES,
|
||||
} DIFFWTD_MASK_TYPE;
|
||||
|
||||
typedef enum {
|
||||
typedef enum ATTRIBUTE_PACKED {
|
||||
KEY_FRAME = 0,
|
||||
INTER_FRAME = 1,
|
||||
INTRA_ONLY_FRAME = 2, // replaces intra-only
|
||||
|
|
@ -57,7 +57,7 @@ static INLINE int is_comp_ref_allowed(BLOCK_SIZE bsize) {
|
|||
}
|
||||
|
||||
static INLINE int is_inter_mode(PREDICTION_MODE mode) {
|
||||
return mode >= NEARESTMV && mode <= NEW_NEWMV;
|
||||
return mode >= INTER_MODE_START && mode < INTER_MODE_END;
|
||||
}
|
||||
|
||||
typedef struct {
|
||||
|
|
@ -66,10 +66,10 @@ typedef struct {
|
|||
} BUFFER_SET;
|
||||
|
||||
static INLINE int is_inter_singleref_mode(PREDICTION_MODE mode) {
|
||||
return mode >= NEARESTMV && mode <= NEWMV;
|
||||
return mode >= SINGLE_INTER_MODE_START && mode < SINGLE_INTER_MODE_END;
|
||||
}
|
||||
static INLINE int is_inter_compound_mode(PREDICTION_MODE mode) {
|
||||
return mode >= NEAREST_NEARESTMV && mode <= NEW_NEWMV;
|
||||
return mode >= COMP_INTER_MODE_START && mode < COMP_INTER_MODE_END;
|
||||
}
|
||||
|
||||
static INLINE PREDICTION_MODE compound_ref0_mode(PREDICTION_MODE mode) {
|
||||
|
|
@ -148,10 +148,6 @@ static INLINE int have_newmv_in_inter_mode(PREDICTION_MODE mode) {
|
|||
mode == NEW_NEARESTMV || mode == NEAR_NEWMV || mode == NEW_NEARMV);
|
||||
}
|
||||
|
||||
static INLINE int use_masked_motion_search(COMPOUND_TYPE type) {
|
||||
return (type == COMPOUND_WEDGE);
|
||||
}
|
||||
|
||||
static INLINE int is_masked_compound_type(COMPOUND_TYPE type) {
|
||||
return (type == COMPOUND_WEDGE || type == COMPOUND_DIFFWTD);
|
||||
}
|
||||
|
|
@ -267,8 +263,8 @@ typedef struct MB_MODE_INFO {
|
|||
int mi_row;
|
||||
int mi_col;
|
||||
#endif
|
||||
int num_proj_ref[2];
|
||||
WarpedMotionParams wm_params[2];
|
||||
int num_proj_ref;
|
||||
WarpedMotionParams wm_params;
|
||||
|
||||
// Index of the alpha Cb and alpha Cr combination
|
||||
int cfl_alpha_idx;
|
||||
|
|
@ -376,7 +372,7 @@ static INLINE void mi_to_pixel_loc(int *pixel_c, int *pixel_r, int mi_col,
|
|||
}
|
||||
#endif
|
||||
|
||||
enum mv_precision { MV_PRECISION_Q3, MV_PRECISION_Q4 };
|
||||
enum ATTRIBUTE_PACKED mv_precision { MV_PRECISION_Q3, MV_PRECISION_Q4 };
|
||||
|
||||
struct buf_2d {
|
||||
uint8_t *buf;
|
||||
|
|
@ -500,6 +496,8 @@ typedef struct jnt_comp_params {
|
|||
int bck_offset;
|
||||
} JNT_COMP_PARAMS;
|
||||
|
||||
// Most/all of the pointers are mere pointers to actual arrays are allocated
|
||||
// elsewhere. This is mostly for coding convenience.
|
||||
typedef struct macroblockd {
|
||||
struct macroblockd_plane plane[MAX_MB_PLANE];
|
||||
|
||||
|
|
@ -544,7 +542,7 @@ typedef struct macroblockd {
|
|||
SgrprojInfo sgrproj_info[MAX_MB_PLANE];
|
||||
|
||||
// block dimension in the unit of mode_info.
|
||||
uint8_t n8_w, n8_h;
|
||||
uint8_t n4_w, n4_h;
|
||||
|
||||
uint8_t ref_mv_count[MODE_CTX_REF_FRAMES];
|
||||
CANDIDATE_MV ref_mv_stack[MODE_CTX_REF_FRAMES][MAX_REF_MV_STACK_SIZE];
|
||||
|
|
@ -599,6 +597,9 @@ typedef struct macroblockd {
|
|||
uint16_t cb_offset[MAX_MB_PLANE];
|
||||
uint16_t txb_offset[MAX_MB_PLANE];
|
||||
uint16_t color_index_map_offset[2];
|
||||
|
||||
CONV_BUF_TYPE *tmp_conv_dst;
|
||||
uint8_t *tmp_obmc_bufs[2];
|
||||
} MACROBLOCKD;
|
||||
|
||||
static INLINE int get_bitdepth_data_path_index(const MACROBLOCKD *xd) {
|
||||
|
|
@ -623,6 +624,11 @@ static INLINE int get_sqr_bsize_idx(BLOCK_SIZE bsize) {
|
|||
}
|
||||
}
|
||||
|
||||
// For a square block size 'bsize', returns the size of the sub-blocks used by
|
||||
// the given partition type. If the partition produces sub-blocks of different
|
||||
// sizes, then the function returns the largest sub-block size.
|
||||
// Implements the Partition_Subsize lookup table in the spec (Section 9.3.
|
||||
// Conversion tables).
|
||||
// Note: the input block size should be square.
|
||||
// Otherwise it's considered invalid.
|
||||
static INLINE BLOCK_SIZE get_partition_subsize(BLOCK_SIZE bsize,
|
||||
|
|
@ -781,6 +787,8 @@ static INLINE TX_TYPE get_default_tx_type(PLANE_TYPE plane_type,
|
|||
return intra_mode_to_tx_type(mbmi, plane_type);
|
||||
}
|
||||
|
||||
// Implements the get_plane_residual_size() function in the spec (Section
|
||||
// 5.11.38. Get plane residual size function).
|
||||
static INLINE BLOCK_SIZE get_plane_block_size(BLOCK_SIZE bsize,
|
||||
int subsampling_x,
|
||||
int subsampling_y) {
|
||||
|
|
@ -952,15 +960,6 @@ typedef void (*foreach_transformed_block_visitor)(int plane, int block,
|
|||
BLOCK_SIZE plane_bsize,
|
||||
TX_SIZE tx_size, void *arg);
|
||||
|
||||
void av1_foreach_transformed_block_in_plane(
|
||||
const MACROBLOCKD *const xd, BLOCK_SIZE bsize, int plane,
|
||||
foreach_transformed_block_visitor visit, void *arg);
|
||||
|
||||
void av1_foreach_transformed_block(const MACROBLOCKD *const xd,
|
||||
BLOCK_SIZE bsize, int mi_row, int mi_col,
|
||||
foreach_transformed_block_visitor visit,
|
||||
void *arg, const int num_planes);
|
||||
|
||||
void av1_set_contexts(const MACROBLOCKD *xd, struct macroblockd_plane *pd,
|
||||
int plane, BLOCK_SIZE plane_bsize, TX_SIZE tx_size,
|
||||
int has_eob, int aoff, int loff);
|
||||
|
|
@ -976,7 +975,7 @@ static INLINE int is_interintra_allowed_bsize(const BLOCK_SIZE bsize) {
|
|||
}
|
||||
|
||||
static INLINE int is_interintra_allowed_mode(const PREDICTION_MODE mode) {
|
||||
return (mode >= NEARESTMV) && (mode <= NEWMV);
|
||||
return (mode >= SINGLE_INTER_MODE_START) && (mode < SINGLE_INTER_MODE_END);
|
||||
}
|
||||
|
||||
static INLINE int is_interintra_allowed_ref(const MV_REFERENCE_FRAME rf[2]) {
|
||||
|
|
@ -1045,7 +1044,7 @@ motion_mode_allowed(const WarpedMotionParams *gm_params, const MACROBLOCKD *xd,
|
|||
is_motion_variation_allowed_compound(mbmi)) {
|
||||
if (!check_num_overlappable_neighbors(mbmi)) return SIMPLE_TRANSLATION;
|
||||
assert(!has_second_ref(mbmi));
|
||||
if (mbmi->num_proj_ref[0] >= 1 &&
|
||||
if (mbmi->num_proj_ref >= 1 &&
|
||||
(allow_warped_motion && !av1_is_scaled(&(xd->block_refs[0]->sf)))) {
|
||||
if (xd->cur_frame_force_integer_mv) {
|
||||
return OBMC_CAUSAL;
|
||||
|
|
@ -1174,4 +1173,4 @@ static INLINE int av1_get_max_eob(TX_SIZE tx_size) {
|
|||
} // extern "C"
|
||||
#endif
|
||||
|
||||
#endif // AV1_COMMON_BLOCKD_H_
|
||||
#endif // AOM_AV1_COMMON_BLOCKD_H_
|
||||
|
|
|
|||
6
third_party/aom/av1/common/cdef.h
vendored
6
third_party/aom/av1/common/cdef.h
vendored
|
|
@ -8,8 +8,8 @@
|
|||
* Media Patent License 1.0 was not distributed with this source code in the
|
||||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
#ifndef AV1_COMMON_CDEF_H_
|
||||
#define AV1_COMMON_CDEF_H_
|
||||
#ifndef AOM_AV1_COMMON_CDEF_H_
|
||||
#define AOM_AV1_COMMON_CDEF_H_
|
||||
|
||||
#define CDEF_STRENGTH_BITS 6
|
||||
|
||||
|
|
@ -48,4 +48,4 @@ void av1_cdef_search(YV12_BUFFER_CONFIG *frame, const YV12_BUFFER_CONFIG *ref,
|
|||
#ifdef __cplusplus
|
||||
} // extern "C"
|
||||
#endif
|
||||
#endif // AV1_COMMON_CDEF_H_
|
||||
#endif // AOM_AV1_COMMON_CDEF_H_
|
||||
|
|
|
|||
6
third_party/aom/av1/common/cdef_block.h
vendored
6
third_party/aom/av1/common/cdef_block.h
vendored
|
|
@ -9,8 +9,8 @@
|
|||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#if !defined(_CDEF_BLOCK_H)
|
||||
#define _CDEF_BLOCK_H (1)
|
||||
#ifndef AOM_AV1_COMMON_CDEF_BLOCK_H_
|
||||
#define AOM_AV1_COMMON_CDEF_BLOCK_H_
|
||||
|
||||
#include "av1/common/odintrin.h"
|
||||
|
||||
|
|
@ -56,4 +56,4 @@ void cdef_filter_fb(uint8_t *dst8, uint16_t *dst16, int dstride, uint16_t *in,
|
|||
cdef_list *dlist, int cdef_count, int level,
|
||||
int sec_strength, int pri_damping, int sec_damping,
|
||||
int coeff_shift);
|
||||
#endif
|
||||
#endif // AOM_AV1_COMMON_CDEF_BLOCK_H_
|
||||
|
|
|
|||
5
third_party/aom/av1/common/cdef_block_simd.h
vendored
5
third_party/aom/av1/common/cdef_block_simd.h
vendored
|
|
@ -9,6 +9,9 @@
|
|||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#ifndef AOM_AV1_COMMON_CDEF_BLOCK_SIMD_H_
|
||||
#define AOM_AV1_COMMON_CDEF_BLOCK_SIMD_H_
|
||||
|
||||
#include "config/av1_rtcd.h"
|
||||
|
||||
#include "av1/common/cdef_block.h"
|
||||
|
|
@ -913,3 +916,5 @@ void SIMD_FUNC(copy_rect8_16bit_to_16bit)(uint16_t *dst, int dstride,
|
|||
}
|
||||
}
|
||||
}
|
||||
|
||||
#endif // AOM_AV1_COMMON_CDEF_BLOCK_SIMD_H_
|
||||
|
|
|
|||
6
third_party/aom/av1/common/cfl.h
vendored
6
third_party/aom/av1/common/cfl.h
vendored
|
|
@ -9,8 +9,8 @@
|
|||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#ifndef AV1_COMMON_CFL_H_
|
||||
#define AV1_COMMON_CFL_H_
|
||||
#ifndef AOM_AV1_COMMON_CFL_H_
|
||||
#define AOM_AV1_COMMON_CFL_H_
|
||||
|
||||
#include "av1/common/blockd.h"
|
||||
#include "av1/common/onyxc_int.h"
|
||||
|
|
@ -299,4 +299,4 @@ void cfl_predict_hbd_null(const int16_t *pred_buf_q3, uint16_t *dst,
|
|||
return pred[tx_size % TX_SIZES_ALL]; \
|
||||
}
|
||||
|
||||
#endif // AV1_COMMON_CFL_H_
|
||||
#endif // AOM_AV1_COMMON_CFL_H_
|
||||
|
|
|
|||
6
third_party/aom/av1/common/common.h
vendored
6
third_party/aom/av1/common/common.h
vendored
|
|
@ -9,8 +9,8 @@
|
|||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#ifndef AV1_COMMON_COMMON_H_
|
||||
#define AV1_COMMON_COMMON_H_
|
||||
#ifndef AOM_AV1_COMMON_COMMON_H_
|
||||
#define AOM_AV1_COMMON_COMMON_H_
|
||||
|
||||
/* Interface header for common constant data structures and lookup tables */
|
||||
|
||||
|
|
@ -60,4 +60,4 @@ static INLINE int get_unsigned_bits(unsigned int num_values) {
|
|||
} // extern "C"
|
||||
#endif
|
||||
|
||||
#endif // AV1_COMMON_COMMON_H_
|
||||
#endif // AOM_AV1_COMMON_COMMON_H_
|
||||
|
|
|
|||
75
third_party/aom/av1/common/common_data.h
vendored
75
third_party/aom/av1/common/common_data.h
vendored
|
|
@ -9,8 +9,8 @@
|
|||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#ifndef AV1_COMMON_COMMON_DATA_H_
|
||||
#define AV1_COMMON_COMMON_DATA_H_
|
||||
#ifndef AOM_AV1_COMMON_COMMON_DATA_H_
|
||||
#define AOM_AV1_COMMON_COMMON_DATA_H_
|
||||
|
||||
#include "av1/common/enums.h"
|
||||
#include "aom/aom_integer.h"
|
||||
|
|
@ -20,34 +20,43 @@
|
|||
extern "C" {
|
||||
#endif
|
||||
|
||||
// Log 2 conversion lookup tables in units of mode info(4x4).
|
||||
// Log 2 conversion lookup tables in units of mode info (4x4).
|
||||
// The Mi_Width_Log2 table in the spec (Section 9.3. Conversion tables).
|
||||
static const uint8_t mi_size_wide_log2[BLOCK_SIZES_ALL] = {
|
||||
0, 0, 1, 1, 1, 2, 2, 2, 3, 3, 3, 4, 4, 4, 5, 5, 0, 2, 1, 3, 2, 4
|
||||
};
|
||||
// The Mi_Height_Log2 table in the spec (Section 9.3. Conversion tables).
|
||||
static const uint8_t mi_size_high_log2[BLOCK_SIZES_ALL] = {
|
||||
0, 1, 0, 1, 2, 1, 2, 3, 2, 3, 4, 3, 4, 5, 4, 5, 2, 0, 3, 1, 4, 2
|
||||
};
|
||||
|
||||
// Width/height lookup tables in units of mode info (4x4).
|
||||
// The Num_4x4_Blocks_Wide table in the spec (Section 9.3. Conversion tables).
|
||||
static const uint8_t mi_size_wide[BLOCK_SIZES_ALL] = {
|
||||
1, 1, 2, 2, 2, 4, 4, 4, 8, 8, 8, 16, 16, 16, 32, 32, 1, 4, 2, 8, 4, 16
|
||||
};
|
||||
|
||||
// The Num_4x4_Blocks_High table in the spec (Section 9.3. Conversion tables).
|
||||
static const uint8_t mi_size_high[BLOCK_SIZES_ALL] = {
|
||||
1, 2, 1, 2, 4, 2, 4, 8, 4, 8, 16, 8, 16, 32, 16, 32, 4, 1, 8, 2, 16, 4
|
||||
};
|
||||
|
||||
// Width/height lookup tables in units of various block sizes
|
||||
// Width/height lookup tables in units of samples.
|
||||
// The Block_Width table in the spec (Section 9.3. Conversion tables).
|
||||
static const uint8_t block_size_wide[BLOCK_SIZES_ALL] = {
|
||||
4, 4, 8, 8, 8, 16, 16, 16, 32, 32, 32,
|
||||
64, 64, 64, 128, 128, 4, 16, 8, 32, 16, 64
|
||||
};
|
||||
|
||||
// The Block_Height table in the spec (Section 9.3. Conversion tables).
|
||||
static const uint8_t block_size_high[BLOCK_SIZES_ALL] = {
|
||||
4, 8, 4, 8, 16, 8, 16, 32, 16, 32, 64,
|
||||
32, 64, 128, 64, 128, 16, 4, 32, 8, 64, 16
|
||||
};
|
||||
|
||||
// AOMMIN(3, AOMMIN(b_width_log2(bsize), b_height_log2(bsize)))
|
||||
// Maps a block size to a context.
|
||||
// The Size_Group table in the spec (Section 9.3. Conversion tables).
|
||||
// AOMMIN(3, AOMMIN(mi_size_wide_log2(bsize), mi_size_high_log2(bsize)))
|
||||
static const uint8_t size_group_lookup[BLOCK_SIZES_ALL] = {
|
||||
0, 0, 0, 1, 1, 1, 2, 2, 2, 3, 3, 3, 3, 3, 3, 3, 0, 0, 1, 1, 2, 2
|
||||
};
|
||||
|
|
@ -56,6 +65,8 @@ static const uint8_t num_pels_log2_lookup[BLOCK_SIZES_ALL] = {
|
|||
4, 5, 5, 6, 7, 7, 8, 9, 9, 10, 11, 11, 12, 13, 13, 14, 6, 6, 8, 8, 10, 10
|
||||
};
|
||||
|
||||
// A compressed version of the Partition_Subsize table in the spec (9.3.
|
||||
// Conversion tables), for square block sizes only.
|
||||
/* clang-format off */
|
||||
static const BLOCK_SIZE subsize_lookup[EXT_PARTITION_TYPES][SQR_BLOCK_SIZES] = {
|
||||
{ // PARTITION_NONE
|
||||
|
|
@ -350,34 +361,36 @@ static const TX_SIZE tx_mode_to_biggest_tx_size[TX_MODES] = {
|
|||
TX_64X64, // TX_MODE_LARGEST
|
||||
TX_64X64, // TX_MODE_SELECT
|
||||
};
|
||||
/* clang-format on */
|
||||
|
||||
// The Subsampled_Size table in the spec (Section 5.11.38. Get plane residual
|
||||
// size function).
|
||||
static const BLOCK_SIZE ss_size_lookup[BLOCK_SIZES_ALL][2][2] = {
|
||||
// ss_x == 0 ss_x == 0 ss_x == 1 ss_x == 1
|
||||
// ss_y == 0 ss_y == 1 ss_y == 0 ss_y == 1
|
||||
{ { BLOCK_4X4, BLOCK_4X4 }, { BLOCK_4X4, BLOCK_4X4 } },
|
||||
{ { BLOCK_4X8, BLOCK_4X4 }, { BLOCK_4X4, BLOCK_4X4 } },
|
||||
{ { BLOCK_8X4, BLOCK_4X4 }, { BLOCK_4X4, BLOCK_4X4 } },
|
||||
{ { BLOCK_8X8, BLOCK_8X4 }, { BLOCK_4X8, BLOCK_4X4 } },
|
||||
{ { BLOCK_8X16, BLOCK_8X8 }, { BLOCK_4X16, BLOCK_4X8 } },
|
||||
{ { BLOCK_16X8, BLOCK_16X4 }, { BLOCK_8X8, BLOCK_8X4 } },
|
||||
{ { BLOCK_16X16, BLOCK_16X8 }, { BLOCK_8X16, BLOCK_8X8 } },
|
||||
{ { BLOCK_16X32, BLOCK_16X16 }, { BLOCK_8X32, BLOCK_8X16 } },
|
||||
{ { BLOCK_32X16, BLOCK_32X8 }, { BLOCK_16X16, BLOCK_16X8 } },
|
||||
{ { BLOCK_32X32, BLOCK_32X16 }, { BLOCK_16X32, BLOCK_16X16 } },
|
||||
{ { BLOCK_32X64, BLOCK_32X32 }, { BLOCK_16X64, BLOCK_16X32 } },
|
||||
{ { BLOCK_64X32, BLOCK_64X16 }, { BLOCK_32X32, BLOCK_32X16 } },
|
||||
{ { BLOCK_64X64, BLOCK_64X32 }, { BLOCK_32X64, BLOCK_32X32 } },
|
||||
{ { BLOCK_64X128, BLOCK_64X64 }, { BLOCK_INVALID, BLOCK_32X64 } },
|
||||
{ { BLOCK_128X64, BLOCK_INVALID }, { BLOCK_64X64, BLOCK_64X32 } },
|
||||
{ { BLOCK_128X128, BLOCK_128X64 }, { BLOCK_64X128, BLOCK_64X64 } },
|
||||
{ { BLOCK_4X16, BLOCK_4X8 }, { BLOCK_4X16, BLOCK_4X8 } },
|
||||
{ { BLOCK_16X4, BLOCK_16X4 }, { BLOCK_8X4, BLOCK_8X4 } },
|
||||
{ { BLOCK_8X32, BLOCK_8X16 }, { BLOCK_INVALID, BLOCK_4X16 } },
|
||||
{ { BLOCK_32X8, BLOCK_INVALID }, { BLOCK_16X8, BLOCK_16X4 } },
|
||||
{ { BLOCK_16X64, BLOCK_16X32 }, { BLOCK_INVALID, BLOCK_8X32 } },
|
||||
{ { BLOCK_64X16, BLOCK_INVALID }, { BLOCK_32X16, BLOCK_32X8 } }
|
||||
// ss_x == 0 ss_x == 0 ss_x == 1 ss_x == 1
|
||||
// ss_y == 0 ss_y == 1 ss_y == 0 ss_y == 1
|
||||
{ { BLOCK_4X4, BLOCK_4X4 }, { BLOCK_4X4, BLOCK_4X4 } },
|
||||
{ { BLOCK_4X8, BLOCK_4X4 }, { BLOCK_INVALID, BLOCK_4X4 } },
|
||||
{ { BLOCK_8X4, BLOCK_INVALID }, { BLOCK_4X4, BLOCK_4X4 } },
|
||||
{ { BLOCK_8X8, BLOCK_8X4 }, { BLOCK_4X8, BLOCK_4X4 } },
|
||||
{ { BLOCK_8X16, BLOCK_8X8 }, { BLOCK_INVALID, BLOCK_4X8 } },
|
||||
{ { BLOCK_16X8, BLOCK_INVALID }, { BLOCK_8X8, BLOCK_8X4 } },
|
||||
{ { BLOCK_16X16, BLOCK_16X8 }, { BLOCK_8X16, BLOCK_8X8 } },
|
||||
{ { BLOCK_16X32, BLOCK_16X16 }, { BLOCK_INVALID, BLOCK_8X16 } },
|
||||
{ { BLOCK_32X16, BLOCK_INVALID }, { BLOCK_16X16, BLOCK_16X8 } },
|
||||
{ { BLOCK_32X32, BLOCK_32X16 }, { BLOCK_16X32, BLOCK_16X16 } },
|
||||
{ { BLOCK_32X64, BLOCK_32X32 }, { BLOCK_INVALID, BLOCK_16X32 } },
|
||||
{ { BLOCK_64X32, BLOCK_INVALID }, { BLOCK_32X32, BLOCK_32X16 } },
|
||||
{ { BLOCK_64X64, BLOCK_64X32 }, { BLOCK_32X64, BLOCK_32X32 } },
|
||||
{ { BLOCK_64X128, BLOCK_64X64 }, { BLOCK_INVALID, BLOCK_32X64 } },
|
||||
{ { BLOCK_128X64, BLOCK_INVALID }, { BLOCK_64X64, BLOCK_64X32 } },
|
||||
{ { BLOCK_128X128, BLOCK_128X64 }, { BLOCK_64X128, BLOCK_64X64 } },
|
||||
{ { BLOCK_4X16, BLOCK_4X8 }, { BLOCK_INVALID, BLOCK_4X8 } },
|
||||
{ { BLOCK_16X4, BLOCK_INVALID }, { BLOCK_8X4, BLOCK_8X4 } },
|
||||
{ { BLOCK_8X32, BLOCK_8X16 }, { BLOCK_INVALID, BLOCK_4X16 } },
|
||||
{ { BLOCK_32X8, BLOCK_INVALID }, { BLOCK_16X8, BLOCK_16X4 } },
|
||||
{ { BLOCK_16X64, BLOCK_16X32 }, { BLOCK_INVALID, BLOCK_8X32 } },
|
||||
{ { BLOCK_64X16, BLOCK_INVALID }, { BLOCK_32X16, BLOCK_32X8 } }
|
||||
};
|
||||
/* clang-format on */
|
||||
|
||||
// Generates 5 bit field in which each bit set to 1 represents
|
||||
// a blocksize partition 11111 means we split 128x128, 64x64, 32x32, 16x16
|
||||
|
|
@ -430,4 +443,4 @@ static const int quant_dist_lookup_table[2][4][2] = {
|
|||
} // extern "C"
|
||||
#endif
|
||||
|
||||
#endif // AV1_COMMON_COMMON_DATA_H_
|
||||
#endif // AOM_AV1_COMMON_COMMON_DATA_H_
|
||||
|
|
|
|||
120
third_party/aom/av1/common/convolve.c
vendored
120
third_party/aom/av1/common/convolve.c
vendored
|
|
@ -173,6 +173,7 @@ void av1_convolve_x_sr_c(const uint8_t *src, int src_stride, uint8_t *dst,
|
|||
// horizontal filter
|
||||
const int16_t *x_filter = av1_get_interp_filter_subpel_kernel(
|
||||
filter_params_x, subpel_x_q4 & SUBPEL_MASK);
|
||||
|
||||
for (int y = 0; y < h; ++y) {
|
||||
for (int x = 0; x < w; ++x) {
|
||||
int32_t res = 0;
|
||||
|
|
@ -510,31 +511,73 @@ static void convolve_2d_scale_wrapper(
|
|||
y_step_qn, conv_params);
|
||||
}
|
||||
|
||||
// TODO(huisu@google.com): bilinear filtering only needs 2 taps in general. So
|
||||
// we may create optimized code to do 2-tap filtering for all bilinear filtering
|
||||
// usages, not just IntraBC.
|
||||
static void convolve_2d_for_intrabc(const uint8_t *src, int src_stride,
|
||||
uint8_t *dst, int dst_stride, int w, int h,
|
||||
int subpel_x_q4, int subpel_y_q4,
|
||||
ConvolveParams *conv_params) {
|
||||
const InterpFilterParams *filter_params_x =
|
||||
subpel_x_q4 ? &av1_intrabc_filter_params : NULL;
|
||||
const InterpFilterParams *filter_params_y =
|
||||
subpel_y_q4 ? &av1_intrabc_filter_params : NULL;
|
||||
if (subpel_x_q4 != 0 && subpel_y_q4 != 0) {
|
||||
av1_convolve_2d_sr_c(src, src_stride, dst, dst_stride, w, h,
|
||||
filter_params_x, filter_params_y, 0, 0, conv_params);
|
||||
} else if (subpel_x_q4 != 0) {
|
||||
av1_convolve_x_sr_c(src, src_stride, dst, dst_stride, w, h, filter_params_x,
|
||||
filter_params_y, 0, 0, conv_params);
|
||||
} else {
|
||||
av1_convolve_y_sr_c(src, src_stride, dst, dst_stride, w, h, filter_params_x,
|
||||
filter_params_y, 0, 0, conv_params);
|
||||
}
|
||||
}
|
||||
|
||||
void av1_convolve_2d_facade(const uint8_t *src, int src_stride, uint8_t *dst,
|
||||
int dst_stride, int w, int h,
|
||||
InterpFilters interp_filters, const int subpel_x_q4,
|
||||
int x_step_q4, const int subpel_y_q4, int y_step_q4,
|
||||
int scaled, ConvolveParams *conv_params,
|
||||
const struct scale_factors *sf) {
|
||||
const struct scale_factors *sf, int is_intrabc) {
|
||||
assert(IMPLIES(is_intrabc, !scaled));
|
||||
(void)x_step_q4;
|
||||
(void)y_step_q4;
|
||||
(void)dst;
|
||||
(void)dst_stride;
|
||||
InterpFilter filter_x = av1_extract_interp_filter(interp_filters, 1);
|
||||
InterpFilter filter_y = av1_extract_interp_filter(interp_filters, 0);
|
||||
const InterpFilterParams *filter_params_x =
|
||||
av1_get_interp_filter_params_with_block_size(filter_x, w);
|
||||
const InterpFilterParams *filter_params_y =
|
||||
av1_get_interp_filter_params_with_block_size(filter_y, h);
|
||||
|
||||
if (scaled)
|
||||
if (is_intrabc && (subpel_x_q4 != 0 || subpel_y_q4 != 0)) {
|
||||
convolve_2d_for_intrabc(src, src_stride, dst, dst_stride, w, h, subpel_x_q4,
|
||||
subpel_y_q4, conv_params);
|
||||
return;
|
||||
}
|
||||
|
||||
InterpFilter filter_x = 0;
|
||||
InterpFilter filter_y = 0;
|
||||
const int need_filter_params_x = (subpel_x_q4 != 0) | scaled;
|
||||
const int need_filter_params_y = (subpel_y_q4 != 0) | scaled;
|
||||
if (need_filter_params_x)
|
||||
filter_x = av1_extract_interp_filter(interp_filters, 1);
|
||||
if (need_filter_params_y)
|
||||
filter_y = av1_extract_interp_filter(interp_filters, 0);
|
||||
const InterpFilterParams *filter_params_x =
|
||||
need_filter_params_x
|
||||
? av1_get_interp_filter_params_with_block_size(filter_x, w)
|
||||
: NULL;
|
||||
const InterpFilterParams *filter_params_y =
|
||||
need_filter_params_y
|
||||
? av1_get_interp_filter_params_with_block_size(filter_y, h)
|
||||
: NULL;
|
||||
|
||||
if (scaled) {
|
||||
convolve_2d_scale_wrapper(src, src_stride, dst, dst_stride, w, h,
|
||||
filter_params_x, filter_params_y, subpel_x_q4,
|
||||
x_step_q4, subpel_y_q4, y_step_q4, conv_params);
|
||||
else
|
||||
} else {
|
||||
sf->convolve[subpel_x_q4 != 0][subpel_y_q4 != 0][conv_params->is_compound](
|
||||
src, src_stride, dst, dst_stride, w, h, filter_params_x,
|
||||
filter_params_y, subpel_x_q4, subpel_y_q4, conv_params);
|
||||
}
|
||||
}
|
||||
|
||||
void av1_highbd_convolve_2d_copy_sr_c(
|
||||
|
|
@ -964,24 +1007,68 @@ void av1_highbd_convolve_2d_scale_c(const uint16_t *src, int src_stride,
|
|||
}
|
||||
}
|
||||
|
||||
static void highbd_convolve_2d_for_intrabc(const uint16_t *src, int src_stride,
|
||||
uint16_t *dst, int dst_stride, int w,
|
||||
int h, int subpel_x_q4,
|
||||
int subpel_y_q4,
|
||||
ConvolveParams *conv_params,
|
||||
int bd) {
|
||||
const InterpFilterParams *filter_params_x =
|
||||
subpel_x_q4 ? &av1_intrabc_filter_params : NULL;
|
||||
const InterpFilterParams *filter_params_y =
|
||||
subpel_y_q4 ? &av1_intrabc_filter_params : NULL;
|
||||
if (subpel_x_q4 != 0 && subpel_y_q4 != 0) {
|
||||
av1_highbd_convolve_2d_sr_c(src, src_stride, dst, dst_stride, w, h,
|
||||
filter_params_x, filter_params_y, 0, 0,
|
||||
conv_params, bd);
|
||||
} else if (subpel_x_q4 != 0) {
|
||||
av1_highbd_convolve_x_sr_c(src, src_stride, dst, dst_stride, w, h,
|
||||
filter_params_x, filter_params_y, 0, 0,
|
||||
conv_params, bd);
|
||||
} else {
|
||||
av1_highbd_convolve_y_sr_c(src, src_stride, dst, dst_stride, w, h,
|
||||
filter_params_x, filter_params_y, 0, 0,
|
||||
conv_params, bd);
|
||||
}
|
||||
}
|
||||
|
||||
void av1_highbd_convolve_2d_facade(const uint8_t *src8, int src_stride,
|
||||
uint8_t *dst8, int dst_stride, int w, int h,
|
||||
InterpFilters interp_filters,
|
||||
const int subpel_x_q4, int x_step_q4,
|
||||
const int subpel_y_q4, int y_step_q4,
|
||||
int scaled, ConvolveParams *conv_params,
|
||||
const struct scale_factors *sf, int bd) {
|
||||
const struct scale_factors *sf,
|
||||
int is_intrabc, int bd) {
|
||||
assert(IMPLIES(is_intrabc, !scaled));
|
||||
(void)x_step_q4;
|
||||
(void)y_step_q4;
|
||||
(void)dst_stride;
|
||||
|
||||
const uint16_t *src = CONVERT_TO_SHORTPTR(src8);
|
||||
InterpFilter filter_x = av1_extract_interp_filter(interp_filters, 1);
|
||||
InterpFilter filter_y = av1_extract_interp_filter(interp_filters, 0);
|
||||
|
||||
if (is_intrabc && (subpel_x_q4 != 0 || subpel_y_q4 != 0)) {
|
||||
uint16_t *dst = CONVERT_TO_SHORTPTR(dst8);
|
||||
highbd_convolve_2d_for_intrabc(src, src_stride, dst, dst_stride, w, h,
|
||||
subpel_x_q4, subpel_y_q4, conv_params, bd);
|
||||
return;
|
||||
}
|
||||
|
||||
InterpFilter filter_x = 0;
|
||||
InterpFilter filter_y = 0;
|
||||
const int need_filter_params_x = (subpel_x_q4 != 0) | scaled;
|
||||
const int need_filter_params_y = (subpel_y_q4 != 0) | scaled;
|
||||
if (need_filter_params_x)
|
||||
filter_x = av1_extract_interp_filter(interp_filters, 1);
|
||||
if (need_filter_params_y)
|
||||
filter_y = av1_extract_interp_filter(interp_filters, 0);
|
||||
const InterpFilterParams *filter_params_x =
|
||||
av1_get_interp_filter_params_with_block_size(filter_x, w);
|
||||
need_filter_params_x
|
||||
? av1_get_interp_filter_params_with_block_size(filter_x, w)
|
||||
: NULL;
|
||||
const InterpFilterParams *filter_params_y =
|
||||
av1_get_interp_filter_params_with_block_size(filter_y, h);
|
||||
need_filter_params_y
|
||||
? av1_get_interp_filter_params_with_block_size(filter_y, h)
|
||||
: NULL;
|
||||
|
||||
if (scaled) {
|
||||
uint16_t *dst = CONVERT_TO_SHORTPTR(dst8);
|
||||
|
|
@ -1111,7 +1198,8 @@ void av1_wiener_convolve_add_src_c(const uint8_t *src, ptrdiff_t src_stride,
|
|||
|
||||
uint16_t temp[WIENER_MAX_EXT_SIZE * MAX_SB_SIZE];
|
||||
const int intermediate_height =
|
||||
(((h - 1) * y_step_q4 + y0_q4) >> SUBPEL_BITS) + SUBPEL_TAPS;
|
||||
(((h - 1) * y_step_q4 + y0_q4) >> SUBPEL_BITS) + SUBPEL_TAPS - 1;
|
||||
memset(temp + (intermediate_height * MAX_SB_SIZE), 0, MAX_SB_SIZE);
|
||||
|
||||
assert(w <= MAX_SB_SIZE);
|
||||
assert(h <= MAX_SB_SIZE);
|
||||
|
|
|
|||
21
third_party/aom/av1/common/convolve.h
vendored
21
third_party/aom/av1/common/convolve.h
vendored
|
|
@ -9,8 +9,8 @@
|
|||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#ifndef AV1_COMMON_AV1_CONVOLVE_H_
|
||||
#define AV1_COMMON_AV1_CONVOLVE_H_
|
||||
#ifndef AOM_AV1_COMMON_CONVOLVE_H_
|
||||
#define AOM_AV1_COMMON_CONVOLVE_H_
|
||||
#include "av1/common/filter.h"
|
||||
|
||||
#ifdef __cplusplus
|
||||
|
|
@ -19,7 +19,6 @@ extern "C" {
|
|||
|
||||
typedef uint16_t CONV_BUF_TYPE;
|
||||
typedef struct ConvolveParams {
|
||||
int ref;
|
||||
int do_average;
|
||||
CONV_BUF_TYPE *dst;
|
||||
int dst_stride;
|
||||
|
|
@ -59,15 +58,13 @@ void av1_convolve_2d_facade(const uint8_t *src, int src_stride, uint8_t *dst,
|
|||
InterpFilters interp_filters, const int subpel_x_q4,
|
||||
int x_step_q4, const int subpel_y_q4, int y_step_q4,
|
||||
int scaled, ConvolveParams *conv_params,
|
||||
const struct scale_factors *sf);
|
||||
const struct scale_factors *sf, int is_intrabc);
|
||||
|
||||
static INLINE ConvolveParams get_conv_params_no_round(int ref, int do_average,
|
||||
int plane,
|
||||
static INLINE ConvolveParams get_conv_params_no_round(int do_average, int plane,
|
||||
CONV_BUF_TYPE *dst,
|
||||
int dst_stride,
|
||||
int is_compound, int bd) {
|
||||
ConvolveParams conv_params;
|
||||
conv_params.ref = ref;
|
||||
conv_params.do_average = do_average;
|
||||
assert(IMPLIES(do_average, is_compound));
|
||||
conv_params.is_compound = is_compound;
|
||||
|
|
@ -88,15 +85,14 @@ static INLINE ConvolveParams get_conv_params_no_round(int ref, int do_average,
|
|||
return conv_params;
|
||||
}
|
||||
|
||||
static INLINE ConvolveParams get_conv_params(int ref, int do_average, int plane,
|
||||
static INLINE ConvolveParams get_conv_params(int do_average, int plane,
|
||||
int bd) {
|
||||
return get_conv_params_no_round(ref, do_average, plane, NULL, 0, 0, bd);
|
||||
return get_conv_params_no_round(do_average, plane, NULL, 0, 0, bd);
|
||||
}
|
||||
|
||||
static INLINE ConvolveParams get_conv_params_wiener(int bd) {
|
||||
ConvolveParams conv_params;
|
||||
(void)bd;
|
||||
conv_params.ref = 0;
|
||||
conv_params.do_average = 0;
|
||||
conv_params.is_compound = 0;
|
||||
conv_params.round_0 = WIENER_ROUND0_BITS;
|
||||
|
|
@ -119,10 +115,11 @@ void av1_highbd_convolve_2d_facade(const uint8_t *src8, int src_stride,
|
|||
const int subpel_x_q4, int x_step_q4,
|
||||
const int subpel_y_q4, int y_step_q4,
|
||||
int scaled, ConvolveParams *conv_params,
|
||||
const struct scale_factors *sf, int bd);
|
||||
const struct scale_factors *sf,
|
||||
int is_intrabc, int bd);
|
||||
|
||||
#ifdef __cplusplus
|
||||
} // extern "C"
|
||||
#endif
|
||||
|
||||
#endif // AV1_COMMON_AV1_CONVOLVE_H_
|
||||
#endif // AOM_AV1_COMMON_CONVOLVE_H_
|
||||
|
|
|
|||
6
third_party/aom/av1/common/entropy.h
vendored
6
third_party/aom/av1/common/entropy.h
vendored
|
|
@ -9,8 +9,8 @@
|
|||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#ifndef AV1_COMMON_ENTROPY_H_
|
||||
#define AV1_COMMON_ENTROPY_H_
|
||||
#ifndef AOM_AV1_COMMON_ENTROPY_H_
|
||||
#define AOM_AV1_COMMON_ENTROPY_H_
|
||||
|
||||
#include "config/aom_config.h"
|
||||
|
||||
|
|
@ -178,4 +178,4 @@ static INLINE TX_SIZE get_txsize_entropy_ctx(TX_SIZE txsize) {
|
|||
} // extern "C"
|
||||
#endif
|
||||
|
||||
#endif // AV1_COMMON_ENTROPY_H_
|
||||
#endif // AOM_AV1_COMMON_ENTROPY_H_
|
||||
|
|
|
|||
8
third_party/aom/av1/common/entropymode.h
vendored
8
third_party/aom/av1/common/entropymode.h
vendored
|
|
@ -9,8 +9,8 @@
|
|||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#ifndef AV1_COMMON_ENTROPYMODE_H_
|
||||
#define AV1_COMMON_ENTROPYMODE_H_
|
||||
#ifndef AOM_AV1_COMMON_ENTROPYMODE_H_
|
||||
#define AOM_AV1_COMMON_ENTROPYMODE_H_
|
||||
|
||||
#include "av1/common/entropy.h"
|
||||
#include "av1/common/entropymv.h"
|
||||
|
|
@ -186,6 +186,8 @@ void av1_set_default_mode_deltas(int8_t *mode_deltas);
|
|||
void av1_setup_frame_contexts(struct AV1Common *cm);
|
||||
void av1_setup_past_independence(struct AV1Common *cm);
|
||||
|
||||
// Returns (int)ceil(log2(n)).
|
||||
// NOTE: This implementation only works for n <= 2^30.
|
||||
static INLINE int av1_ceil_log2(int n) {
|
||||
if (n < 2) return 0;
|
||||
int i = 1, p = 2;
|
||||
|
|
@ -207,4 +209,4 @@ int av1_get_palette_color_index_context(const uint8_t *color_map, int stride,
|
|||
} // extern "C"
|
||||
#endif
|
||||
|
||||
#endif // AV1_COMMON_ENTROPYMODE_H_
|
||||
#endif // AOM_AV1_COMMON_ENTROPYMODE_H_
|
||||
|
|
|
|||
55
third_party/aom/av1/common/entropymv.c
vendored
55
third_party/aom/av1/common/entropymv.c
vendored
|
|
@ -60,61 +60,6 @@ static const nmv_context default_nmv_context = {
|
|||
} },
|
||||
};
|
||||
|
||||
static const uint8_t log_in_base_2[] = {
|
||||
0, 0, 1, 1, 2, 2, 2, 2, 3, 3, 3, 3, 3, 3, 3, 3, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4,
|
||||
4, 4, 4, 4, 4, 4, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5,
|
||||
5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6,
|
||||
6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6,
|
||||
6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 7, 7,
|
||||
7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7,
|
||||
7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7,
|
||||
7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7,
|
||||
7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7,
|
||||
7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 8, 8, 8, 8,
|
||||
8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8,
|
||||
8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8,
|
||||
8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8,
|
||||
8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8,
|
||||
8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8,
|
||||
8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8,
|
||||
8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8,
|
||||
8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8,
|
||||
8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8,
|
||||
8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 9, 9, 9, 9, 9, 9, 9, 9,
|
||||
9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9,
|
||||
9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9,
|
||||
9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9,
|
||||
9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9,
|
||||
9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9,
|
||||
9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9,
|
||||
9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9,
|
||||
9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9,
|
||||
9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9,
|
||||
9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9,
|
||||
9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9,
|
||||
9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9,
|
||||
9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9,
|
||||
9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9,
|
||||
9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9,
|
||||
9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9,
|
||||
9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9,
|
||||
9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9,
|
||||
9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9,
|
||||
9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 10
|
||||
};
|
||||
|
||||
static INLINE int mv_class_base(MV_CLASS_TYPE c) {
|
||||
return c ? CLASS0_SIZE << (c + 2) : 0;
|
||||
}
|
||||
|
||||
MV_CLASS_TYPE av1_get_mv_class(int z, int *offset) {
|
||||
const MV_CLASS_TYPE c = (z >= CLASS0_SIZE * 4096)
|
||||
? MV_CLASS_10
|
||||
: (MV_CLASS_TYPE)log_in_base_2[z >> 3];
|
||||
if (offset) *offset = z - mv_class_base(c);
|
||||
return c;
|
||||
}
|
||||
|
||||
void av1_init_mv_probs(AV1_COMMON *cm) {
|
||||
// NB: this sets CDFs too
|
||||
cm->fc->nmvc = default_nmv_context;
|
||||
|
|
|
|||
16
third_party/aom/av1/common/entropymv.h
vendored
16
third_party/aom/av1/common/entropymv.h
vendored
|
|
@ -9,8 +9,8 @@
|
|||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#ifndef AV1_COMMON_ENTROPYMV_H_
|
||||
#define AV1_COMMON_ENTROPYMV_H_
|
||||
#ifndef AOM_AV1_COMMON_ENTROPYMV_H_
|
||||
#define AOM_AV1_COMMON_ENTROPYMV_H_
|
||||
|
||||
#include "config/aom_config.h"
|
||||
|
||||
|
|
@ -91,16 +91,6 @@ typedef struct {
|
|||
nmv_component comps[2];
|
||||
} nmv_context;
|
||||
|
||||
static INLINE MV_JOINT_TYPE av1_get_mv_joint(const MV *mv) {
|
||||
if (mv->row == 0) {
|
||||
return mv->col == 0 ? MV_JOINT_ZERO : MV_JOINT_HNZVZ;
|
||||
} else {
|
||||
return mv->col == 0 ? MV_JOINT_HZVNZ : MV_JOINT_HNZVNZ;
|
||||
}
|
||||
}
|
||||
|
||||
MV_CLASS_TYPE av1_get_mv_class(int z, int *offset);
|
||||
|
||||
typedef enum {
|
||||
MV_SUBPEL_NONE = -1,
|
||||
MV_SUBPEL_LOW_PRECISION = 0,
|
||||
|
|
@ -111,4 +101,4 @@ typedef enum {
|
|||
} // extern "C"
|
||||
#endif
|
||||
|
||||
#endif // AV1_COMMON_ENTROPYMV_H_
|
||||
#endif // AOM_AV1_COMMON_ENTROPYMV_H_
|
||||
|
|
|
|||
12
third_party/aom/av1/common/enums.h
vendored
12
third_party/aom/av1/common/enums.h
vendored
|
|
@ -9,8 +9,8 @@
|
|||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#ifndef AV1_COMMON_ENUMS_H_
|
||||
#define AV1_COMMON_ENUMS_H_
|
||||
#ifndef AOM_AV1_COMMON_ENUMS_H_
|
||||
#define AOM_AV1_COMMON_ENUMS_H_
|
||||
|
||||
#include "config/aom_config.h"
|
||||
|
||||
|
|
@ -274,7 +274,7 @@ typedef enum ATTRIBUTE_PACKED {
|
|||
TX_TYPES,
|
||||
} TX_TYPE;
|
||||
|
||||
typedef enum {
|
||||
typedef enum ATTRIBUTE_PACKED {
|
||||
REG_REG,
|
||||
REG_SMOOTH,
|
||||
REG_SHARP,
|
||||
|
|
@ -438,6 +438,8 @@ typedef enum ATTRIBUTE_PACKED {
|
|||
COMP_INTER_MODE_START = NEAREST_NEARESTMV,
|
||||
COMP_INTER_MODE_END = MB_MODE_COUNT,
|
||||
COMP_INTER_MODE_NUM = COMP_INTER_MODE_END - COMP_INTER_MODE_START,
|
||||
INTER_MODE_START = NEARESTMV,
|
||||
INTER_MODE_END = MB_MODE_COUNT,
|
||||
INTRA_MODES = PAETH_PRED + 1, // PAETH_PRED has to be the last intra mode.
|
||||
INTRA_INVALID = MB_MODE_COUNT // For uv_mode in inter blocks
|
||||
} PREDICTION_MODE;
|
||||
|
|
@ -478,7 +480,7 @@ typedef enum ATTRIBUTE_PACKED {
|
|||
INTERINTRA_MODES
|
||||
} INTERINTRA_MODE;
|
||||
|
||||
typedef enum {
|
||||
typedef enum ATTRIBUTE_PACKED {
|
||||
COMPOUND_AVERAGE,
|
||||
COMPOUND_WEDGE,
|
||||
COMPOUND_DIFFWTD,
|
||||
|
|
@ -614,4 +616,4 @@ typedef enum ATTRIBUTE_PACKED {
|
|||
} // extern "C"
|
||||
#endif
|
||||
|
||||
#endif // AV1_COMMON_ENUMS_H_
|
||||
#endif // AOM_AV1_COMMON_ENUMS_H_
|
||||
|
|
|
|||
22
third_party/aom/av1/common/filter.h
vendored
22
third_party/aom/av1/common/filter.h
vendored
|
|
@ -9,8 +9,8 @@
|
|||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#ifndef AV1_COMMON_FILTER_H_
|
||||
#define AV1_COMMON_FILTER_H_
|
||||
#ifndef AOM_AV1_COMMON_FILTER_H_
|
||||
#define AOM_AV1_COMMON_FILTER_H_
|
||||
|
||||
#include <assert.h>
|
||||
|
||||
|
|
@ -139,6 +139,17 @@ static const InterpFilterParams
|
|||
BILINEAR }
|
||||
};
|
||||
|
||||
// A special 2-tap bilinear filter for IntraBC chroma. IntraBC uses full pixel
|
||||
// MV for luma. If sub-sampling exists, chroma may possibly use half-pel MV.
|
||||
DECLARE_ALIGNED(256, static const int16_t, av1_intrabc_bilinear_filter[2]) = {
|
||||
64,
|
||||
64,
|
||||
};
|
||||
|
||||
static const InterpFilterParams av1_intrabc_filter_params = {
|
||||
av1_intrabc_bilinear_filter, 2, 0, BILINEAR
|
||||
};
|
||||
|
||||
DECLARE_ALIGNED(256, static const InterpKernel,
|
||||
av1_sub_pel_filters_4[SUBPEL_SHIFTS]) = {
|
||||
{ 0, 0, 0, 128, 0, 0, 0, 0 }, { 0, 0, -4, 126, 8, -2, 0, 0 },
|
||||
|
|
@ -181,6 +192,11 @@ av1_get_interp_filter_params_with_block_size(const InterpFilter interp_filter,
|
|||
return &av1_interp_filter_params_list[interp_filter];
|
||||
}
|
||||
|
||||
static INLINE const InterpFilterParams *av1_get_4tap_interp_filter_params(
|
||||
const InterpFilter interp_filter) {
|
||||
return &av1_interp_4tap[interp_filter];
|
||||
}
|
||||
|
||||
static INLINE const int16_t *av1_get_interp_filter_kernel(
|
||||
const InterpFilter interp_filter) {
|
||||
return av1_interp_filter_params_list[interp_filter].filter_ptr;
|
||||
|
|
@ -195,4 +211,4 @@ static INLINE const int16_t *av1_get_interp_filter_subpel_kernel(
|
|||
} // extern "C"
|
||||
#endif
|
||||
|
||||
#endif // AV1_COMMON_FILTER_H_
|
||||
#endif // AOM_AV1_COMMON_FILTER_H_
|
||||
|
|
|
|||
11
third_party/aom/av1/common/frame_buffers.c
vendored
11
third_party/aom/av1/common/frame_buffers.c
vendored
|
|
@ -38,6 +38,17 @@ void av1_free_internal_frame_buffers(InternalFrameBufferList *list) {
|
|||
list->int_fb = NULL;
|
||||
}
|
||||
|
||||
void av1_zero_unused_internal_frame_buffers(InternalFrameBufferList *list) {
|
||||
int i;
|
||||
|
||||
assert(list != NULL);
|
||||
|
||||
for (i = 0; i < list->num_internal_frame_buffers; ++i) {
|
||||
if (list->int_fb[i].data && !list->int_fb[i].in_use)
|
||||
memset(list->int_fb[i].data, 0, list->int_fb[i].size);
|
||||
}
|
||||
}
|
||||
|
||||
int av1_get_frame_buffer(void *cb_priv, size_t min_size,
|
||||
aom_codec_frame_buffer_t *fb) {
|
||||
int i;
|
||||
|
|
|
|||
12
third_party/aom/av1/common/frame_buffers.h
vendored
12
third_party/aom/av1/common/frame_buffers.h
vendored
|
|
@ -9,8 +9,8 @@
|
|||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#ifndef AV1_COMMON_FRAME_BUFFERS_H_
|
||||
#define AV1_COMMON_FRAME_BUFFERS_H_
|
||||
#ifndef AOM_AV1_COMMON_FRAME_BUFFERS_H_
|
||||
#define AOM_AV1_COMMON_FRAME_BUFFERS_H_
|
||||
|
||||
#include "aom/aom_frame_buffer.h"
|
||||
#include "aom/aom_integer.h"
|
||||
|
|
@ -36,6 +36,12 @@ int av1_alloc_internal_frame_buffers(InternalFrameBufferList *list);
|
|||
// Free any data allocated to the frame buffers.
|
||||
void av1_free_internal_frame_buffers(InternalFrameBufferList *list);
|
||||
|
||||
// Zeros all unused internal frame buffers. In particular, this zeros the
|
||||
// frame borders. Call this function after a sequence header change to
|
||||
// re-initialize the frame borders for the different width, height, or bit
|
||||
// depth.
|
||||
void av1_zero_unused_internal_frame_buffers(InternalFrameBufferList *list);
|
||||
|
||||
// Callback used by libaom to request an external frame buffer. |cb_priv|
|
||||
// Callback private data, which points to an InternalFrameBufferList.
|
||||
// |min_size| is the minimum size in bytes needed to decode the next frame.
|
||||
|
|
@ -51,4 +57,4 @@ int av1_release_frame_buffer(void *cb_priv, aom_codec_frame_buffer_t *fb);
|
|||
} // extern "C"
|
||||
#endif
|
||||
|
||||
#endif // AV1_COMMON_FRAME_BUFFERS_H_
|
||||
#endif // AOM_AV1_COMMON_FRAME_BUFFERS_H_
|
||||
|
|
|
|||
284
third_party/aom/av1/common/idct.c
vendored
284
third_party/aom/av1/common/idct.c
vendored
|
|
@ -31,21 +31,16 @@ int av1_get_tx_scale(const TX_SIZE tx_size) {
|
|||
// that input and output could be the same buffer.
|
||||
|
||||
// idct
|
||||
static void highbd_iwht4x4_add(const tran_low_t *input, uint8_t *dest,
|
||||
int stride, int eob, int bd) {
|
||||
void av1_highbd_iwht4x4_add(const tran_low_t *input, uint8_t *dest, int stride,
|
||||
int eob, int bd) {
|
||||
if (eob > 1)
|
||||
av1_highbd_iwht4x4_16_add(input, dest, stride, bd);
|
||||
else
|
||||
av1_highbd_iwht4x4_1_add(input, dest, stride, bd);
|
||||
}
|
||||
|
||||
static const int32_t *cast_to_int32(const tran_low_t *input) {
|
||||
assert(sizeof(int32_t) == sizeof(tran_low_t));
|
||||
return (const int32_t *)input;
|
||||
}
|
||||
|
||||
void av1_highbd_inv_txfm_add_4x4(const tran_low_t *input, uint8_t *dest,
|
||||
int stride, const TxfmParam *txfm_param) {
|
||||
void av1_highbd_inv_txfm_add_4x4_c(const tran_low_t *input, uint8_t *dest,
|
||||
int stride, const TxfmParam *txfm_param) {
|
||||
assert(av1_ext_tx_used[txfm_param->tx_set_type][txfm_param->tx_type]);
|
||||
int eob = txfm_param->eob;
|
||||
int bd = txfm_param->bd;
|
||||
|
|
@ -54,206 +49,150 @@ void av1_highbd_inv_txfm_add_4x4(const tran_low_t *input, uint8_t *dest,
|
|||
const TX_TYPE tx_type = txfm_param->tx_type;
|
||||
if (lossless) {
|
||||
assert(tx_type == DCT_DCT);
|
||||
highbd_iwht4x4_add(input, dest, stride, eob, bd);
|
||||
av1_highbd_iwht4x4_add(input, dest, stride, eob, bd);
|
||||
return;
|
||||
}
|
||||
switch (tx_type) {
|
||||
// Assembly version doesn't support some transform types, so use C version
|
||||
// for those.
|
||||
case V_DCT:
|
||||
case H_DCT:
|
||||
case V_ADST:
|
||||
case H_ADST:
|
||||
case V_FLIPADST:
|
||||
case H_FLIPADST:
|
||||
case IDTX:
|
||||
av1_inv_txfm2d_add_4x4_c(src, CONVERT_TO_SHORTPTR(dest), stride, tx_type,
|
||||
bd);
|
||||
break;
|
||||
default:
|
||||
av1_inv_txfm2d_add_4x4(src, CONVERT_TO_SHORTPTR(dest), stride, tx_type,
|
||||
bd);
|
||||
break;
|
||||
}
|
||||
|
||||
av1_inv_txfm2d_add_4x4_c(src, CONVERT_TO_SHORTPTR(dest), stride, tx_type, bd);
|
||||
}
|
||||
|
||||
static void highbd_inv_txfm_add_4x8(const tran_low_t *input, uint8_t *dest,
|
||||
int stride, const TxfmParam *txfm_param) {
|
||||
void av1_highbd_inv_txfm_add_4x8(const tran_low_t *input, uint8_t *dest,
|
||||
int stride, const TxfmParam *txfm_param) {
|
||||
assert(av1_ext_tx_used[txfm_param->tx_set_type][txfm_param->tx_type]);
|
||||
const int32_t *src = cast_to_int32(input);
|
||||
av1_inv_txfm2d_add_4x8(src, CONVERT_TO_SHORTPTR(dest), stride,
|
||||
txfm_param->tx_type, txfm_param->bd);
|
||||
av1_inv_txfm2d_add_4x8_c(src, CONVERT_TO_SHORTPTR(dest), stride,
|
||||
txfm_param->tx_type, txfm_param->bd);
|
||||
}
|
||||
|
||||
static void highbd_inv_txfm_add_8x4(const tran_low_t *input, uint8_t *dest,
|
||||
int stride, const TxfmParam *txfm_param) {
|
||||
void av1_highbd_inv_txfm_add_8x4(const tran_low_t *input, uint8_t *dest,
|
||||
int stride, const TxfmParam *txfm_param) {
|
||||
assert(av1_ext_tx_used[txfm_param->tx_set_type][txfm_param->tx_type]);
|
||||
const int32_t *src = cast_to_int32(input);
|
||||
av1_inv_txfm2d_add_8x4(src, CONVERT_TO_SHORTPTR(dest), stride,
|
||||
txfm_param->tx_type, txfm_param->bd);
|
||||
}
|
||||
|
||||
static void highbd_inv_txfm_add_8x16(const tran_low_t *input, uint8_t *dest,
|
||||
int stride, const TxfmParam *txfm_param) {
|
||||
const int32_t *src = cast_to_int32(input);
|
||||
av1_inv_txfm2d_add_8x16(src, CONVERT_TO_SHORTPTR(dest), stride,
|
||||
txfm_param->tx_type, txfm_param->bd);
|
||||
}
|
||||
|
||||
static void highbd_inv_txfm_add_16x8(const tran_low_t *input, uint8_t *dest,
|
||||
int stride, const TxfmParam *txfm_param) {
|
||||
const int32_t *src = cast_to_int32(input);
|
||||
av1_inv_txfm2d_add_16x8(src, CONVERT_TO_SHORTPTR(dest), stride,
|
||||
txfm_param->tx_type, txfm_param->bd);
|
||||
}
|
||||
|
||||
static void highbd_inv_txfm_add_16x32(const tran_low_t *input, uint8_t *dest,
|
||||
int stride, const TxfmParam *txfm_param) {
|
||||
const int32_t *src = cast_to_int32(input);
|
||||
av1_inv_txfm2d_add_16x32(src, CONVERT_TO_SHORTPTR(dest), stride,
|
||||
av1_inv_txfm2d_add_8x4_c(src, CONVERT_TO_SHORTPTR(dest), stride,
|
||||
txfm_param->tx_type, txfm_param->bd);
|
||||
}
|
||||
|
||||
static void highbd_inv_txfm_add_32x16(const tran_low_t *input, uint8_t *dest,
|
||||
int stride, const TxfmParam *txfm_param) {
|
||||
void av1_highbd_inv_txfm_add_16x32(const tran_low_t *input, uint8_t *dest,
|
||||
int stride, const TxfmParam *txfm_param) {
|
||||
const int32_t *src = cast_to_int32(input);
|
||||
av1_inv_txfm2d_add_32x16(src, CONVERT_TO_SHORTPTR(dest), stride,
|
||||
txfm_param->tx_type, txfm_param->bd);
|
||||
av1_inv_txfm2d_add_16x32_c(src, CONVERT_TO_SHORTPTR(dest), stride,
|
||||
txfm_param->tx_type, txfm_param->bd);
|
||||
}
|
||||
|
||||
static void highbd_inv_txfm_add_16x4(const tran_low_t *input, uint8_t *dest,
|
||||
int stride, const TxfmParam *txfm_param) {
|
||||
void av1_highbd_inv_txfm_add_32x16(const tran_low_t *input, uint8_t *dest,
|
||||
int stride, const TxfmParam *txfm_param) {
|
||||
const int32_t *src = cast_to_int32(input);
|
||||
av1_inv_txfm2d_add_16x4(src, CONVERT_TO_SHORTPTR(dest), stride,
|
||||
txfm_param->tx_type, txfm_param->bd);
|
||||
av1_inv_txfm2d_add_32x16_c(src, CONVERT_TO_SHORTPTR(dest), stride,
|
||||
txfm_param->tx_type, txfm_param->bd);
|
||||
}
|
||||
|
||||
static void highbd_inv_txfm_add_4x16(const tran_low_t *input, uint8_t *dest,
|
||||
int stride, const TxfmParam *txfm_param) {
|
||||
void av1_highbd_inv_txfm_add_16x4(const tran_low_t *input, uint8_t *dest,
|
||||
int stride, const TxfmParam *txfm_param) {
|
||||
const int32_t *src = cast_to_int32(input);
|
||||
av1_inv_txfm2d_add_4x16(src, CONVERT_TO_SHORTPTR(dest), stride,
|
||||
txfm_param->tx_type, txfm_param->bd);
|
||||
av1_inv_txfm2d_add_16x4_c(src, CONVERT_TO_SHORTPTR(dest), stride,
|
||||
txfm_param->tx_type, txfm_param->bd);
|
||||
}
|
||||
|
||||
static void highbd_inv_txfm_add_32x8(const tran_low_t *input, uint8_t *dest,
|
||||
int stride, const TxfmParam *txfm_param) {
|
||||
void av1_highbd_inv_txfm_add_4x16(const tran_low_t *input, uint8_t *dest,
|
||||
int stride, const TxfmParam *txfm_param) {
|
||||
const int32_t *src = cast_to_int32(input);
|
||||
av1_inv_txfm2d_add_32x8(src, CONVERT_TO_SHORTPTR(dest), stride,
|
||||
txfm_param->tx_type, txfm_param->bd);
|
||||
av1_inv_txfm2d_add_4x16_c(src, CONVERT_TO_SHORTPTR(dest), stride,
|
||||
txfm_param->tx_type, txfm_param->bd);
|
||||
}
|
||||
|
||||
static void highbd_inv_txfm_add_8x32(const tran_low_t *input, uint8_t *dest,
|
||||
int stride, const TxfmParam *txfm_param) {
|
||||
void av1_highbd_inv_txfm_add_32x8(const tran_low_t *input, uint8_t *dest,
|
||||
int stride, const TxfmParam *txfm_param) {
|
||||
const int32_t *src = cast_to_int32(input);
|
||||
av1_inv_txfm2d_add_8x32(src, CONVERT_TO_SHORTPTR(dest), stride,
|
||||
txfm_param->tx_type, txfm_param->bd);
|
||||
av1_inv_txfm2d_add_32x8_c(src, CONVERT_TO_SHORTPTR(dest), stride,
|
||||
txfm_param->tx_type, txfm_param->bd);
|
||||
}
|
||||
|
||||
static void highbd_inv_txfm_add_32x64(const tran_low_t *input, uint8_t *dest,
|
||||
int stride, const TxfmParam *txfm_param) {
|
||||
void av1_highbd_inv_txfm_add_8x32(const tran_low_t *input, uint8_t *dest,
|
||||
int stride, const TxfmParam *txfm_param) {
|
||||
const int32_t *src = cast_to_int32(input);
|
||||
av1_inv_txfm2d_add_32x64(src, CONVERT_TO_SHORTPTR(dest), stride,
|
||||
txfm_param->tx_type, txfm_param->bd);
|
||||
av1_inv_txfm2d_add_8x32_c(src, CONVERT_TO_SHORTPTR(dest), stride,
|
||||
txfm_param->tx_type, txfm_param->bd);
|
||||
}
|
||||
|
||||
static void highbd_inv_txfm_add_64x32(const tran_low_t *input, uint8_t *dest,
|
||||
int stride, const TxfmParam *txfm_param) {
|
||||
void av1_highbd_inv_txfm_add_32x64(const tran_low_t *input, uint8_t *dest,
|
||||
int stride, const TxfmParam *txfm_param) {
|
||||
const int32_t *src = cast_to_int32(input);
|
||||
av1_inv_txfm2d_add_64x32(src, CONVERT_TO_SHORTPTR(dest), stride,
|
||||
txfm_param->tx_type, txfm_param->bd);
|
||||
av1_inv_txfm2d_add_32x64_c(src, CONVERT_TO_SHORTPTR(dest), stride,
|
||||
txfm_param->tx_type, txfm_param->bd);
|
||||
}
|
||||
|
||||
static void highbd_inv_txfm_add_16x64(const tran_low_t *input, uint8_t *dest,
|
||||
int stride, const TxfmParam *txfm_param) {
|
||||
void av1_highbd_inv_txfm_add_64x32(const tran_low_t *input, uint8_t *dest,
|
||||
int stride, const TxfmParam *txfm_param) {
|
||||
const int32_t *src = cast_to_int32(input);
|
||||
av1_inv_txfm2d_add_16x64(src, CONVERT_TO_SHORTPTR(dest), stride,
|
||||
txfm_param->tx_type, txfm_param->bd);
|
||||
av1_inv_txfm2d_add_64x32_c(src, CONVERT_TO_SHORTPTR(dest), stride,
|
||||
txfm_param->tx_type, txfm_param->bd);
|
||||
}
|
||||
|
||||
static void highbd_inv_txfm_add_64x16(const tran_low_t *input, uint8_t *dest,
|
||||
int stride, const TxfmParam *txfm_param) {
|
||||
void av1_highbd_inv_txfm_add_16x64(const tran_low_t *input, uint8_t *dest,
|
||||
int stride, const TxfmParam *txfm_param) {
|
||||
const int32_t *src = cast_to_int32(input);
|
||||
av1_inv_txfm2d_add_64x16(src, CONVERT_TO_SHORTPTR(dest), stride,
|
||||
txfm_param->tx_type, txfm_param->bd);
|
||||
av1_inv_txfm2d_add_16x64_c(src, CONVERT_TO_SHORTPTR(dest), stride,
|
||||
txfm_param->tx_type, txfm_param->bd);
|
||||
}
|
||||
|
||||
static void highbd_inv_txfm_add_8x8(const tran_low_t *input, uint8_t *dest,
|
||||
int stride, const TxfmParam *txfm_param) {
|
||||
void av1_highbd_inv_txfm_add_64x16(const tran_low_t *input, uint8_t *dest,
|
||||
int stride, const TxfmParam *txfm_param) {
|
||||
const int32_t *src = cast_to_int32(input);
|
||||
av1_inv_txfm2d_add_64x16_c(src, CONVERT_TO_SHORTPTR(dest), stride,
|
||||
txfm_param->tx_type, txfm_param->bd);
|
||||
}
|
||||
|
||||
void av1_highbd_inv_txfm_add_8x8_c(const tran_low_t *input, uint8_t *dest,
|
||||
int stride, const TxfmParam *txfm_param) {
|
||||
int bd = txfm_param->bd;
|
||||
const TX_TYPE tx_type = txfm_param->tx_type;
|
||||
const int32_t *src = cast_to_int32(input);
|
||||
switch (tx_type) {
|
||||
// Assembly version doesn't support some transform types, so use C version
|
||||
// for those.
|
||||
case V_DCT:
|
||||
case H_DCT:
|
||||
case V_ADST:
|
||||
case H_ADST:
|
||||
case V_FLIPADST:
|
||||
case H_FLIPADST:
|
||||
case IDTX:
|
||||
av1_inv_txfm2d_add_8x8_c(src, CONVERT_TO_SHORTPTR(dest), stride, tx_type,
|
||||
bd);
|
||||
break;
|
||||
default:
|
||||
av1_inv_txfm2d_add_8x8(src, CONVERT_TO_SHORTPTR(dest), stride, tx_type,
|
||||
bd);
|
||||
break;
|
||||
}
|
||||
|
||||
av1_inv_txfm2d_add_8x8_c(src, CONVERT_TO_SHORTPTR(dest), stride, tx_type, bd);
|
||||
}
|
||||
|
||||
static void highbd_inv_txfm_add_16x16(const tran_low_t *input, uint8_t *dest,
|
||||
int stride, const TxfmParam *txfm_param) {
|
||||
void av1_highbd_inv_txfm_add_16x16_c(const tran_low_t *input, uint8_t *dest,
|
||||
int stride, const TxfmParam *txfm_param) {
|
||||
int bd = txfm_param->bd;
|
||||
const TX_TYPE tx_type = txfm_param->tx_type;
|
||||
const int32_t *src = cast_to_int32(input);
|
||||
switch (tx_type) {
|
||||
// Assembly version doesn't support some transform types, so use C version
|
||||
// for those.
|
||||
case V_DCT:
|
||||
case H_DCT:
|
||||
case V_ADST:
|
||||
case H_ADST:
|
||||
case V_FLIPADST:
|
||||
case H_FLIPADST:
|
||||
case IDTX:
|
||||
av1_inv_txfm2d_add_16x16_c(src, CONVERT_TO_SHORTPTR(dest), stride,
|
||||
tx_type, bd);
|
||||
break;
|
||||
default:
|
||||
av1_inv_txfm2d_add_16x16(src, CONVERT_TO_SHORTPTR(dest), stride, tx_type,
|
||||
bd);
|
||||
break;
|
||||
}
|
||||
|
||||
av1_inv_txfm2d_add_16x16_c(src, CONVERT_TO_SHORTPTR(dest), stride, tx_type,
|
||||
bd);
|
||||
}
|
||||
|
||||
static void highbd_inv_txfm_add_32x32(const tran_low_t *input, uint8_t *dest,
|
||||
int stride, const TxfmParam *txfm_param) {
|
||||
void av1_highbd_inv_txfm_add_8x16_c(const tran_low_t *input, uint8_t *dest,
|
||||
int stride, const TxfmParam *txfm_param) {
|
||||
const int32_t *src = cast_to_int32(input);
|
||||
av1_inv_txfm2d_add_8x16_c(src, CONVERT_TO_SHORTPTR(dest), stride,
|
||||
txfm_param->tx_type, txfm_param->bd);
|
||||
}
|
||||
|
||||
void av1_highbd_inv_txfm_add_16x8_c(const tran_low_t *input, uint8_t *dest,
|
||||
int stride, const TxfmParam *txfm_param) {
|
||||
const int32_t *src = cast_to_int32(input);
|
||||
av1_inv_txfm2d_add_16x8_c(src, CONVERT_TO_SHORTPTR(dest), stride,
|
||||
txfm_param->tx_type, txfm_param->bd);
|
||||
}
|
||||
|
||||
void av1_highbd_inv_txfm_add_32x32_c(const tran_low_t *input, uint8_t *dest,
|
||||
int stride, const TxfmParam *txfm_param) {
|
||||
const int bd = txfm_param->bd;
|
||||
const TX_TYPE tx_type = txfm_param->tx_type;
|
||||
const int32_t *src = cast_to_int32(input);
|
||||
switch (tx_type) {
|
||||
case DCT_DCT:
|
||||
av1_inv_txfm2d_add_32x32(src, CONVERT_TO_SHORTPTR(dest), stride, tx_type,
|
||||
bd);
|
||||
break;
|
||||
// Assembly version doesn't support IDTX, so use C version for it.
|
||||
case IDTX:
|
||||
av1_inv_txfm2d_add_32x32_c(src, CONVERT_TO_SHORTPTR(dest), stride,
|
||||
tx_type, bd);
|
||||
break;
|
||||
|
||||
default: assert(0);
|
||||
}
|
||||
av1_inv_txfm2d_add_32x32_c(src, CONVERT_TO_SHORTPTR(dest), stride, tx_type,
|
||||
bd);
|
||||
}
|
||||
|
||||
static void highbd_inv_txfm_add_64x64(const tran_low_t *input, uint8_t *dest,
|
||||
int stride, const TxfmParam *txfm_param) {
|
||||
void av1_highbd_inv_txfm_add_64x64_c(const tran_low_t *input, uint8_t *dest,
|
||||
int stride, const TxfmParam *txfm_param) {
|
||||
const int bd = txfm_param->bd;
|
||||
const TX_TYPE tx_type = txfm_param->tx_type;
|
||||
const int32_t *src = cast_to_int32(input);
|
||||
assert(tx_type == DCT_DCT);
|
||||
av1_inv_txfm2d_add_64x64(src, CONVERT_TO_SHORTPTR(dest), stride, tx_type, bd);
|
||||
av1_inv_txfm2d_add_64x64_c(src, CONVERT_TO_SHORTPTR(dest), stride, tx_type,
|
||||
bd);
|
||||
}
|
||||
|
||||
static void init_txfm_param(const MACROBLOCKD *xd, int plane, TX_SIZE tx_size,
|
||||
|
|
@ -270,70 +209,70 @@ static void init_txfm_param(const MACROBLOCKD *xd, int plane, TX_SIZE tx_size,
|
|||
txfm_param->tx_size, is_inter_block(xd->mi[0]), reduced_tx_set);
|
||||
}
|
||||
|
||||
static void highbd_inv_txfm_add(const tran_low_t *input, uint8_t *dest,
|
||||
int stride, const TxfmParam *txfm_param) {
|
||||
void av1_highbd_inv_txfm_add_c(const tran_low_t *input, uint8_t *dest,
|
||||
int stride, const TxfmParam *txfm_param) {
|
||||
assert(av1_ext_tx_used[txfm_param->tx_set_type][txfm_param->tx_type]);
|
||||
const TX_SIZE tx_size = txfm_param->tx_size;
|
||||
switch (tx_size) {
|
||||
case TX_32X32:
|
||||
highbd_inv_txfm_add_32x32(input, dest, stride, txfm_param);
|
||||
av1_highbd_inv_txfm_add_32x32_c(input, dest, stride, txfm_param);
|
||||
break;
|
||||
case TX_16X16:
|
||||
highbd_inv_txfm_add_16x16(input, dest, stride, txfm_param);
|
||||
av1_highbd_inv_txfm_add_16x16_c(input, dest, stride, txfm_param);
|
||||
break;
|
||||
case TX_8X8:
|
||||
highbd_inv_txfm_add_8x8(input, dest, stride, txfm_param);
|
||||
av1_highbd_inv_txfm_add_8x8_c(input, dest, stride, txfm_param);
|
||||
break;
|
||||
case TX_4X8:
|
||||
highbd_inv_txfm_add_4x8(input, dest, stride, txfm_param);
|
||||
av1_highbd_inv_txfm_add_4x8(input, dest, stride, txfm_param);
|
||||
break;
|
||||
case TX_8X4:
|
||||
highbd_inv_txfm_add_8x4(input, dest, stride, txfm_param);
|
||||
av1_highbd_inv_txfm_add_8x4(input, dest, stride, txfm_param);
|
||||
break;
|
||||
case TX_8X16:
|
||||
highbd_inv_txfm_add_8x16(input, dest, stride, txfm_param);
|
||||
av1_highbd_inv_txfm_add_8x16_c(input, dest, stride, txfm_param);
|
||||
break;
|
||||
case TX_16X8:
|
||||
highbd_inv_txfm_add_16x8(input, dest, stride, txfm_param);
|
||||
av1_highbd_inv_txfm_add_16x8_c(input, dest, stride, txfm_param);
|
||||
break;
|
||||
case TX_16X32:
|
||||
highbd_inv_txfm_add_16x32(input, dest, stride, txfm_param);
|
||||
av1_highbd_inv_txfm_add_16x32(input, dest, stride, txfm_param);
|
||||
break;
|
||||
case TX_32X16:
|
||||
highbd_inv_txfm_add_32x16(input, dest, stride, txfm_param);
|
||||
av1_highbd_inv_txfm_add_32x16(input, dest, stride, txfm_param);
|
||||
break;
|
||||
case TX_64X64:
|
||||
highbd_inv_txfm_add_64x64(input, dest, stride, txfm_param);
|
||||
av1_highbd_inv_txfm_add_64x64_c(input, dest, stride, txfm_param);
|
||||
break;
|
||||
case TX_32X64:
|
||||
highbd_inv_txfm_add_32x64(input, dest, stride, txfm_param);
|
||||
av1_highbd_inv_txfm_add_32x64(input, dest, stride, txfm_param);
|
||||
break;
|
||||
case TX_64X32:
|
||||
highbd_inv_txfm_add_64x32(input, dest, stride, txfm_param);
|
||||
av1_highbd_inv_txfm_add_64x32(input, dest, stride, txfm_param);
|
||||
break;
|
||||
case TX_16X64:
|
||||
highbd_inv_txfm_add_16x64(input, dest, stride, txfm_param);
|
||||
av1_highbd_inv_txfm_add_16x64(input, dest, stride, txfm_param);
|
||||
break;
|
||||
case TX_64X16:
|
||||
highbd_inv_txfm_add_64x16(input, dest, stride, txfm_param);
|
||||
av1_highbd_inv_txfm_add_64x16(input, dest, stride, txfm_param);
|
||||
break;
|
||||
case TX_4X4:
|
||||
// this is like av1_short_idct4x4 but has a special case around eob<=1
|
||||
// which is significant (not just an optimization) for the lossless
|
||||
// case.
|
||||
av1_highbd_inv_txfm_add_4x4(input, dest, stride, txfm_param);
|
||||
av1_highbd_inv_txfm_add_4x4_c(input, dest, stride, txfm_param);
|
||||
break;
|
||||
case TX_16X4:
|
||||
highbd_inv_txfm_add_16x4(input, dest, stride, txfm_param);
|
||||
av1_highbd_inv_txfm_add_16x4(input, dest, stride, txfm_param);
|
||||
break;
|
||||
case TX_4X16:
|
||||
highbd_inv_txfm_add_4x16(input, dest, stride, txfm_param);
|
||||
av1_highbd_inv_txfm_add_4x16(input, dest, stride, txfm_param);
|
||||
break;
|
||||
case TX_8X32:
|
||||
highbd_inv_txfm_add_8x32(input, dest, stride, txfm_param);
|
||||
av1_highbd_inv_txfm_add_8x32(input, dest, stride, txfm_param);
|
||||
break;
|
||||
case TX_32X8:
|
||||
highbd_inv_txfm_add_32x8(input, dest, stride, txfm_param);
|
||||
av1_highbd_inv_txfm_add_32x8(input, dest, stride, txfm_param);
|
||||
break;
|
||||
default: assert(0 && "Invalid transform size"); break;
|
||||
}
|
||||
|
|
@ -352,7 +291,8 @@ void av1_inv_txfm_add_c(const tran_low_t *dqcoeff, uint8_t *dst, int stride,
|
|||
}
|
||||
}
|
||||
|
||||
highbd_inv_txfm_add(dqcoeff, CONVERT_TO_BYTEPTR(tmp), tmp_stride, txfm_param);
|
||||
av1_highbd_inv_txfm_add(dqcoeff, CONVERT_TO_BYTEPTR(tmp), tmp_stride,
|
||||
txfm_param);
|
||||
|
||||
for (int r = 0; r < h; ++r) {
|
||||
for (int c = 0; c < w; ++c) {
|
||||
|
|
@ -375,7 +315,7 @@ void av1_inverse_transform_block(const MACROBLOCKD *xd,
|
|||
assert(av1_ext_tx_used[txfm_param.tx_set_type][txfm_param.tx_type]);
|
||||
|
||||
if (txfm_param.is_hbd) {
|
||||
highbd_inv_txfm_add(dqcoeff, dst, stride, &txfm_param);
|
||||
av1_highbd_inv_txfm_add(dqcoeff, dst, stride, &txfm_param);
|
||||
} else {
|
||||
av1_inv_txfm_add(dqcoeff, dst, stride, &txfm_param);
|
||||
}
|
||||
|
|
|
|||
31
third_party/aom/av1/common/idct.h
vendored
31
third_party/aom/av1/common/idct.h
vendored
|
|
@ -9,8 +9,8 @@
|
|||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#ifndef AV1_COMMON_IDCT_H_
|
||||
#define AV1_COMMON_IDCT_H_
|
||||
#ifndef AOM_AV1_COMMON_IDCT_H_
|
||||
#define AOM_AV1_COMMON_IDCT_H_
|
||||
|
||||
#include "config/aom_config.h"
|
||||
|
||||
|
|
@ -36,11 +36,32 @@ void av1_inverse_transform_block(const MACROBLOCKD *xd,
|
|||
const tran_low_t *dqcoeff, int plane,
|
||||
TX_TYPE tx_type, TX_SIZE tx_size, uint8_t *dst,
|
||||
int stride, int eob, int reduced_tx_set);
|
||||
void av1_highbd_iwht4x4_add(const tran_low_t *input, uint8_t *dest, int stride,
|
||||
int eob, int bd);
|
||||
|
||||
static INLINE const int32_t *cast_to_int32(const tran_low_t *input) {
|
||||
assert(sizeof(int32_t) == sizeof(tran_low_t));
|
||||
return (const int32_t *)input;
|
||||
}
|
||||
|
||||
typedef void(highbd_inv_txfm_add)(const tran_low_t *input, uint8_t *dest,
|
||||
int stride, const TxfmParam *param);
|
||||
|
||||
highbd_inv_txfm_add av1_highbd_inv_txfm_add_4x8;
|
||||
highbd_inv_txfm_add av1_highbd_inv_txfm_add_8x4;
|
||||
highbd_inv_txfm_add av1_highbd_inv_txfm_add_16x32;
|
||||
highbd_inv_txfm_add av1_highbd_inv_txfm_add_32x16;
|
||||
highbd_inv_txfm_add av1_highbd_inv_txfm_add_32x64;
|
||||
highbd_inv_txfm_add av1_highbd_inv_txfm_add_64x32;
|
||||
highbd_inv_txfm_add av1_highbd_inv_txfm_add_16x64;
|
||||
highbd_inv_txfm_add av1_highbd_inv_txfm_add_64x16;
|
||||
highbd_inv_txfm_add av1_highbd_inv_txfm_add_16x4;
|
||||
highbd_inv_txfm_add av1_highbd_inv_txfm_add_4x16;
|
||||
highbd_inv_txfm_add av1_highbd_inv_txfm_add_8x32;
|
||||
highbd_inv_txfm_add av1_highbd_inv_txfm_add_32x8;
|
||||
|
||||
void av1_highbd_inv_txfm_add_4x4(const tran_low_t *input, uint8_t *dest,
|
||||
int stride, const TxfmParam *param);
|
||||
#ifdef __cplusplus
|
||||
} // extern "C"
|
||||
#endif
|
||||
|
||||
#endif // AV1_COMMON_IDCT_H_
|
||||
#endif // AOM_AV1_COMMON_IDCT_H_
|
||||
|
|
|
|||
8
third_party/aom/av1/common/mv.h
vendored
8
third_party/aom/av1/common/mv.h
vendored
|
|
@ -9,8 +9,8 @@
|
|||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#ifndef AV1_COMMON_MV_H_
|
||||
#define AV1_COMMON_MV_H_
|
||||
#ifndef AOM_AV1_COMMON_MV_H_
|
||||
#define AOM_AV1_COMMON_MV_H_
|
||||
|
||||
#include "av1/common/common.h"
|
||||
#include "av1/common/common_data.h"
|
||||
|
|
@ -56,7 +56,7 @@ typedef struct mv32 {
|
|||
#define WARPEDDIFF_PREC_BITS (WARPEDMODEL_PREC_BITS - WARPEDPIXEL_PREC_BITS)
|
||||
|
||||
/* clang-format off */
|
||||
typedef enum {
|
||||
typedef enum ATTRIBUTE_PACKED {
|
||||
IDENTITY = 0, // identity transformation, 0-parameter
|
||||
TRANSLATION = 1, // translational motion 2-parameter
|
||||
ROTZOOM = 2, // simplified affine with rotation + zoom only, 4-parameter
|
||||
|
|
@ -298,4 +298,4 @@ static INLINE void clamp_mv(MV *mv, int min_col, int max_col, int min_row,
|
|||
} // extern "C"
|
||||
#endif
|
||||
|
||||
#endif // AV1_COMMON_MV_H_
|
||||
#endif // AOM_AV1_COMMON_MV_H_
|
||||
|
|
|
|||
379
third_party/aom/av1/common/mvref_common.c
vendored
379
third_party/aom/av1/common/mvref_common.c
vendored
|
|
@ -27,16 +27,19 @@ static void get_mv_projection(MV *output, MV ref, int num, int den) {
|
|||
den = AOMMIN(den, MAX_FRAME_DISTANCE);
|
||||
num = num > 0 ? AOMMIN(num, MAX_FRAME_DISTANCE)
|
||||
: AOMMAX(num, -MAX_FRAME_DISTANCE);
|
||||
int mv_row = ROUND_POWER_OF_TWO_SIGNED(ref.row * num * div_mult[den], 14);
|
||||
int mv_col = ROUND_POWER_OF_TWO_SIGNED(ref.col * num * div_mult[den], 14);
|
||||
const int mv_row =
|
||||
ROUND_POWER_OF_TWO_SIGNED(ref.row * num * div_mult[den], 14);
|
||||
const int mv_col =
|
||||
ROUND_POWER_OF_TWO_SIGNED(ref.col * num * div_mult[den], 14);
|
||||
const int clamp_max = MV_UPP - 1;
|
||||
const int clamp_min = MV_LOW + 1;
|
||||
output->row = (int16_t)clamp(mv_row, clamp_min, clamp_max);
|
||||
output->col = (int16_t)clamp(mv_col, clamp_min, clamp_max);
|
||||
}
|
||||
|
||||
void av1_copy_frame_mvs(const AV1_COMMON *const cm, MB_MODE_INFO *mi,
|
||||
int mi_row, int mi_col, int x_mis, int y_mis) {
|
||||
void av1_copy_frame_mvs(const AV1_COMMON *const cm,
|
||||
const MB_MODE_INFO *const mi, int mi_row, int mi_col,
|
||||
int x_mis, int y_mis) {
|
||||
const int frame_mvs_stride = ROUND_POWER_OF_TWO(cm->mi_cols, 1);
|
||||
MV_REF *frame_mvs =
|
||||
cm->cur_frame->mvs + (mi_row >> 1) * frame_mvs_stride + (mi_col >> 1);
|
||||
|
|
@ -141,38 +144,37 @@ static void scan_row_mbmi(const AV1_COMMON *cm, const MACROBLOCKD *xd,
|
|||
uint8_t *ref_match_count, uint8_t *newmv_count,
|
||||
int_mv *gm_mv_candidates, int max_row_offset,
|
||||
int *processed_rows) {
|
||||
int end_mi = AOMMIN(xd->n8_w, cm->mi_cols - mi_col);
|
||||
int end_mi = AOMMIN(xd->n4_w, cm->mi_cols - mi_col);
|
||||
end_mi = AOMMIN(end_mi, mi_size_wide[BLOCK_64X64]);
|
||||
const int n8_w_8 = mi_size_wide[BLOCK_8X8];
|
||||
const int n8_w_16 = mi_size_wide[BLOCK_16X16];
|
||||
int i;
|
||||
int col_offset = 0;
|
||||
const int shift = 0;
|
||||
// TODO(jingning): Revisit this part after cb4x4 is stable.
|
||||
if (abs(row_offset) > 1) {
|
||||
col_offset = 1;
|
||||
if ((mi_col & 0x01) && xd->n8_w < n8_w_8) --col_offset;
|
||||
if ((mi_col & 0x01) && xd->n4_w < n8_w_8) --col_offset;
|
||||
}
|
||||
const int use_step_16 = (xd->n8_w >= 16);
|
||||
const int use_step_16 = (xd->n4_w >= 16);
|
||||
MB_MODE_INFO **const candidate_mi0 = xd->mi + row_offset * xd->mi_stride;
|
||||
(void)mi_row;
|
||||
|
||||
for (i = 0; i < end_mi;) {
|
||||
const MB_MODE_INFO *const candidate = candidate_mi0[col_offset + i];
|
||||
const int candidate_bsize = candidate->sb_type;
|
||||
const int n8_w = mi_size_wide[candidate_bsize];
|
||||
int len = AOMMIN(xd->n8_w, n8_w);
|
||||
const int n4_w = mi_size_wide[candidate_bsize];
|
||||
int len = AOMMIN(xd->n4_w, n4_w);
|
||||
if (use_step_16)
|
||||
len = AOMMAX(n8_w_16, len);
|
||||
else if (abs(row_offset) > 1)
|
||||
len = AOMMAX(len, n8_w_8);
|
||||
|
||||
int weight = 2;
|
||||
if (xd->n8_w >= n8_w_8 && xd->n8_w <= n8_w) {
|
||||
if (xd->n4_w >= n8_w_8 && xd->n4_w <= n4_w) {
|
||||
int inc = AOMMIN(-max_row_offset + row_offset + 1,
|
||||
mi_size_high[candidate_bsize]);
|
||||
// Obtain range used in weight calculation.
|
||||
weight = AOMMAX(weight, (inc << shift));
|
||||
weight = AOMMAX(weight, inc);
|
||||
// Update processed rows.
|
||||
*processed_rows = inc - row_offset - 1;
|
||||
}
|
||||
|
|
@ -192,37 +194,36 @@ static void scan_col_mbmi(const AV1_COMMON *cm, const MACROBLOCKD *xd,
|
|||
uint8_t *ref_match_count, uint8_t *newmv_count,
|
||||
int_mv *gm_mv_candidates, int max_col_offset,
|
||||
int *processed_cols) {
|
||||
int end_mi = AOMMIN(xd->n8_h, cm->mi_rows - mi_row);
|
||||
int end_mi = AOMMIN(xd->n4_h, cm->mi_rows - mi_row);
|
||||
end_mi = AOMMIN(end_mi, mi_size_high[BLOCK_64X64]);
|
||||
const int n8_h_8 = mi_size_high[BLOCK_8X8];
|
||||
const int n8_h_16 = mi_size_high[BLOCK_16X16];
|
||||
int i;
|
||||
int row_offset = 0;
|
||||
const int shift = 0;
|
||||
if (abs(col_offset) > 1) {
|
||||
row_offset = 1;
|
||||
if ((mi_row & 0x01) && xd->n8_h < n8_h_8) --row_offset;
|
||||
if ((mi_row & 0x01) && xd->n4_h < n8_h_8) --row_offset;
|
||||
}
|
||||
const int use_step_16 = (xd->n8_h >= 16);
|
||||
const int use_step_16 = (xd->n4_h >= 16);
|
||||
(void)mi_col;
|
||||
|
||||
for (i = 0; i < end_mi;) {
|
||||
const MB_MODE_INFO *const candidate =
|
||||
xd->mi[(row_offset + i) * xd->mi_stride + col_offset];
|
||||
const int candidate_bsize = candidate->sb_type;
|
||||
const int n8_h = mi_size_high[candidate_bsize];
|
||||
int len = AOMMIN(xd->n8_h, n8_h);
|
||||
const int n4_h = mi_size_high[candidate_bsize];
|
||||
int len = AOMMIN(xd->n4_h, n4_h);
|
||||
if (use_step_16)
|
||||
len = AOMMAX(n8_h_16, len);
|
||||
else if (abs(col_offset) > 1)
|
||||
len = AOMMAX(len, n8_h_8);
|
||||
|
||||
int weight = 2;
|
||||
if (xd->n8_h >= n8_h_8 && xd->n8_h <= n8_h) {
|
||||
if (xd->n4_h >= n8_h_8 && xd->n4_h <= n4_h) {
|
||||
int inc = AOMMIN(-max_col_offset + col_offset + 1,
|
||||
mi_size_wide[candidate_bsize]);
|
||||
// Obtain range used in weight calculation.
|
||||
weight = AOMMAX(weight, (inc << shift));
|
||||
weight = AOMMAX(weight, inc);
|
||||
// Update processed cols.
|
||||
*processed_cols = inc - col_offset - 1;
|
||||
}
|
||||
|
|
@ -248,7 +249,7 @@ static void scan_blk_mbmi(const AV1_COMMON *cm, const MACROBLOCKD *xd,
|
|||
mi_pos.row = row_offset;
|
||||
mi_pos.col = col_offset;
|
||||
|
||||
if (is_inside(tile, mi_col, mi_row, cm->mi_rows, &mi_pos)) {
|
||||
if (is_inside(tile, mi_col, mi_row, &mi_pos)) {
|
||||
const MB_MODE_INFO *const candidate =
|
||||
xd->mi[mi_pos.row * xd->mi_stride + mi_pos.col];
|
||||
const int len = mi_size_wide[BLOCK_8X8];
|
||||
|
|
@ -290,19 +291,19 @@ static int has_top_right(const AV1_COMMON *cm, const MACROBLOCKD *xd,
|
|||
|
||||
// The left hand of two vertical rectangles always has a top right (as the
|
||||
// block above will have been decoded)
|
||||
if (xd->n8_w < xd->n8_h)
|
||||
if (xd->n4_w < xd->n4_h)
|
||||
if (!xd->is_sec_rect) has_tr = 1;
|
||||
|
||||
// The bottom of two horizontal rectangles never has a top right (as the block
|
||||
// to the right won't have been decoded)
|
||||
if (xd->n8_w > xd->n8_h)
|
||||
if (xd->n4_w > xd->n4_h)
|
||||
if (xd->is_sec_rect) has_tr = 0;
|
||||
|
||||
// The bottom left square of a Vertical A (in the old format) does
|
||||
// not have a top right as it is decoded before the right hand
|
||||
// rectangle of the partition
|
||||
if (xd->mi[0]->partition == PARTITION_VERT_A) {
|
||||
if (xd->n8_w == xd->n8_h)
|
||||
if (xd->n4_w == xd->n4_h)
|
||||
if (mask_row & bs) has_tr = 0;
|
||||
}
|
||||
|
||||
|
|
@ -335,7 +336,7 @@ static int add_tpl_ref_mv(const AV1_COMMON *cm, const MACROBLOCKD *xd,
|
|||
mi_pos.row = (mi_row & 0x01) ? blk_row : blk_row + 1;
|
||||
mi_pos.col = (mi_col & 0x01) ? blk_col : blk_col + 1;
|
||||
|
||||
if (!is_inside(&xd->tile, mi_col, mi_row, cm->mi_rows, &mi_pos)) return 0;
|
||||
if (!is_inside(&xd->tile, mi_col, mi_row, &mi_pos)) return 0;
|
||||
|
||||
const TPL_MV_REF *prev_frame_mvs =
|
||||
cm->tpl_mvs + ((mi_row + mi_pos.row) >> 1) * (cm->mi_stride >> 1) +
|
||||
|
|
@ -430,20 +431,75 @@ static int add_tpl_ref_mv(const AV1_COMMON *cm, const MACROBLOCKD *xd,
|
|||
return 0;
|
||||
}
|
||||
|
||||
static void process_compound_ref_mv_candidate(
|
||||
const MB_MODE_INFO *const candidate, const AV1_COMMON *const cm,
|
||||
const MV_REFERENCE_FRAME *const rf, int_mv ref_id[2][2],
|
||||
int ref_id_count[2], int_mv ref_diff[2][2], int ref_diff_count[2]) {
|
||||
for (int rf_idx = 0; rf_idx < 2; ++rf_idx) {
|
||||
MV_REFERENCE_FRAME can_rf = candidate->ref_frame[rf_idx];
|
||||
|
||||
for (int cmp_idx = 0; cmp_idx < 2; ++cmp_idx) {
|
||||
if (can_rf == rf[cmp_idx] && ref_id_count[cmp_idx] < 2) {
|
||||
ref_id[cmp_idx][ref_id_count[cmp_idx]] = candidate->mv[rf_idx];
|
||||
++ref_id_count[cmp_idx];
|
||||
} else if (can_rf > INTRA_FRAME && ref_diff_count[cmp_idx] < 2) {
|
||||
int_mv this_mv = candidate->mv[rf_idx];
|
||||
if (cm->ref_frame_sign_bias[can_rf] !=
|
||||
cm->ref_frame_sign_bias[rf[cmp_idx]]) {
|
||||
this_mv.as_mv.row = -this_mv.as_mv.row;
|
||||
this_mv.as_mv.col = -this_mv.as_mv.col;
|
||||
}
|
||||
ref_diff[cmp_idx][ref_diff_count[cmp_idx]] = this_mv;
|
||||
++ref_diff_count[cmp_idx];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static void process_single_ref_mv_candidate(
|
||||
const MB_MODE_INFO *const candidate, const AV1_COMMON *const cm,
|
||||
MV_REFERENCE_FRAME ref_frame, uint8_t refmv_count[MODE_CTX_REF_FRAMES],
|
||||
CANDIDATE_MV ref_mv_stack[][MAX_REF_MV_STACK_SIZE]) {
|
||||
for (int rf_idx = 0; rf_idx < 2; ++rf_idx) {
|
||||
if (candidate->ref_frame[rf_idx] > INTRA_FRAME) {
|
||||
int_mv this_mv = candidate->mv[rf_idx];
|
||||
if (cm->ref_frame_sign_bias[candidate->ref_frame[rf_idx]] !=
|
||||
cm->ref_frame_sign_bias[ref_frame]) {
|
||||
this_mv.as_mv.row = -this_mv.as_mv.row;
|
||||
this_mv.as_mv.col = -this_mv.as_mv.col;
|
||||
}
|
||||
int stack_idx;
|
||||
for (stack_idx = 0; stack_idx < refmv_count[ref_frame]; ++stack_idx) {
|
||||
const int_mv stack_mv = ref_mv_stack[ref_frame][stack_idx].this_mv;
|
||||
if (this_mv.as_int == stack_mv.as_int) break;
|
||||
}
|
||||
|
||||
if (stack_idx == refmv_count[ref_frame]) {
|
||||
ref_mv_stack[ref_frame][stack_idx].this_mv = this_mv;
|
||||
|
||||
// TODO(jingning): Set an arbitrary small number here. The weight
|
||||
// doesn't matter as long as it is properly initialized.
|
||||
ref_mv_stack[ref_frame][stack_idx].weight = 2;
|
||||
++refmv_count[ref_frame];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static void setup_ref_mv_list(
|
||||
const AV1_COMMON *cm, const MACROBLOCKD *xd, MV_REFERENCE_FRAME ref_frame,
|
||||
uint8_t refmv_count[MODE_CTX_REF_FRAMES],
|
||||
CANDIDATE_MV ref_mv_stack[][MAX_REF_MV_STACK_SIZE],
|
||||
int_mv mv_ref_list[][MAX_MV_REF_CANDIDATES], int_mv *gm_mv_candidates,
|
||||
int mi_row, int mi_col, int16_t *mode_context) {
|
||||
const int bs = AOMMAX(xd->n8_w, xd->n8_h);
|
||||
const int bs = AOMMAX(xd->n4_w, xd->n4_h);
|
||||
const int has_tr = has_top_right(cm, xd, mi_row, mi_col, bs);
|
||||
MV_REFERENCE_FRAME rf[2];
|
||||
|
||||
const TileInfo *const tile = &xd->tile;
|
||||
int max_row_offset = 0, max_col_offset = 0;
|
||||
const int row_adj = (xd->n8_h < mi_size_high[BLOCK_8X8]) && (mi_row & 0x01);
|
||||
const int col_adj = (xd->n8_w < mi_size_wide[BLOCK_8X8]) && (mi_col & 0x01);
|
||||
const int row_adj = (xd->n4_h < mi_size_high[BLOCK_8X8]) && (mi_row & 0x01);
|
||||
const int col_adj = (xd->n4_w < mi_size_wide[BLOCK_8X8]) && (mi_col & 0x01);
|
||||
int processed_rows = 0;
|
||||
int processed_cols = 0;
|
||||
|
||||
|
|
@ -455,17 +511,16 @@ static void setup_ref_mv_list(
|
|||
if (xd->up_available) {
|
||||
max_row_offset = -(MVREF_ROW_COLS << 1) + row_adj;
|
||||
|
||||
if (xd->n8_h < mi_size_high[BLOCK_8X8])
|
||||
if (xd->n4_h < mi_size_high[BLOCK_8X8])
|
||||
max_row_offset = -(2 << 1) + row_adj;
|
||||
|
||||
max_row_offset =
|
||||
find_valid_row_offset(tile, mi_row, cm->mi_rows, max_row_offset);
|
||||
max_row_offset = find_valid_row_offset(tile, mi_row, max_row_offset);
|
||||
}
|
||||
|
||||
if (xd->left_available) {
|
||||
max_col_offset = -(MVREF_ROW_COLS << 1) + col_adj;
|
||||
|
||||
if (xd->n8_w < mi_size_wide[BLOCK_8X8])
|
||||
if (xd->n4_w < mi_size_wide[BLOCK_8X8])
|
||||
max_col_offset = -(2 << 1) + col_adj;
|
||||
|
||||
max_col_offset = find_valid_col_offset(tile, mi_col, max_col_offset);
|
||||
|
|
@ -487,12 +542,12 @@ static void setup_ref_mv_list(
|
|||
gm_mv_candidates, max_col_offset, &processed_cols);
|
||||
// Check top-right boundary
|
||||
if (has_tr)
|
||||
scan_blk_mbmi(cm, xd, mi_row, mi_col, rf, -1, xd->n8_w,
|
||||
scan_blk_mbmi(cm, xd, mi_row, mi_col, rf, -1, xd->n4_w,
|
||||
ref_mv_stack[ref_frame], &row_match_count, &newmv_count,
|
||||
gm_mv_candidates, &refmv_count[ref_frame]);
|
||||
|
||||
uint8_t nearest_match = (row_match_count > 0) + (col_match_count > 0);
|
||||
uint8_t nearest_refmv_count = refmv_count[ref_frame];
|
||||
const uint8_t nearest_match = (row_match_count > 0) + (col_match_count > 0);
|
||||
const uint8_t nearest_refmv_count = refmv_count[ref_frame];
|
||||
|
||||
// TODO(yunqing): for comp_search, do it for all 3 cases.
|
||||
for (int idx = 0; idx < nearest_refmv_count; ++idx)
|
||||
|
|
@ -500,27 +555,27 @@ static void setup_ref_mv_list(
|
|||
|
||||
if (cm->allow_ref_frame_mvs) {
|
||||
int is_available = 0;
|
||||
const int voffset = AOMMAX(mi_size_high[BLOCK_8X8], xd->n8_h);
|
||||
const int hoffset = AOMMAX(mi_size_wide[BLOCK_8X8], xd->n8_w);
|
||||
const int blk_row_end = AOMMIN(xd->n8_h, mi_size_high[BLOCK_64X64]);
|
||||
const int blk_col_end = AOMMIN(xd->n8_w, mi_size_wide[BLOCK_64X64]);
|
||||
const int voffset = AOMMAX(mi_size_high[BLOCK_8X8], xd->n4_h);
|
||||
const int hoffset = AOMMAX(mi_size_wide[BLOCK_8X8], xd->n4_w);
|
||||
const int blk_row_end = AOMMIN(xd->n4_h, mi_size_high[BLOCK_64X64]);
|
||||
const int blk_col_end = AOMMIN(xd->n4_w, mi_size_wide[BLOCK_64X64]);
|
||||
|
||||
const int tpl_sample_pos[3][2] = {
|
||||
{ voffset, -2 },
|
||||
{ voffset, hoffset },
|
||||
{ voffset - 2, hoffset },
|
||||
};
|
||||
const int allow_extension = (xd->n8_h >= mi_size_high[BLOCK_8X8]) &&
|
||||
(xd->n8_h < mi_size_high[BLOCK_64X64]) &&
|
||||
(xd->n8_w >= mi_size_wide[BLOCK_8X8]) &&
|
||||
(xd->n8_w < mi_size_wide[BLOCK_64X64]);
|
||||
const int allow_extension = (xd->n4_h >= mi_size_high[BLOCK_8X8]) &&
|
||||
(xd->n4_h < mi_size_high[BLOCK_64X64]) &&
|
||||
(xd->n4_w >= mi_size_wide[BLOCK_8X8]) &&
|
||||
(xd->n4_w < mi_size_wide[BLOCK_64X64]);
|
||||
|
||||
int step_h = (xd->n8_h >= mi_size_high[BLOCK_64X64])
|
||||
? mi_size_high[BLOCK_16X16]
|
||||
: mi_size_high[BLOCK_8X8];
|
||||
int step_w = (xd->n8_w >= mi_size_wide[BLOCK_64X64])
|
||||
? mi_size_wide[BLOCK_16X16]
|
||||
: mi_size_wide[BLOCK_8X8];
|
||||
const int step_h = (xd->n4_h >= mi_size_high[BLOCK_64X64])
|
||||
? mi_size_high[BLOCK_16X16]
|
||||
: mi_size_high[BLOCK_8X8];
|
||||
const int step_w = (xd->n4_w >= mi_size_wide[BLOCK_64X64])
|
||||
? mi_size_wide[BLOCK_16X16]
|
||||
: mi_size_wide[BLOCK_8X8];
|
||||
|
||||
for (int blk_row = 0; blk_row < blk_row_end; blk_row += step_h) {
|
||||
for (int blk_col = 0; blk_col < blk_col_end; blk_col += step_w) {
|
||||
|
|
@ -569,7 +624,7 @@ static void setup_ref_mv_list(
|
|||
max_col_offset, &processed_cols);
|
||||
}
|
||||
|
||||
uint8_t ref_match_count = (row_match_count > 0) + (col_match_count > 0);
|
||||
const uint8_t ref_match_count = (row_match_count > 0) + (col_match_count > 0);
|
||||
|
||||
switch (nearest_match) {
|
||||
case 0:
|
||||
|
|
@ -636,62 +691,24 @@ static void setup_ref_mv_list(
|
|||
int_mv ref_id[2][2], ref_diff[2][2];
|
||||
int ref_id_count[2] = { 0 }, ref_diff_count[2] = { 0 };
|
||||
|
||||
int mi_width = AOMMIN(mi_size_wide[BLOCK_64X64], xd->n8_w);
|
||||
int mi_width = AOMMIN(mi_size_wide[BLOCK_64X64], xd->n4_w);
|
||||
mi_width = AOMMIN(mi_width, cm->mi_cols - mi_col);
|
||||
int mi_height = AOMMIN(mi_size_high[BLOCK_64X64], xd->n8_h);
|
||||
int mi_height = AOMMIN(mi_size_high[BLOCK_64X64], xd->n4_h);
|
||||
mi_height = AOMMIN(mi_height, cm->mi_rows - mi_row);
|
||||
int mi_size = AOMMIN(mi_width, mi_height);
|
||||
|
||||
for (int idx = 0; abs(max_row_offset) >= 1 && idx < mi_size;) {
|
||||
const MB_MODE_INFO *const candidate = xd->mi[-xd->mi_stride + idx];
|
||||
const int candidate_bsize = candidate->sb_type;
|
||||
|
||||
for (int rf_idx = 0; rf_idx < 2; ++rf_idx) {
|
||||
MV_REFERENCE_FRAME can_rf = candidate->ref_frame[rf_idx];
|
||||
|
||||
for (int cmp_idx = 0; cmp_idx < 2; ++cmp_idx) {
|
||||
if (can_rf == rf[cmp_idx] && ref_id_count[cmp_idx] < 2) {
|
||||
ref_id[cmp_idx][ref_id_count[cmp_idx]] = candidate->mv[rf_idx];
|
||||
++ref_id_count[cmp_idx];
|
||||
} else if (can_rf > INTRA_FRAME && ref_diff_count[cmp_idx] < 2) {
|
||||
int_mv this_mv = candidate->mv[rf_idx];
|
||||
if (cm->ref_frame_sign_bias[can_rf] !=
|
||||
cm->ref_frame_sign_bias[rf[cmp_idx]]) {
|
||||
this_mv.as_mv.row = -this_mv.as_mv.row;
|
||||
this_mv.as_mv.col = -this_mv.as_mv.col;
|
||||
}
|
||||
ref_diff[cmp_idx][ref_diff_count[cmp_idx]] = this_mv;
|
||||
++ref_diff_count[cmp_idx];
|
||||
}
|
||||
}
|
||||
}
|
||||
idx += mi_size_wide[candidate_bsize];
|
||||
process_compound_ref_mv_candidate(
|
||||
candidate, cm, rf, ref_id, ref_id_count, ref_diff, ref_diff_count);
|
||||
idx += mi_size_wide[candidate->sb_type];
|
||||
}
|
||||
|
||||
for (int idx = 0; abs(max_col_offset) >= 1 && idx < mi_size;) {
|
||||
const MB_MODE_INFO *const candidate = xd->mi[idx * xd->mi_stride - 1];
|
||||
const int candidate_bsize = candidate->sb_type;
|
||||
|
||||
for (int rf_idx = 0; rf_idx < 2; ++rf_idx) {
|
||||
MV_REFERENCE_FRAME can_rf = candidate->ref_frame[rf_idx];
|
||||
|
||||
for (int cmp_idx = 0; cmp_idx < 2; ++cmp_idx) {
|
||||
if (can_rf == rf[cmp_idx] && ref_id_count[cmp_idx] < 2) {
|
||||
ref_id[cmp_idx][ref_id_count[cmp_idx]] = candidate->mv[rf_idx];
|
||||
++ref_id_count[cmp_idx];
|
||||
} else if (can_rf > INTRA_FRAME && ref_diff_count[cmp_idx] < 2) {
|
||||
int_mv this_mv = candidate->mv[rf_idx];
|
||||
if (cm->ref_frame_sign_bias[can_rf] !=
|
||||
cm->ref_frame_sign_bias[rf[cmp_idx]]) {
|
||||
this_mv.as_mv.row = -this_mv.as_mv.row;
|
||||
this_mv.as_mv.col = -this_mv.as_mv.col;
|
||||
}
|
||||
ref_diff[cmp_idx][ref_diff_count[cmp_idx]] = this_mv;
|
||||
++ref_diff_count[cmp_idx];
|
||||
}
|
||||
}
|
||||
}
|
||||
idx += mi_size_high[candidate_bsize];
|
||||
process_compound_ref_mv_candidate(
|
||||
candidate, cm, rf, ref_id, ref_id_count, ref_diff, ref_diff_count);
|
||||
idx += mi_size_high[candidate->sb_type];
|
||||
}
|
||||
|
||||
// Build up the compound mv predictor
|
||||
|
|
@ -743,87 +760,37 @@ static void setup_ref_mv_list(
|
|||
|
||||
for (int idx = 0; idx < refmv_count[ref_frame]; ++idx) {
|
||||
clamp_mv_ref(&ref_mv_stack[ref_frame][idx].this_mv.as_mv,
|
||||
xd->n8_w << MI_SIZE_LOG2, xd->n8_h << MI_SIZE_LOG2, xd);
|
||||
xd->n4_w << MI_SIZE_LOG2, xd->n4_h << MI_SIZE_LOG2, xd);
|
||||
clamp_mv_ref(&ref_mv_stack[ref_frame][idx].comp_mv.as_mv,
|
||||
xd->n8_w << MI_SIZE_LOG2, xd->n8_h << MI_SIZE_LOG2, xd);
|
||||
xd->n4_w << MI_SIZE_LOG2, xd->n4_h << MI_SIZE_LOG2, xd);
|
||||
}
|
||||
} else {
|
||||
// Handle single reference frame extension
|
||||
int mi_width = AOMMIN(mi_size_wide[BLOCK_64X64], xd->n8_w);
|
||||
int mi_width = AOMMIN(mi_size_wide[BLOCK_64X64], xd->n4_w);
|
||||
mi_width = AOMMIN(mi_width, cm->mi_cols - mi_col);
|
||||
int mi_height = AOMMIN(mi_size_high[BLOCK_64X64], xd->n8_h);
|
||||
int mi_height = AOMMIN(mi_size_high[BLOCK_64X64], xd->n4_h);
|
||||
mi_height = AOMMIN(mi_height, cm->mi_rows - mi_row);
|
||||
int mi_size = AOMMIN(mi_width, mi_height);
|
||||
|
||||
for (int idx = 0; abs(max_row_offset) >= 1 && idx < mi_size &&
|
||||
refmv_count[ref_frame] < MAX_MV_REF_CANDIDATES;) {
|
||||
const MB_MODE_INFO *const candidate = xd->mi[-xd->mi_stride + idx];
|
||||
const int candidate_bsize = candidate->sb_type;
|
||||
|
||||
// TODO(jingning): Refactor the following code.
|
||||
for (int rf_idx = 0; rf_idx < 2; ++rf_idx) {
|
||||
if (candidate->ref_frame[rf_idx] > INTRA_FRAME) {
|
||||
int_mv this_mv = candidate->mv[rf_idx];
|
||||
if (cm->ref_frame_sign_bias[candidate->ref_frame[rf_idx]] !=
|
||||
cm->ref_frame_sign_bias[ref_frame]) {
|
||||
this_mv.as_mv.row = -this_mv.as_mv.row;
|
||||
this_mv.as_mv.col = -this_mv.as_mv.col;
|
||||
}
|
||||
int stack_idx;
|
||||
for (stack_idx = 0; stack_idx < refmv_count[ref_frame]; ++stack_idx) {
|
||||
int_mv stack_mv = ref_mv_stack[ref_frame][stack_idx].this_mv;
|
||||
if (this_mv.as_int == stack_mv.as_int) break;
|
||||
}
|
||||
|
||||
if (stack_idx == refmv_count[ref_frame]) {
|
||||
ref_mv_stack[ref_frame][stack_idx].this_mv = this_mv;
|
||||
|
||||
// TODO(jingning): Set an arbitrary small number here. The weight
|
||||
// doesn't matter as long as it is properly initialized.
|
||||
ref_mv_stack[ref_frame][stack_idx].weight = 2;
|
||||
++refmv_count[ref_frame];
|
||||
}
|
||||
}
|
||||
}
|
||||
idx += mi_size_wide[candidate_bsize];
|
||||
process_single_ref_mv_candidate(candidate, cm, ref_frame, refmv_count,
|
||||
ref_mv_stack);
|
||||
idx += mi_size_wide[candidate->sb_type];
|
||||
}
|
||||
|
||||
for (int idx = 0; abs(max_col_offset) >= 1 && idx < mi_size &&
|
||||
refmv_count[ref_frame] < MAX_MV_REF_CANDIDATES;) {
|
||||
const MB_MODE_INFO *const candidate = xd->mi[idx * xd->mi_stride - 1];
|
||||
const int candidate_bsize = candidate->sb_type;
|
||||
|
||||
// TODO(jingning): Refactor the following code.
|
||||
for (int rf_idx = 0; rf_idx < 2; ++rf_idx) {
|
||||
if (candidate->ref_frame[rf_idx] > INTRA_FRAME) {
|
||||
int_mv this_mv = candidate->mv[rf_idx];
|
||||
if (cm->ref_frame_sign_bias[candidate->ref_frame[rf_idx]] !=
|
||||
cm->ref_frame_sign_bias[ref_frame]) {
|
||||
this_mv.as_mv.row = -this_mv.as_mv.row;
|
||||
this_mv.as_mv.col = -this_mv.as_mv.col;
|
||||
}
|
||||
int stack_idx;
|
||||
for (stack_idx = 0; stack_idx < refmv_count[ref_frame]; ++stack_idx) {
|
||||
int_mv stack_mv = ref_mv_stack[ref_frame][stack_idx].this_mv;
|
||||
if (this_mv.as_int == stack_mv.as_int) break;
|
||||
}
|
||||
|
||||
if (stack_idx == refmv_count[ref_frame]) {
|
||||
ref_mv_stack[ref_frame][stack_idx].this_mv = this_mv;
|
||||
|
||||
// TODO(jingning): Set an arbitrary small number here. The weight
|
||||
// doesn't matter as long as it is properly initialized.
|
||||
ref_mv_stack[ref_frame][stack_idx].weight = 2;
|
||||
++refmv_count[ref_frame];
|
||||
}
|
||||
}
|
||||
}
|
||||
idx += mi_size_high[candidate_bsize];
|
||||
process_single_ref_mv_candidate(candidate, cm, ref_frame, refmv_count,
|
||||
ref_mv_stack);
|
||||
idx += mi_size_high[candidate->sb_type];
|
||||
}
|
||||
|
||||
for (int idx = 0; idx < refmv_count[ref_frame]; ++idx) {
|
||||
clamp_mv_ref(&ref_mv_stack[ref_frame][idx].this_mv.as_mv,
|
||||
xd->n8_w << MI_SIZE_LOG2, xd->n8_h << MI_SIZE_LOG2, xd);
|
||||
xd->n4_w << MI_SIZE_LOG2, xd->n4_h << MI_SIZE_LOG2, xd);
|
||||
}
|
||||
|
||||
if (mv_ref_list != NULL) {
|
||||
|
|
@ -936,8 +903,10 @@ static int get_block_position(AV1_COMMON *cm, int *mi_r, int *mi_c, int blk_row,
|
|||
const int col_offset = (mv.col >= 0) ? (mv.col >> (4 + MI_SIZE_LOG2))
|
||||
: -((-mv.col) >> (4 + MI_SIZE_LOG2));
|
||||
|
||||
int row = (sign_bias == 1) ? blk_row - row_offset : blk_row + row_offset;
|
||||
int col = (sign_bias == 1) ? blk_col - col_offset : blk_col + col_offset;
|
||||
const int row =
|
||||
(sign_bias == 1) ? blk_row - row_offset : blk_row + row_offset;
|
||||
const int col =
|
||||
(sign_bias == 1) ? blk_col - col_offset : blk_col + col_offset;
|
||||
|
||||
if (row < 0 || row >= (cm->mi_rows >> 1) || col < 0 ||
|
||||
col >= (cm->mi_cols >> 1))
|
||||
|
|
@ -955,37 +924,44 @@ static int get_block_position(AV1_COMMON *cm, int *mi_r, int *mi_c, int blk_row,
|
|||
return 1;
|
||||
}
|
||||
|
||||
static int motion_field_projection(AV1_COMMON *cm, MV_REFERENCE_FRAME ref_frame,
|
||||
int dir) {
|
||||
// Note: motion_filed_projection finds motion vectors of current frame's
|
||||
// reference frame, and projects them to current frame. To make it clear,
|
||||
// let's call current frame's reference frame as start frame.
|
||||
// Call Start frame's reference frames as reference frames.
|
||||
// Call ref_offset as frame distances between start frame and its reference
|
||||
// frames.
|
||||
static int motion_field_projection(AV1_COMMON *cm,
|
||||
MV_REFERENCE_FRAME start_frame, int dir) {
|
||||
TPL_MV_REF *tpl_mvs_base = cm->tpl_mvs;
|
||||
int ref_offset[REF_FRAMES] = { 0 };
|
||||
|
||||
(void)dir;
|
||||
|
||||
int ref_frame_idx = cm->frame_refs[FWD_RF_OFFSET(ref_frame)].idx;
|
||||
if (ref_frame_idx < 0) return 0;
|
||||
const int start_frame_idx = cm->frame_refs[FWD_RF_OFFSET(start_frame)].idx;
|
||||
if (start_frame_idx < 0) return 0;
|
||||
|
||||
if (cm->buffer_pool->frame_bufs[ref_frame_idx].intra_only) return 0;
|
||||
if (cm->buffer_pool->frame_bufs[start_frame_idx].intra_only) return 0;
|
||||
|
||||
if (cm->buffer_pool->frame_bufs[ref_frame_idx].mi_rows != cm->mi_rows ||
|
||||
cm->buffer_pool->frame_bufs[ref_frame_idx].mi_cols != cm->mi_cols)
|
||||
if (cm->buffer_pool->frame_bufs[start_frame_idx].mi_rows != cm->mi_rows ||
|
||||
cm->buffer_pool->frame_bufs[start_frame_idx].mi_cols != cm->mi_cols)
|
||||
return 0;
|
||||
|
||||
int ref_frame_index =
|
||||
cm->buffer_pool->frame_bufs[ref_frame_idx].cur_frame_offset;
|
||||
unsigned int *ref_rf_idx =
|
||||
&cm->buffer_pool->frame_bufs[ref_frame_idx].ref_frame_offset[0];
|
||||
int cur_frame_index = cm->cur_frame->cur_frame_offset;
|
||||
int ref_to_cur = get_relative_dist(cm, ref_frame_index, cur_frame_index);
|
||||
const int start_frame_offset =
|
||||
cm->buffer_pool->frame_bufs[start_frame_idx].cur_frame_offset;
|
||||
const unsigned int *const ref_frame_offsets =
|
||||
&cm->buffer_pool->frame_bufs[start_frame_idx].ref_frame_offset[0];
|
||||
const int cur_frame_offset = cm->cur_frame->cur_frame_offset;
|
||||
int start_to_current_frame_offset =
|
||||
get_relative_dist(cm, start_frame_offset, cur_frame_offset);
|
||||
|
||||
for (MV_REFERENCE_FRAME rf = LAST_FRAME; rf <= INTER_REFS_PER_FRAME; ++rf) {
|
||||
ref_offset[rf] =
|
||||
get_relative_dist(cm, ref_frame_index, ref_rf_idx[rf - LAST_FRAME]);
|
||||
ref_offset[rf] = get_relative_dist(cm, start_frame_offset,
|
||||
ref_frame_offsets[rf - LAST_FRAME]);
|
||||
}
|
||||
|
||||
if (dir == 2) ref_to_cur = -ref_to_cur;
|
||||
if (dir == 2) start_to_current_frame_offset = -start_to_current_frame_offset;
|
||||
|
||||
MV_REF *mv_ref_base = cm->buffer_pool->frame_bufs[ref_frame_idx].mvs;
|
||||
MV_REF *mv_ref_base = cm->buffer_pool->frame_bufs[start_frame_idx].mvs;
|
||||
const int mvs_rows = (cm->mi_rows + 1) >> 1;
|
||||
const int mvs_cols = (cm->mi_cols + 1) >> 1;
|
||||
|
||||
|
|
@ -999,19 +975,20 @@ static int motion_field_projection(AV1_COMMON *cm, MV_REFERENCE_FRAME ref_frame,
|
|||
int mi_r, mi_c;
|
||||
const int ref_frame_offset = ref_offset[mv_ref->ref_frame];
|
||||
|
||||
int pos_valid = abs(ref_frame_offset) <= MAX_FRAME_DISTANCE &&
|
||||
ref_frame_offset > 0 &&
|
||||
abs(ref_to_cur) <= MAX_FRAME_DISTANCE;
|
||||
int pos_valid =
|
||||
abs(ref_frame_offset) <= MAX_FRAME_DISTANCE &&
|
||||
ref_frame_offset > 0 &&
|
||||
abs(start_to_current_frame_offset) <= MAX_FRAME_DISTANCE;
|
||||
|
||||
if (pos_valid) {
|
||||
get_mv_projection(&this_mv.as_mv, fwd_mv, ref_to_cur,
|
||||
ref_frame_offset);
|
||||
get_mv_projection(&this_mv.as_mv, fwd_mv,
|
||||
start_to_current_frame_offset, ref_frame_offset);
|
||||
pos_valid = get_block_position(cm, &mi_r, &mi_c, blk_row, blk_col,
|
||||
this_mv.as_mv, dir >> 1);
|
||||
}
|
||||
|
||||
if (pos_valid) {
|
||||
int mi_offset = mi_r * (cm->mi_stride >> 1) + mi_c;
|
||||
const int mi_offset = mi_r * (cm->mi_stride >> 1) + mi_c;
|
||||
|
||||
tpl_mvs_base[mi_offset].mfmv0.as_mv.row = fwd_mv.row;
|
||||
tpl_mvs_base[mi_offset].mfmv0.as_mv.col = fwd_mv.col;
|
||||
|
|
@ -1167,14 +1144,14 @@ int findSamples(const AV1_COMMON *cm, MACROBLOCKD *xd, int mi_row, int mi_col,
|
|||
if (up_available) {
|
||||
int mi_row_offset = -1;
|
||||
MB_MODE_INFO *mbmi = xd->mi[mi_row_offset * xd->mi_stride];
|
||||
uint8_t n8_w = mi_size_wide[mbmi->sb_type];
|
||||
uint8_t n4_w = mi_size_wide[mbmi->sb_type];
|
||||
|
||||
if (xd->n8_w <= n8_w) {
|
||||
if (xd->n4_w <= n4_w) {
|
||||
// Handle "current block width <= above block width" case.
|
||||
int col_offset = -mi_col % n8_w;
|
||||
int col_offset = -mi_col % n4_w;
|
||||
|
||||
if (col_offset < 0) do_tl = 0;
|
||||
if (col_offset + n8_w > xd->n8_w) do_tr = 0;
|
||||
if (col_offset + n4_w > xd->n4_w) do_tr = 0;
|
||||
|
||||
if (mbmi->ref_frame[0] == ref_frame && mbmi->ref_frame[1] == NONE_FRAME) {
|
||||
record_samples(mbmi, pts, pts_inref, 0, -1, col_offset, 1);
|
||||
|
|
@ -1185,11 +1162,11 @@ int findSamples(const AV1_COMMON *cm, MACROBLOCKD *xd, int mi_row, int mi_col,
|
|||
}
|
||||
} else {
|
||||
// Handle "current block width > above block width" case.
|
||||
for (i = 0; i < AOMMIN(xd->n8_w, cm->mi_cols - mi_col); i += mi_step) {
|
||||
for (i = 0; i < AOMMIN(xd->n4_w, cm->mi_cols - mi_col); i += mi_step) {
|
||||
int mi_col_offset = i;
|
||||
mbmi = xd->mi[mi_col_offset + mi_row_offset * xd->mi_stride];
|
||||
n8_w = mi_size_wide[mbmi->sb_type];
|
||||
mi_step = AOMMIN(xd->n8_w, n8_w);
|
||||
n4_w = mi_size_wide[mbmi->sb_type];
|
||||
mi_step = AOMMIN(xd->n4_w, n4_w);
|
||||
|
||||
if (mbmi->ref_frame[0] == ref_frame &&
|
||||
mbmi->ref_frame[1] == NONE_FRAME) {
|
||||
|
|
@ -1209,11 +1186,11 @@ int findSamples(const AV1_COMMON *cm, MACROBLOCKD *xd, int mi_row, int mi_col,
|
|||
int mi_col_offset = -1;
|
||||
|
||||
MB_MODE_INFO *mbmi = xd->mi[mi_col_offset];
|
||||
uint8_t n8_h = mi_size_high[mbmi->sb_type];
|
||||
uint8_t n4_h = mi_size_high[mbmi->sb_type];
|
||||
|
||||
if (xd->n8_h <= n8_h) {
|
||||
if (xd->n4_h <= n4_h) {
|
||||
// Handle "current block height <= above block height" case.
|
||||
int row_offset = -mi_row % n8_h;
|
||||
int row_offset = -mi_row % n4_h;
|
||||
|
||||
if (row_offset < 0) do_tl = 0;
|
||||
|
||||
|
|
@ -1226,11 +1203,11 @@ int findSamples(const AV1_COMMON *cm, MACROBLOCKD *xd, int mi_row, int mi_col,
|
|||
}
|
||||
} else {
|
||||
// Handle "current block height > above block height" case.
|
||||
for (i = 0; i < AOMMIN(xd->n8_h, cm->mi_rows - mi_row); i += mi_step) {
|
||||
for (i = 0; i < AOMMIN(xd->n4_h, cm->mi_rows - mi_row); i += mi_step) {
|
||||
int mi_row_offset = i;
|
||||
mbmi = xd->mi[mi_col_offset + mi_row_offset * xd->mi_stride];
|
||||
n8_h = mi_size_high[mbmi->sb_type];
|
||||
mi_step = AOMMIN(xd->n8_h, n8_h);
|
||||
n4_h = mi_size_high[mbmi->sb_type];
|
||||
mi_step = AOMMIN(xd->n4_h, n4_h);
|
||||
|
||||
if (mbmi->ref_frame[0] == ref_frame &&
|
||||
mbmi->ref_frame[1] == NONE_FRAME) {
|
||||
|
|
@ -1264,18 +1241,18 @@ int findSamples(const AV1_COMMON *cm, MACROBLOCKD *xd, int mi_row, int mi_col,
|
|||
|
||||
// Top-right block
|
||||
if (do_tr &&
|
||||
has_top_right(cm, xd, mi_row, mi_col, AOMMAX(xd->n8_w, xd->n8_h))) {
|
||||
POSITION trb_pos = { -1, xd->n8_w };
|
||||
has_top_right(cm, xd, mi_row, mi_col, AOMMAX(xd->n4_w, xd->n4_h))) {
|
||||
POSITION trb_pos = { -1, xd->n4_w };
|
||||
|
||||
if (is_inside(tile, mi_col, mi_row, cm->mi_rows, &trb_pos)) {
|
||||
if (is_inside(tile, mi_col, mi_row, &trb_pos)) {
|
||||
int mi_row_offset = -1;
|
||||
int mi_col_offset = xd->n8_w;
|
||||
int mi_col_offset = xd->n4_w;
|
||||
|
||||
MB_MODE_INFO *mbmi =
|
||||
xd->mi[mi_col_offset + mi_row_offset * xd->mi_stride];
|
||||
|
||||
if (mbmi->ref_frame[0] == ref_frame && mbmi->ref_frame[1] == NONE_FRAME) {
|
||||
record_samples(mbmi, pts, pts_inref, 0, -1, xd->n8_w, 1);
|
||||
record_samples(mbmi, pts, pts_inref, 0, -1, xd->n4_w, 1);
|
||||
np++;
|
||||
if (np >= LEAST_SQUARES_SAMPLES_MAX) return LEAST_SQUARES_SAMPLES_MAX;
|
||||
}
|
||||
|
|
@ -1372,7 +1349,7 @@ static int compare_ref_frame_info(const void *arg_a, const void *arg_b) {
|
|||
|
||||
static void set_ref_frame_info(AV1_COMMON *const cm, int frame_idx,
|
||||
REF_FRAME_INFO *ref_info) {
|
||||
assert(frame_idx >= 0 && frame_idx <= INTER_REFS_PER_FRAME);
|
||||
assert(frame_idx >= 0 && frame_idx < INTER_REFS_PER_FRAME);
|
||||
|
||||
const int buf_idx = ref_info->buf_idx;
|
||||
|
||||
|
|
|
|||
43
third_party/aom/av1/common/mvref_common.h
vendored
43
third_party/aom/av1/common/mvref_common.h
vendored
|
|
@ -8,8 +8,8 @@
|
|||
* Media Patent License 1.0 was not distributed with this source code in the
|
||||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
#ifndef AV1_COMMON_MVREF_COMMON_H_
|
||||
#define AV1_COMMON_MVREF_COMMON_H_
|
||||
#ifndef AOM_AV1_COMMON_MVREF_COMMON_H_
|
||||
#define AOM_AV1_COMMON_MVREF_COMMON_H_
|
||||
|
||||
#include "av1/common/onyxc_int.h"
|
||||
#include "av1/common/blockd.h"
|
||||
|
|
@ -85,29 +85,17 @@ static INLINE int_mv scale_mv(const MB_MODE_INFO *mbmi, int ref,
|
|||
// Checks that the given mi_row, mi_col and search point
|
||||
// are inside the borders of the tile.
|
||||
static INLINE int is_inside(const TileInfo *const tile, int mi_col, int mi_row,
|
||||
int mi_rows, const POSITION *mi_pos) {
|
||||
const int dependent_horz_tile_flag = 0;
|
||||
if (dependent_horz_tile_flag && !tile->tg_horz_boundary) {
|
||||
return !(mi_row + mi_pos->row < 0 ||
|
||||
mi_col + mi_pos->col < tile->mi_col_start ||
|
||||
mi_row + mi_pos->row >= mi_rows ||
|
||||
mi_col + mi_pos->col >= tile->mi_col_end);
|
||||
} else {
|
||||
return !(mi_row + mi_pos->row < tile->mi_row_start ||
|
||||
mi_col + mi_pos->col < tile->mi_col_start ||
|
||||
mi_row + mi_pos->row >= tile->mi_row_end ||
|
||||
mi_col + mi_pos->col >= tile->mi_col_end);
|
||||
}
|
||||
const POSITION *mi_pos) {
|
||||
return !(mi_row + mi_pos->row < tile->mi_row_start ||
|
||||
mi_col + mi_pos->col < tile->mi_col_start ||
|
||||
mi_row + mi_pos->row >= tile->mi_row_end ||
|
||||
mi_col + mi_pos->col >= tile->mi_col_end);
|
||||
}
|
||||
|
||||
static INLINE int find_valid_row_offset(const TileInfo *const tile, int mi_row,
|
||||
int mi_rows, int row_offset) {
|
||||
const int dependent_horz_tile_flag = 0;
|
||||
if (dependent_horz_tile_flag && !tile->tg_horz_boundary)
|
||||
return clamp(row_offset, -mi_row, mi_rows - mi_row - 1);
|
||||
else
|
||||
return clamp(row_offset, tile->mi_row_start - mi_row,
|
||||
tile->mi_row_end - mi_row - 1);
|
||||
int row_offset) {
|
||||
return clamp(row_offset, tile->mi_row_start - mi_row,
|
||||
tile->mi_row_end - mi_row - 1);
|
||||
}
|
||||
|
||||
static INLINE int find_valid_col_offset(const TileInfo *const tile, int mi_col,
|
||||
|
|
@ -263,8 +251,9 @@ static INLINE void av1_collect_neighbors_ref_counts(MACROBLOCKD *const xd) {
|
|||
}
|
||||
}
|
||||
|
||||
void av1_copy_frame_mvs(const AV1_COMMON *const cm, MB_MODE_INFO *mi,
|
||||
int mi_row, int mi_col, int x_mis, int y_mis);
|
||||
void av1_copy_frame_mvs(const AV1_COMMON *const cm,
|
||||
const MB_MODE_INFO *const mi, int mi_row, int mi_col,
|
||||
int x_mis, int y_mis);
|
||||
|
||||
void av1_find_mv_refs(const AV1_COMMON *cm, const MACROBLOCKD *xd,
|
||||
MB_MODE_INFO *mi, MV_REFERENCE_FRAME ref_frame,
|
||||
|
|
@ -286,7 +275,6 @@ int findSamples(const AV1_COMMON *cm, MACROBLOCKD *xd, int mi_row, int mi_col,
|
|||
|
||||
#define INTRABC_DELAY_PIXELS 256 // Delay of 256 pixels
|
||||
#define INTRABC_DELAY_SB64 (INTRABC_DELAY_PIXELS / 64)
|
||||
#define USE_WAVE_FRONT 1 // Use only top left area of frame for reference.
|
||||
|
||||
static INLINE void av1_find_ref_dv(int_mv *ref_dv, const TileInfo *const tile,
|
||||
int mib_size, int mi_row, int mi_col) {
|
||||
|
|
@ -356,13 +344,12 @@ static INLINE int av1_is_dv_valid(const MV dv, const AV1_COMMON *cm,
|
|||
const int src_sb64 = src_sb_row * total_sb64_per_row + src_sb64_col;
|
||||
if (src_sb64 >= active_sb64 - INTRABC_DELAY_SB64) return 0;
|
||||
|
||||
#if USE_WAVE_FRONT
|
||||
// Wavefront constraint: use only top left area of frame for reference.
|
||||
const int gradient = 1 + INTRABC_DELAY_SB64 + (sb_size > 64);
|
||||
const int wf_offset = gradient * (active_sb_row - src_sb_row);
|
||||
if (src_sb_row > active_sb_row ||
|
||||
src_sb64_col >= active_sb64_col - INTRABC_DELAY_SB64 + wf_offset)
|
||||
return 0;
|
||||
#endif
|
||||
|
||||
return 1;
|
||||
}
|
||||
|
|
@ -371,4 +358,4 @@ static INLINE int av1_is_dv_valid(const MV dv, const AV1_COMMON *cm,
|
|||
} // extern "C"
|
||||
#endif
|
||||
|
||||
#endif // AV1_COMMON_MVREF_COMMON_H_
|
||||
#endif // AOM_AV1_COMMON_MVREF_COMMON_H_
|
||||
|
|
|
|||
14
third_party/aom/av1/common/obmc.h
vendored
14
third_party/aom/av1/common/obmc.h
vendored
|
|
@ -9,8 +9,8 @@
|
|||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#ifndef AV1_COMMON_OBMC_H_
|
||||
#define AV1_COMMON_OBMC_H_
|
||||
#ifndef AOM_AV1_COMMON_OBMC_H_
|
||||
#define AOM_AV1_COMMON_OBMC_H_
|
||||
|
||||
typedef void (*overlappable_nb_visitor_t)(MACROBLOCKD *xd, int rel_mi_pos,
|
||||
uint8_t nb_mi_size,
|
||||
|
|
@ -30,7 +30,7 @@ static INLINE void foreach_overlappable_nb_above(const AV1_COMMON *cm,
|
|||
// prev_row_mi points into the mi array, starting at the beginning of the
|
||||
// previous row.
|
||||
MB_MODE_INFO **prev_row_mi = xd->mi - mi_col - 1 * xd->mi_stride;
|
||||
const int end_col = AOMMIN(mi_col + xd->n8_w, cm->mi_cols);
|
||||
const int end_col = AOMMIN(mi_col + xd->n4_w, cm->mi_cols);
|
||||
uint8_t mi_step;
|
||||
for (int above_mi_col = mi_col; above_mi_col < end_col && nb_count < nb_max;
|
||||
above_mi_col += mi_step) {
|
||||
|
|
@ -49,7 +49,7 @@ static INLINE void foreach_overlappable_nb_above(const AV1_COMMON *cm,
|
|||
}
|
||||
if (is_neighbor_overlappable(*above_mi)) {
|
||||
++nb_count;
|
||||
fun(xd, above_mi_col - mi_col, AOMMIN(xd->n8_w, mi_step), *above_mi,
|
||||
fun(xd, above_mi_col - mi_col, AOMMIN(xd->n4_w, mi_step), *above_mi,
|
||||
fun_ctxt, num_planes);
|
||||
}
|
||||
}
|
||||
|
|
@ -68,7 +68,7 @@ static INLINE void foreach_overlappable_nb_left(const AV1_COMMON *cm,
|
|||
// prev_col_mi points into the mi array, starting at the top of the
|
||||
// previous column
|
||||
MB_MODE_INFO **prev_col_mi = xd->mi - 1 - mi_row * xd->mi_stride;
|
||||
const int end_row = AOMMIN(mi_row + xd->n8_h, cm->mi_rows);
|
||||
const int end_row = AOMMIN(mi_row + xd->n4_h, cm->mi_rows);
|
||||
uint8_t mi_step;
|
||||
for (int left_mi_row = mi_row; left_mi_row < end_row && nb_count < nb_max;
|
||||
left_mi_row += mi_step) {
|
||||
|
|
@ -82,10 +82,10 @@ static INLINE void foreach_overlappable_nb_left(const AV1_COMMON *cm,
|
|||
}
|
||||
if (is_neighbor_overlappable(*left_mi)) {
|
||||
++nb_count;
|
||||
fun(xd, left_mi_row - mi_row, AOMMIN(xd->n8_h, mi_step), *left_mi,
|
||||
fun(xd, left_mi_row - mi_row, AOMMIN(xd->n4_h, mi_step), *left_mi,
|
||||
fun_ctxt, num_planes);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#endif // AV1_COMMON_OBMC_H_
|
||||
#endif // AOM_AV1_COMMON_OBMC_H_
|
||||
|
|
|
|||
147
third_party/aom/av1/common/obu_util.c
vendored
Normal file
147
third_party/aom/av1/common/obu_util.c
vendored
Normal file
|
|
@ -0,0 +1,147 @@
|
|||
/*
|
||||
* Copyright (c) 2018, Alliance for Open Media. All rights reserved
|
||||
*
|
||||
* This source code is subject to the terms of the BSD 2 Clause License and
|
||||
* the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
|
||||
* was not distributed with this source code in the LICENSE file, you can
|
||||
* obtain it at www.aomedia.org/license/software. If the Alliance for Open
|
||||
* Media Patent License 1.0 was not distributed with this source code in the
|
||||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
#include "av1/common/obu_util.h"
|
||||
|
||||
#include "aom_dsp/bitreader_buffer.h"
|
||||
|
||||
// Returns 1 when OBU type is valid, and 0 otherwise.
|
||||
static int valid_obu_type(int obu_type) {
|
||||
int valid_type = 0;
|
||||
switch (obu_type) {
|
||||
case OBU_SEQUENCE_HEADER:
|
||||
case OBU_TEMPORAL_DELIMITER:
|
||||
case OBU_FRAME_HEADER:
|
||||
case OBU_TILE_GROUP:
|
||||
case OBU_METADATA:
|
||||
case OBU_FRAME:
|
||||
case OBU_REDUNDANT_FRAME_HEADER:
|
||||
case OBU_TILE_LIST:
|
||||
case OBU_PADDING: valid_type = 1; break;
|
||||
default: break;
|
||||
}
|
||||
return valid_type;
|
||||
}
|
||||
|
||||
static aom_codec_err_t read_obu_size(const uint8_t *data,
|
||||
size_t bytes_available,
|
||||
size_t *const obu_size,
|
||||
size_t *const length_field_size) {
|
||||
uint64_t u_obu_size = 0;
|
||||
if (aom_uleb_decode(data, bytes_available, &u_obu_size, length_field_size) !=
|
||||
0) {
|
||||
return AOM_CODEC_CORRUPT_FRAME;
|
||||
}
|
||||
|
||||
if (u_obu_size > UINT32_MAX) return AOM_CODEC_CORRUPT_FRAME;
|
||||
*obu_size = (size_t)u_obu_size;
|
||||
return AOM_CODEC_OK;
|
||||
}
|
||||
|
||||
// Parses OBU header and stores values in 'header'.
|
||||
static aom_codec_err_t read_obu_header(struct aom_read_bit_buffer *rb,
|
||||
int is_annexb, ObuHeader *header) {
|
||||
if (!rb || !header) return AOM_CODEC_INVALID_PARAM;
|
||||
|
||||
const ptrdiff_t bit_buffer_byte_length = rb->bit_buffer_end - rb->bit_buffer;
|
||||
if (bit_buffer_byte_length < 1) return AOM_CODEC_CORRUPT_FRAME;
|
||||
|
||||
header->size = 1;
|
||||
|
||||
if (aom_rb_read_bit(rb) != 0) {
|
||||
// Forbidden bit. Must not be set.
|
||||
return AOM_CODEC_CORRUPT_FRAME;
|
||||
}
|
||||
|
||||
header->type = (OBU_TYPE)aom_rb_read_literal(rb, 4);
|
||||
|
||||
if (!valid_obu_type(header->type)) return AOM_CODEC_CORRUPT_FRAME;
|
||||
|
||||
header->has_extension = aom_rb_read_bit(rb);
|
||||
header->has_size_field = aom_rb_read_bit(rb);
|
||||
|
||||
if (!header->has_size_field && !is_annexb) {
|
||||
// section 5 obu streams must have obu_size field set.
|
||||
return AOM_CODEC_UNSUP_BITSTREAM;
|
||||
}
|
||||
|
||||
if (aom_rb_read_bit(rb) != 0) {
|
||||
// obu_reserved_1bit must be set to 0.
|
||||
return AOM_CODEC_CORRUPT_FRAME;
|
||||
}
|
||||
|
||||
if (header->has_extension) {
|
||||
if (bit_buffer_byte_length == 1) return AOM_CODEC_CORRUPT_FRAME;
|
||||
|
||||
header->size += 1;
|
||||
header->temporal_layer_id = aom_rb_read_literal(rb, 3);
|
||||
header->spatial_layer_id = aom_rb_read_literal(rb, 2);
|
||||
if (aom_rb_read_literal(rb, 3) != 0) {
|
||||
// extension_header_reserved_3bits must be set to 0.
|
||||
return AOM_CODEC_CORRUPT_FRAME;
|
||||
}
|
||||
}
|
||||
|
||||
return AOM_CODEC_OK;
|
||||
}
|
||||
|
||||
aom_codec_err_t aom_read_obu_header(uint8_t *buffer, size_t buffer_length,
|
||||
size_t *consumed, ObuHeader *header,
|
||||
int is_annexb) {
|
||||
if (buffer_length < 1 || !consumed || !header) return AOM_CODEC_INVALID_PARAM;
|
||||
|
||||
// TODO(tomfinegan): Set the error handler here and throughout this file, and
|
||||
// confirm parsing work done via aom_read_bit_buffer is successful.
|
||||
struct aom_read_bit_buffer rb = { buffer, buffer + buffer_length, 0, NULL,
|
||||
NULL };
|
||||
aom_codec_err_t parse_result = read_obu_header(&rb, is_annexb, header);
|
||||
if (parse_result == AOM_CODEC_OK) *consumed = header->size;
|
||||
return parse_result;
|
||||
}
|
||||
|
||||
aom_codec_err_t aom_read_obu_header_and_size(const uint8_t *data,
|
||||
size_t bytes_available,
|
||||
int is_annexb,
|
||||
ObuHeader *obu_header,
|
||||
size_t *const payload_size,
|
||||
size_t *const bytes_read) {
|
||||
size_t length_field_size = 0, obu_size = 0;
|
||||
aom_codec_err_t status;
|
||||
|
||||
if (is_annexb) {
|
||||
// Size field comes before the OBU header, and includes the OBU header
|
||||
status =
|
||||
read_obu_size(data, bytes_available, &obu_size, &length_field_size);
|
||||
|
||||
if (status != AOM_CODEC_OK) return status;
|
||||
}
|
||||
|
||||
struct aom_read_bit_buffer rb = { data + length_field_size,
|
||||
data + bytes_available, 0, NULL, NULL };
|
||||
|
||||
status = read_obu_header(&rb, is_annexb, obu_header);
|
||||
if (status != AOM_CODEC_OK) return status;
|
||||
|
||||
if (is_annexb) {
|
||||
// Derive the payload size from the data we've already read
|
||||
if (obu_size < obu_header->size) return AOM_CODEC_CORRUPT_FRAME;
|
||||
|
||||
*payload_size = obu_size - obu_header->size;
|
||||
} else {
|
||||
// Size field comes after the OBU header, and is just the payload size
|
||||
status = read_obu_size(data + obu_header->size,
|
||||
bytes_available - obu_header->size, payload_size,
|
||||
&length_field_size);
|
||||
if (status != AOM_CODEC_OK) return status;
|
||||
}
|
||||
|
||||
*bytes_read = length_field_size + obu_header->size;
|
||||
return AOM_CODEC_OK;
|
||||
}
|
||||
47
third_party/aom/av1/common/obu_util.h
vendored
Normal file
47
third_party/aom/av1/common/obu_util.h
vendored
Normal file
|
|
@ -0,0 +1,47 @@
|
|||
/*
|
||||
* Copyright (c) 2018, Alliance for Open Media. All rights reserved
|
||||
*
|
||||
* This source code is subject to the terms of the BSD 2 Clause License and
|
||||
* the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
|
||||
* was not distributed with this source code in the LICENSE file, you can
|
||||
* obtain it at www.aomedia.org/license/software. If the Alliance for Open
|
||||
* Media Patent License 1.0 was not distributed with this source code in the
|
||||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
#ifndef AOM_AV1_COMMON_OBU_UTIL_H_
|
||||
#define AOM_AV1_COMMON_OBU_UTIL_H_
|
||||
|
||||
#include "aom/aom_codec.h"
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
|
||||
typedef struct {
|
||||
size_t size; // Size (1 or 2 bytes) of the OBU header (including the
|
||||
// optional OBU extension header) in the bitstream.
|
||||
OBU_TYPE type;
|
||||
int has_size_field;
|
||||
int has_extension;
|
||||
// The following fields come from the OBU extension header and therefore are
|
||||
// only used if has_extension is true.
|
||||
int temporal_layer_id;
|
||||
int spatial_layer_id;
|
||||
} ObuHeader;
|
||||
|
||||
aom_codec_err_t aom_read_obu_header(uint8_t *buffer, size_t buffer_length,
|
||||
size_t *consumed, ObuHeader *header,
|
||||
int is_annexb);
|
||||
|
||||
aom_codec_err_t aom_read_obu_header_and_size(const uint8_t *data,
|
||||
size_t bytes_available,
|
||||
int is_annexb,
|
||||
ObuHeader *obu_header,
|
||||
size_t *const payload_size,
|
||||
size_t *const bytes_read);
|
||||
|
||||
#ifdef __cplusplus
|
||||
} // extern "C"
|
||||
#endif
|
||||
|
||||
#endif // AOM_AV1_COMMON_OBU_UTIL_H_
|
||||
12
third_party/aom/av1/common/odintrin.h
vendored
12
third_party/aom/av1/common/odintrin.h
vendored
|
|
@ -11,8 +11,8 @@
|
|||
|
||||
/* clang-format off */
|
||||
|
||||
#ifndef AV1_COMMON_ODINTRIN_H_
|
||||
#define AV1_COMMON_ODINTRIN_H_
|
||||
#ifndef AOM_AV1_COMMON_ODINTRIN_H_
|
||||
#define AOM_AV1_COMMON_ODINTRIN_H_
|
||||
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
|
|
@ -46,9 +46,9 @@ extern uint32_t OD_DIVU_SMALL_CONSTS[OD_DIVU_DMAX][2];
|
|||
#define OD_MAXI AOMMAX
|
||||
#define OD_CLAMPI(min, val, max) (OD_MAXI(min, OD_MINI(val, max)))
|
||||
|
||||
#define OD_CLZ0 (1)
|
||||
#define OD_CLZ(x) (-get_msb(x))
|
||||
#define OD_ILOG_NZ(x) (OD_CLZ0 - OD_CLZ(x))
|
||||
/*Integer logarithm (base 2) of a nonzero unsigned 32-bit integer.
|
||||
OD_ILOG_NZ(x) = (int)floor(log2(x)) + 1.*/
|
||||
#define OD_ILOG_NZ(x) (1 + get_msb(x))
|
||||
|
||||
/*Enable special features for gcc and compatible compilers.*/
|
||||
#if defined(__GNUC__) && defined(__GNUC_MINOR__) && defined(__GNUC_PATCHLEVEL__)
|
||||
|
|
@ -93,4 +93,4 @@ extern uint32_t OD_DIVU_SMALL_CONSTS[OD_DIVU_DMAX][2];
|
|||
} // extern "C"
|
||||
#endif
|
||||
|
||||
#endif // AV1_COMMON_ODINTRIN_H_
|
||||
#endif // AOM_AV1_COMMON_ODINTRIN_H_
|
||||
|
|
|
|||
29
third_party/aom/av1/common/onyxc_int.h
vendored
29
third_party/aom/av1/common/onyxc_int.h
vendored
|
|
@ -9,8 +9,8 @@
|
|||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#ifndef AV1_COMMON_ONYXC_INT_H_
|
||||
#define AV1_COMMON_ONYXC_INT_H_
|
||||
#ifndef AOM_AV1_COMMON_ONYXC_INT_H_
|
||||
#define AOM_AV1_COMMON_ONYXC_INT_H_
|
||||
|
||||
#include "config/aom_config.h"
|
||||
#include "config/av1_rtcd.h"
|
||||
|
|
@ -480,6 +480,7 @@ typedef struct AV1Common {
|
|||
|
||||
int byte_alignment;
|
||||
int skip_loop_filter;
|
||||
int skip_film_grain;
|
||||
|
||||
// Private data associated with the frame buffer callbacks.
|
||||
void *cb_priv;
|
||||
|
|
@ -823,18 +824,18 @@ static INLINE void set_mi_row_col(MACROBLOCKD *xd, const TileInfo *const tile,
|
|||
xd->chroma_left_mbmi = chroma_left_mi;
|
||||
}
|
||||
|
||||
xd->n8_h = bh;
|
||||
xd->n8_w = bw;
|
||||
xd->n4_h = bh;
|
||||
xd->n4_w = bw;
|
||||
xd->is_sec_rect = 0;
|
||||
if (xd->n8_w < xd->n8_h) {
|
||||
if (xd->n4_w < xd->n4_h) {
|
||||
// Only mark is_sec_rect as 1 for the last block.
|
||||
// For PARTITION_VERT_4, it would be (0, 0, 0, 1);
|
||||
// For other partitions, it would be (0, 1).
|
||||
if (!((mi_col + xd->n8_w) & (xd->n8_h - 1))) xd->is_sec_rect = 1;
|
||||
if (!((mi_col + xd->n4_w) & (xd->n4_h - 1))) xd->is_sec_rect = 1;
|
||||
}
|
||||
|
||||
if (xd->n8_w > xd->n8_h)
|
||||
if (mi_row & (xd->n8_w - 1)) xd->is_sec_rect = 1;
|
||||
if (xd->n4_w > xd->n4_h)
|
||||
if (mi_row & (xd->n4_w - 1)) xd->is_sec_rect = 1;
|
||||
}
|
||||
|
||||
static INLINE aom_cdf_prob *get_y_mode_cdf(FRAME_CONTEXT *tile_ctx,
|
||||
|
|
@ -1115,18 +1116,18 @@ static INLINE void set_txfm_ctx(TXFM_CONTEXT *txfm_ctx, uint8_t txs, int len) {
|
|||
for (i = 0; i < len; ++i) txfm_ctx[i] = txs;
|
||||
}
|
||||
|
||||
static INLINE void set_txfm_ctxs(TX_SIZE tx_size, int n8_w, int n8_h, int skip,
|
||||
static INLINE void set_txfm_ctxs(TX_SIZE tx_size, int n4_w, int n4_h, int skip,
|
||||
const MACROBLOCKD *xd) {
|
||||
uint8_t bw = tx_size_wide[tx_size];
|
||||
uint8_t bh = tx_size_high[tx_size];
|
||||
|
||||
if (skip) {
|
||||
bw = n8_w * MI_SIZE;
|
||||
bh = n8_h * MI_SIZE;
|
||||
bw = n4_w * MI_SIZE;
|
||||
bh = n4_h * MI_SIZE;
|
||||
}
|
||||
|
||||
set_txfm_ctx(xd->above_txfm_context, bw, n8_w);
|
||||
set_txfm_ctx(xd->left_txfm_context, bh, n8_h);
|
||||
set_txfm_ctx(xd->above_txfm_context, bw, n4_w);
|
||||
set_txfm_ctx(xd->left_txfm_context, bh, n4_h);
|
||||
}
|
||||
|
||||
static INLINE void txfm_partition_update(TXFM_CONTEXT *above_ctx,
|
||||
|
|
@ -1338,4 +1339,4 @@ static INLINE uint8_t major_minor_to_seq_level_idx(BitstreamLevel bl) {
|
|||
} // extern "C"
|
||||
#endif
|
||||
|
||||
#endif // AV1_COMMON_ONYXC_INT_H_
|
||||
#endif // AOM_AV1_COMMON_ONYXC_INT_H_
|
||||
|
|
|
|||
85
third_party/aom/av1/common/ppc/cfl_ppc.c
vendored
85
third_party/aom/av1/common/ppc/cfl_ppc.c
vendored
|
|
@ -24,19 +24,21 @@
|
|||
#define CFL_LINE_2 128
|
||||
#define CFL_LINE_3 192
|
||||
|
||||
typedef vector int8_t int8x16_t;
|
||||
typedef vector uint8_t uint8x16_t;
|
||||
typedef vector int16_t int16x8_t;
|
||||
typedef vector uint16_t uint16x8_t;
|
||||
typedef vector int32_t int32x4_t;
|
||||
typedef vector uint32_t uint32x4_t;
|
||||
typedef vector uint64_t uint64x2_t;
|
||||
typedef vector signed char int8x16_t; // NOLINT(runtime/int)
|
||||
typedef vector unsigned char uint8x16_t; // NOLINT(runtime/int)
|
||||
typedef vector signed short int16x8_t; // NOLINT(runtime/int)
|
||||
typedef vector unsigned short uint16x8_t; // NOLINT(runtime/int)
|
||||
typedef vector signed int int32x4_t; // NOLINT(runtime/int)
|
||||
typedef vector unsigned int uint32x4_t; // NOLINT(runtime/int)
|
||||
typedef vector unsigned long long uint64x2_t; // NOLINT(runtime/int)
|
||||
|
||||
static INLINE void subtract_average_vsx(int16_t *pred_buf, int width,
|
||||
int height, int round_offset,
|
||||
static INLINE void subtract_average_vsx(const uint16_t *src_ptr, int16_t *dst,
|
||||
int width, int height, int round_offset,
|
||||
int num_pel_log2) {
|
||||
const int16_t *end = pred_buf + height * CFL_BUF_LINE;
|
||||
const int16_t *sum_buf = pred_buf;
|
||||
// int16_t *dst = dst_ptr;
|
||||
const int16_t *dst_end = dst + height * CFL_BUF_LINE;
|
||||
const int16_t *sum_buf = (const int16_t *)src_ptr;
|
||||
const int16_t *end = sum_buf + height * CFL_BUF_LINE;
|
||||
const uint32x4_t div_shift = vec_splats((uint32_t)num_pel_log2);
|
||||
const uint8x16_t mask_64 = { 0x08, 0x09, 0x0A, 0x0B, 0x0C, 0x0D, 0x0E, 0x0F,
|
||||
0x00, 0x01, 0x02, 0x03, 0x04, 0x05, 0x06, 0x07 };
|
||||
|
|
@ -71,43 +73,40 @@ static INLINE void subtract_average_vsx(int16_t *pred_buf, int width,
|
|||
const int32x4_t avg = vec_sr(sum_32x4, div_shift);
|
||||
const int16x8_t vec_avg = vec_pack(avg, avg);
|
||||
do {
|
||||
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_0, pred_buf), vec_avg), OFF_0, pred_buf);
|
||||
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_0 + CFL_LINE_1, pred_buf), vec_avg),
|
||||
OFF_0 + CFL_BUF_LINE_BYTES, pred_buf);
|
||||
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_0 + CFL_LINE_2, pred_buf), vec_avg),
|
||||
OFF_0 + CFL_LINE_2, pred_buf);
|
||||
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_0 + CFL_LINE_3, pred_buf), vec_avg),
|
||||
OFF_0 + CFL_LINE_3, pred_buf);
|
||||
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_0, dst), vec_avg), OFF_0, dst);
|
||||
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_0 + CFL_LINE_1, dst), vec_avg),
|
||||
OFF_0 + CFL_BUF_LINE_BYTES, dst);
|
||||
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_0 + CFL_LINE_2, dst), vec_avg),
|
||||
OFF_0 + CFL_LINE_2, dst);
|
||||
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_0 + CFL_LINE_3, dst), vec_avg),
|
||||
OFF_0 + CFL_LINE_3, dst);
|
||||
if (width >= 16) {
|
||||
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_1, pred_buf), vec_avg), OFF_1,
|
||||
pred_buf);
|
||||
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_1 + CFL_LINE_1, pred_buf), vec_avg),
|
||||
OFF_1 + CFL_LINE_1, pred_buf);
|
||||
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_1 + CFL_LINE_2, pred_buf), vec_avg),
|
||||
OFF_1 + CFL_LINE_2, pred_buf);
|
||||
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_1 + CFL_LINE_3, pred_buf), vec_avg),
|
||||
OFF_1 + CFL_LINE_3, pred_buf);
|
||||
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_1, dst), vec_avg), OFF_1, dst);
|
||||
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_1 + CFL_LINE_1, dst), vec_avg),
|
||||
OFF_1 + CFL_LINE_1, dst);
|
||||
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_1 + CFL_LINE_2, dst), vec_avg),
|
||||
OFF_1 + CFL_LINE_2, dst);
|
||||
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_1 + CFL_LINE_3, dst), vec_avg),
|
||||
OFF_1 + CFL_LINE_3, dst);
|
||||
}
|
||||
if (width == 32) {
|
||||
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_2, pred_buf), vec_avg), OFF_2,
|
||||
pred_buf);
|
||||
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_2 + CFL_LINE_1, pred_buf), vec_avg),
|
||||
OFF_2 + CFL_LINE_1, pred_buf);
|
||||
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_2 + CFL_LINE_2, pred_buf), vec_avg),
|
||||
OFF_2 + CFL_LINE_2, pred_buf);
|
||||
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_2 + CFL_LINE_3, pred_buf), vec_avg),
|
||||
OFF_2 + CFL_LINE_3, pred_buf);
|
||||
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_2, dst), vec_avg), OFF_2, dst);
|
||||
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_2 + CFL_LINE_1, dst), vec_avg),
|
||||
OFF_2 + CFL_LINE_1, dst);
|
||||
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_2 + CFL_LINE_2, dst), vec_avg),
|
||||
OFF_2 + CFL_LINE_2, dst);
|
||||
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_2 + CFL_LINE_3, dst), vec_avg),
|
||||
OFF_2 + CFL_LINE_3, dst);
|
||||
|
||||
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_3, pred_buf), vec_avg), OFF_3,
|
||||
pred_buf);
|
||||
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_3 + CFL_LINE_1, pred_buf), vec_avg),
|
||||
OFF_3 + CFL_LINE_1, pred_buf);
|
||||
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_3 + CFL_LINE_2, pred_buf), vec_avg),
|
||||
OFF_3 + CFL_LINE_2, pred_buf);
|
||||
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_3 + CFL_LINE_3, pred_buf), vec_avg),
|
||||
OFF_3 + CFL_LINE_3, pred_buf);
|
||||
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_3, dst), vec_avg), OFF_3, dst);
|
||||
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_3 + CFL_LINE_1, dst), vec_avg),
|
||||
OFF_3 + CFL_LINE_1, dst);
|
||||
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_3 + CFL_LINE_2, dst), vec_avg),
|
||||
OFF_3 + CFL_LINE_2, dst);
|
||||
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_3 + CFL_LINE_3, dst), vec_avg),
|
||||
OFF_3 + CFL_LINE_3, dst);
|
||||
}
|
||||
} while ((pred_buf += CFL_BUF_LINE * 4) < end);
|
||||
} while ((dst += CFL_BUF_LINE * 4) < dst_end);
|
||||
}
|
||||
|
||||
// Declare wrappers for VSX sizes
|
||||
|
|
|
|||
4
third_party/aom/av1/common/pred_common.c
vendored
4
third_party/aom/av1/common/pred_common.c
vendored
|
|
@ -31,8 +31,8 @@ int av1_get_pred_context_switchable_interp(const MACROBLOCKD *xd, int dir) {
|
|||
const MB_MODE_INFO *const mbmi = xd->mi[0];
|
||||
const int ctx_offset =
|
||||
(mbmi->ref_frame[1] > INTRA_FRAME) * INTER_FILTER_COMP_OFFSET;
|
||||
MV_REFERENCE_FRAME ref_frame =
|
||||
(dir < 2) ? mbmi->ref_frame[0] : mbmi->ref_frame[1];
|
||||
assert(dir == 0 || dir == 1);
|
||||
const MV_REFERENCE_FRAME ref_frame = mbmi->ref_frame[0];
|
||||
// Note:
|
||||
// The mode info data structure has a one element border above and to the
|
||||
// left of the entries corresponding to real macroblocks.
|
||||
|
|
|
|||
6
third_party/aom/av1/common/pred_common.h
vendored
6
third_party/aom/av1/common/pred_common.h
vendored
|
|
@ -9,8 +9,8 @@
|
|||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#ifndef AV1_COMMON_PRED_COMMON_H_
|
||||
#define AV1_COMMON_PRED_COMMON_H_
|
||||
#ifndef AOM_AV1_COMMON_PRED_COMMON_H_
|
||||
#define AOM_AV1_COMMON_PRED_COMMON_H_
|
||||
|
||||
#include "av1/common/blockd.h"
|
||||
#include "av1/common/mvref_common.h"
|
||||
|
|
@ -357,4 +357,4 @@ static INLINE int get_tx_size_context(const MACROBLOCKD *xd) {
|
|||
} // extern "C"
|
||||
#endif
|
||||
|
||||
#endif // AV1_COMMON_PRED_COMMON_H_
|
||||
#endif // AOM_AV1_COMMON_PRED_COMMON_H_
|
||||
|
|
|
|||
6
third_party/aom/av1/common/quant_common.h
vendored
6
third_party/aom/av1/common/quant_common.h
vendored
|
|
@ -9,8 +9,8 @@
|
|||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#ifndef AV1_COMMON_QUANT_COMMON_H_
|
||||
#define AV1_COMMON_QUANT_COMMON_H_
|
||||
#ifndef AOM_AV1_COMMON_QUANT_COMMON_H_
|
||||
#define AOM_AV1_COMMON_QUANT_COMMON_H_
|
||||
|
||||
#include "aom/aom_codec.h"
|
||||
#include "av1/common/seg_common.h"
|
||||
|
|
@ -60,4 +60,4 @@ const qm_val_t *av1_qmatrix(struct AV1Common *cm, int qindex, int comp,
|
|||
} // extern "C"
|
||||
#endif
|
||||
|
||||
#endif // AV1_COMMON_QUANT_COMMON_H_
|
||||
#endif // AOM_AV1_COMMON_QUANT_COMMON_H_
|
||||
|
|
|
|||
652
third_party/aom/av1/common/reconinter.c
vendored
652
third_party/aom/av1/common/reconinter.c
vendored
|
|
@ -44,10 +44,9 @@ int av1_allow_warp(const MB_MODE_INFO *const mbmi,
|
|||
|
||||
if (build_for_obmc) return 0;
|
||||
|
||||
if (warp_types->local_warp_allowed && !mbmi->wm_params[0].invalid) {
|
||||
if (warp_types->local_warp_allowed && !mbmi->wm_params.invalid) {
|
||||
if (final_warp_params != NULL)
|
||||
memcpy(final_warp_params, &mbmi->wm_params[0],
|
||||
sizeof(*final_warp_params));
|
||||
memcpy(final_warp_params, &mbmi->wm_params, sizeof(*final_warp_params));
|
||||
return 1;
|
||||
} else if (warp_types->global_warp_allowed && !gm_params->invalid) {
|
||||
if (final_warp_params != NULL)
|
||||
|
|
@ -78,6 +77,9 @@ void av1_make_inter_predictor(const uint8_t *src, int src_stride, uint8_t *dst,
|
|||
av1_allow_warp(mi, warp_types, &xd->global_motion[mi->ref_frame[ref]],
|
||||
build_for_obmc, subpel_params->xs, subpel_params->ys,
|
||||
&final_warp_params));
|
||||
const int is_intrabc = mi->use_intrabc;
|
||||
assert(IMPLIES(is_intrabc, !do_warp));
|
||||
|
||||
if (do_warp && xd->cur_frame_force_integer_mv == 0) {
|
||||
const struct macroblockd_plane *const pd = &xd->plane[plane];
|
||||
const struct buf_2d *const pre_buf = &pd->pre[ref];
|
||||
|
|
@ -88,10 +90,11 @@ void av1_make_inter_predictor(const uint8_t *src, int src_stride, uint8_t *dst,
|
|||
pd->subsampling_x, pd->subsampling_y, conv_params);
|
||||
} else if (xd->cur_buf->flags & YV12_FLAG_HIGHBITDEPTH) {
|
||||
highbd_inter_predictor(src, src_stride, dst, dst_stride, subpel_params, sf,
|
||||
w, h, conv_params, interp_filters, xd->bd);
|
||||
w, h, conv_params, interp_filters, is_intrabc,
|
||||
xd->bd);
|
||||
} else {
|
||||
inter_predictor(src, src_stride, dst, dst_stride, subpel_params, sf, w, h,
|
||||
conv_params, interp_filters);
|
||||
conv_params, interp_filters, is_intrabc);
|
||||
}
|
||||
}
|
||||
|
||||
|
|
@ -574,37 +577,6 @@ static void build_masked_compound_no_round(
|
|||
h, subw, subh, conv_params);
|
||||
}
|
||||
|
||||
static void build_masked_compound(
|
||||
uint8_t *dst, int dst_stride, const uint8_t *src0, int src0_stride,
|
||||
const uint8_t *src1, int src1_stride,
|
||||
const INTERINTER_COMPOUND_DATA *const comp_data, BLOCK_SIZE sb_type, int h,
|
||||
int w) {
|
||||
// Derive subsampling from h and w passed in. May be refactored to
|
||||
// pass in subsampling factors directly.
|
||||
const int subh = (2 << mi_size_high_log2[sb_type]) == h;
|
||||
const int subw = (2 << mi_size_wide_log2[sb_type]) == w;
|
||||
const uint8_t *mask = av1_get_compound_type_mask(comp_data, sb_type);
|
||||
aom_blend_a64_mask(dst, dst_stride, src0, src0_stride, src1, src1_stride,
|
||||
mask, block_size_wide[sb_type], w, h, subw, subh);
|
||||
}
|
||||
|
||||
static void build_masked_compound_highbd(
|
||||
uint8_t *dst_8, int dst_stride, const uint8_t *src0_8, int src0_stride,
|
||||
const uint8_t *src1_8, int src1_stride,
|
||||
const INTERINTER_COMPOUND_DATA *const comp_data, BLOCK_SIZE sb_type, int h,
|
||||
int w, int bd) {
|
||||
// Derive subsampling from h and w passed in. May be refactored to
|
||||
// pass in subsampling factors directly.
|
||||
const int subh = (2 << mi_size_high_log2[sb_type]) == h;
|
||||
const int subw = (2 << mi_size_wide_log2[sb_type]) == w;
|
||||
const uint8_t *mask = av1_get_compound_type_mask(comp_data, sb_type);
|
||||
// const uint8_t *mask =
|
||||
// av1_get_contiguous_soft_mask(wedge_index, wedge_sign, sb_type);
|
||||
aom_highbd_blend_a64_mask(dst_8, dst_stride, src0_8, src0_stride, src1_8,
|
||||
src1_stride, mask, block_size_wide[sb_type], w, h,
|
||||
subw, subh, bd);
|
||||
}
|
||||
|
||||
void av1_make_masked_inter_predictor(
|
||||
const uint8_t *pre, int pre_stride, uint8_t *dst, int dst_stride,
|
||||
const SubpelParams *subpel_params, const struct scale_factors *sf, int w,
|
||||
|
|
@ -653,63 +625,6 @@ void av1_make_masked_inter_predictor(
|
|||
mi->sb_type, h, w, conv_params, xd);
|
||||
}
|
||||
|
||||
// TODO(sarahparker) av1_highbd_build_inter_predictor and
|
||||
// av1_build_inter_predictor should be combined with
|
||||
// av1_make_inter_predictor
|
||||
void av1_highbd_build_inter_predictor(
|
||||
const uint8_t *src, int src_stride, uint8_t *dst, int dst_stride,
|
||||
const MV *src_mv, const struct scale_factors *sf, int w, int h, int ref,
|
||||
InterpFilters interp_filters, const WarpTypesAllowed *warp_types, int p_col,
|
||||
int p_row, int plane, enum mv_precision precision, int x, int y,
|
||||
const MACROBLOCKD *xd, int can_use_previous) {
|
||||
const int is_q4 = precision == MV_PRECISION_Q4;
|
||||
const MV mv_q4 = { is_q4 ? src_mv->row : src_mv->row * 2,
|
||||
is_q4 ? src_mv->col : src_mv->col * 2 };
|
||||
MV32 mv = av1_scale_mv(&mv_q4, x, y, sf);
|
||||
mv.col += SCALE_EXTRA_OFF;
|
||||
mv.row += SCALE_EXTRA_OFF;
|
||||
const SubpelParams subpel_params = { sf->x_step_q4, sf->y_step_q4,
|
||||
mv.col & SCALE_SUBPEL_MASK,
|
||||
mv.row & SCALE_SUBPEL_MASK };
|
||||
ConvolveParams conv_params = get_conv_params(ref, 0, plane, xd->bd);
|
||||
|
||||
src += (mv.row >> SCALE_SUBPEL_BITS) * src_stride +
|
||||
(mv.col >> SCALE_SUBPEL_BITS);
|
||||
|
||||
av1_make_inter_predictor(src, src_stride, dst, dst_stride, &subpel_params, sf,
|
||||
w, h, &conv_params, interp_filters, warp_types,
|
||||
p_col, p_row, plane, ref, xd->mi[0], 0, xd,
|
||||
can_use_previous);
|
||||
}
|
||||
|
||||
void av1_build_inter_predictor(const uint8_t *src, int src_stride, uint8_t *dst,
|
||||
int dst_stride, const MV *src_mv,
|
||||
const struct scale_factors *sf, int w, int h,
|
||||
ConvolveParams *conv_params,
|
||||
InterpFilters interp_filters,
|
||||
const WarpTypesAllowed *warp_types, int p_col,
|
||||
int p_row, int plane, int ref,
|
||||
enum mv_precision precision, int x, int y,
|
||||
const MACROBLOCKD *xd, int can_use_previous) {
|
||||
const int is_q4 = precision == MV_PRECISION_Q4;
|
||||
const MV mv_q4 = { is_q4 ? src_mv->row : src_mv->row * 2,
|
||||
is_q4 ? src_mv->col : src_mv->col * 2 };
|
||||
MV32 mv = av1_scale_mv(&mv_q4, x, y, sf);
|
||||
mv.col += SCALE_EXTRA_OFF;
|
||||
mv.row += SCALE_EXTRA_OFF;
|
||||
|
||||
const SubpelParams subpel_params = { sf->x_step_q4, sf->y_step_q4,
|
||||
mv.col & SCALE_SUBPEL_MASK,
|
||||
mv.row & SCALE_SUBPEL_MASK };
|
||||
src += (mv.row >> SCALE_SUBPEL_BITS) * src_stride +
|
||||
(mv.col >> SCALE_SUBPEL_BITS);
|
||||
|
||||
av1_make_inter_predictor(src, src_stride, dst, dst_stride, &subpel_params, sf,
|
||||
w, h, conv_params, interp_filters, warp_types, p_col,
|
||||
p_row, plane, ref, xd->mi[0], 0, xd,
|
||||
can_use_previous);
|
||||
}
|
||||
|
||||
void av1_jnt_comp_weight_assign(const AV1_COMMON *cm, const MB_MODE_INFO *mbmi,
|
||||
int order_idx, int *fwd_offset, int *bck_offset,
|
||||
int *use_jnt_comp_avg, int is_compound) {
|
||||
|
|
@ -759,279 +674,6 @@ void av1_jnt_comp_weight_assign(const AV1_COMMON *cm, const MB_MODE_INFO *mbmi,
|
|||
*bck_offset = quant_dist_lookup_table[order_idx][i][1 - order];
|
||||
}
|
||||
|
||||
static INLINE void calc_subpel_params(
|
||||
MACROBLOCKD *xd, const struct scale_factors *const sf, const MV mv,
|
||||
int plane, const int pre_x, const int pre_y, int x, int y,
|
||||
struct buf_2d *const pre_buf, uint8_t **pre, SubpelParams *subpel_params,
|
||||
int bw, int bh) {
|
||||
struct macroblockd_plane *const pd = &xd->plane[plane];
|
||||
const int is_scaled = av1_is_scaled(sf);
|
||||
if (is_scaled) {
|
||||
int ssx = pd->subsampling_x;
|
||||
int ssy = pd->subsampling_y;
|
||||
int orig_pos_y = (pre_y + y) << SUBPEL_BITS;
|
||||
orig_pos_y += mv.row * (1 << (1 - ssy));
|
||||
int orig_pos_x = (pre_x + x) << SUBPEL_BITS;
|
||||
orig_pos_x += mv.col * (1 << (1 - ssx));
|
||||
int pos_y = sf->scale_value_y(orig_pos_y, sf);
|
||||
int pos_x = sf->scale_value_x(orig_pos_x, sf);
|
||||
pos_x += SCALE_EXTRA_OFF;
|
||||
pos_y += SCALE_EXTRA_OFF;
|
||||
|
||||
const int top = -AOM_LEFT_TOP_MARGIN_SCALED(ssy);
|
||||
const int left = -AOM_LEFT_TOP_MARGIN_SCALED(ssx);
|
||||
const int bottom = (pre_buf->height + AOM_INTERP_EXTEND)
|
||||
<< SCALE_SUBPEL_BITS;
|
||||
const int right = (pre_buf->width + AOM_INTERP_EXTEND) << SCALE_SUBPEL_BITS;
|
||||
pos_y = clamp(pos_y, top, bottom);
|
||||
pos_x = clamp(pos_x, left, right);
|
||||
|
||||
*pre = pre_buf->buf0 + (pos_y >> SCALE_SUBPEL_BITS) * pre_buf->stride +
|
||||
(pos_x >> SCALE_SUBPEL_BITS);
|
||||
subpel_params->subpel_x = pos_x & SCALE_SUBPEL_MASK;
|
||||
subpel_params->subpel_y = pos_y & SCALE_SUBPEL_MASK;
|
||||
subpel_params->xs = sf->x_step_q4;
|
||||
subpel_params->ys = sf->y_step_q4;
|
||||
} else {
|
||||
const MV mv_q4 = clamp_mv_to_umv_border_sb(
|
||||
xd, &mv, bw, bh, pd->subsampling_x, pd->subsampling_y);
|
||||
subpel_params->xs = subpel_params->ys = SCALE_SUBPEL_SHIFTS;
|
||||
subpel_params->subpel_x = (mv_q4.col & SUBPEL_MASK) << SCALE_EXTRA_BITS;
|
||||
subpel_params->subpel_y = (mv_q4.row & SUBPEL_MASK) << SCALE_EXTRA_BITS;
|
||||
*pre = pre_buf->buf + (y + (mv_q4.row >> SUBPEL_BITS)) * pre_buf->stride +
|
||||
(x + (mv_q4.col >> SUBPEL_BITS));
|
||||
}
|
||||
}
|
||||
|
||||
static INLINE void build_inter_predictors(const AV1_COMMON *cm, MACROBLOCKD *xd,
|
||||
int plane, const MB_MODE_INFO *mi,
|
||||
int build_for_obmc, int bw, int bh,
|
||||
int mi_x, int mi_y) {
|
||||
struct macroblockd_plane *const pd = &xd->plane[plane];
|
||||
int is_compound = has_second_ref(mi);
|
||||
int ref;
|
||||
const int is_intrabc = is_intrabc_block(mi);
|
||||
assert(IMPLIES(is_intrabc, !is_compound));
|
||||
int is_global[2] = { 0, 0 };
|
||||
for (ref = 0; ref < 1 + is_compound; ++ref) {
|
||||
const WarpedMotionParams *const wm = &xd->global_motion[mi->ref_frame[ref]];
|
||||
is_global[ref] = is_global_mv_block(mi, wm->wmtype);
|
||||
}
|
||||
|
||||
const BLOCK_SIZE bsize = mi->sb_type;
|
||||
const int ss_x = pd->subsampling_x;
|
||||
const int ss_y = pd->subsampling_y;
|
||||
int sub8x8_inter = (block_size_wide[bsize] < 8 && ss_x) ||
|
||||
(block_size_high[bsize] < 8 && ss_y);
|
||||
|
||||
if (is_intrabc) sub8x8_inter = 0;
|
||||
|
||||
// For sub8x8 chroma blocks, we may be covering more than one luma block's
|
||||
// worth of pixels. Thus (mi_x, mi_y) may not be the correct coordinates for
|
||||
// the top-left corner of the prediction source - the correct top-left corner
|
||||
// is at (pre_x, pre_y).
|
||||
const int row_start =
|
||||
(block_size_high[bsize] == 4) && ss_y && !build_for_obmc ? -1 : 0;
|
||||
const int col_start =
|
||||
(block_size_wide[bsize] == 4) && ss_x && !build_for_obmc ? -1 : 0;
|
||||
const int pre_x = (mi_x + MI_SIZE * col_start) >> ss_x;
|
||||
const int pre_y = (mi_y + MI_SIZE * row_start) >> ss_y;
|
||||
|
||||
sub8x8_inter = sub8x8_inter && !build_for_obmc;
|
||||
if (sub8x8_inter) {
|
||||
for (int row = row_start; row <= 0 && sub8x8_inter; ++row) {
|
||||
for (int col = col_start; col <= 0; ++col) {
|
||||
const MB_MODE_INFO *this_mbmi = xd->mi[row * xd->mi_stride + col];
|
||||
if (!is_inter_block(this_mbmi)) sub8x8_inter = 0;
|
||||
if (is_intrabc_block(this_mbmi)) sub8x8_inter = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (sub8x8_inter) {
|
||||
// block size
|
||||
const int b4_w = block_size_wide[bsize] >> ss_x;
|
||||
const int b4_h = block_size_high[bsize] >> ss_y;
|
||||
const BLOCK_SIZE plane_bsize = scale_chroma_bsize(bsize, ss_x, ss_y);
|
||||
const int b8_w = block_size_wide[plane_bsize] >> ss_x;
|
||||
const int b8_h = block_size_high[plane_bsize] >> ss_y;
|
||||
assert(!is_compound);
|
||||
|
||||
const struct buf_2d orig_pred_buf[2] = { pd->pre[0], pd->pre[1] };
|
||||
|
||||
int row = row_start;
|
||||
for (int y = 0; y < b8_h; y += b4_h) {
|
||||
int col = col_start;
|
||||
for (int x = 0; x < b8_w; x += b4_w) {
|
||||
MB_MODE_INFO *this_mbmi = xd->mi[row * xd->mi_stride + col];
|
||||
is_compound = has_second_ref(this_mbmi);
|
||||
DECLARE_ALIGNED(32, CONV_BUF_TYPE, tmp_dst[8 * 8]);
|
||||
int tmp_dst_stride = 8;
|
||||
assert(bw < 8 || bh < 8);
|
||||
ConvolveParams conv_params = get_conv_params_no_round(
|
||||
0, 0, plane, tmp_dst, tmp_dst_stride, is_compound, xd->bd);
|
||||
conv_params.use_jnt_comp_avg = 0;
|
||||
struct buf_2d *const dst_buf = &pd->dst;
|
||||
uint8_t *dst = dst_buf->buf + dst_buf->stride * y + x;
|
||||
|
||||
ref = 0;
|
||||
const RefBuffer *ref_buf =
|
||||
&cm->frame_refs[this_mbmi->ref_frame[ref] - LAST_FRAME];
|
||||
|
||||
pd->pre[ref].buf0 =
|
||||
(plane == 1) ? ref_buf->buf->u_buffer : ref_buf->buf->v_buffer;
|
||||
pd->pre[ref].buf =
|
||||
pd->pre[ref].buf0 + scaled_buffer_offset(pre_x, pre_y,
|
||||
ref_buf->buf->uv_stride,
|
||||
&ref_buf->sf);
|
||||
pd->pre[ref].width = ref_buf->buf->uv_crop_width;
|
||||
pd->pre[ref].height = ref_buf->buf->uv_crop_height;
|
||||
pd->pre[ref].stride = ref_buf->buf->uv_stride;
|
||||
|
||||
const struct scale_factors *const sf =
|
||||
is_intrabc ? &cm->sf_identity : &ref_buf->sf;
|
||||
struct buf_2d *const pre_buf = is_intrabc ? dst_buf : &pd->pre[ref];
|
||||
|
||||
const MV mv = this_mbmi->mv[ref].as_mv;
|
||||
|
||||
uint8_t *pre;
|
||||
SubpelParams subpel_params;
|
||||
WarpTypesAllowed warp_types;
|
||||
warp_types.global_warp_allowed = is_global[ref];
|
||||
warp_types.local_warp_allowed = this_mbmi->motion_mode == WARPED_CAUSAL;
|
||||
|
||||
calc_subpel_params(xd, sf, mv, plane, pre_x, pre_y, x, y, pre_buf, &pre,
|
||||
&subpel_params, bw, bh);
|
||||
|
||||
conv_params.ref = ref;
|
||||
conv_params.do_average = ref;
|
||||
if (is_masked_compound_type(mi->interinter_comp.type)) {
|
||||
// masked compound type has its own average mechanism
|
||||
conv_params.do_average = 0;
|
||||
}
|
||||
|
||||
av1_make_inter_predictor(
|
||||
pre, pre_buf->stride, dst, dst_buf->stride, &subpel_params, sf,
|
||||
b4_w, b4_h, &conv_params, this_mbmi->interp_filters, &warp_types,
|
||||
(mi_x >> pd->subsampling_x) + x, (mi_y >> pd->subsampling_y) + y,
|
||||
plane, ref, mi, build_for_obmc, xd, cm->allow_warped_motion);
|
||||
|
||||
++col;
|
||||
}
|
||||
++row;
|
||||
}
|
||||
|
||||
for (ref = 0; ref < 2; ++ref) pd->pre[ref] = orig_pred_buf[ref];
|
||||
return;
|
||||
}
|
||||
|
||||
{
|
||||
DECLARE_ALIGNED(32, uint16_t, tmp_dst[MAX_SB_SIZE * MAX_SB_SIZE]);
|
||||
ConvolveParams conv_params = get_conv_params_no_round(
|
||||
0, 0, plane, tmp_dst, MAX_SB_SIZE, is_compound, xd->bd);
|
||||
av1_jnt_comp_weight_assign(cm, mi, 0, &conv_params.fwd_offset,
|
||||
&conv_params.bck_offset,
|
||||
&conv_params.use_jnt_comp_avg, is_compound);
|
||||
|
||||
struct buf_2d *const dst_buf = &pd->dst;
|
||||
uint8_t *const dst = dst_buf->buf;
|
||||
for (ref = 0; ref < 1 + is_compound; ++ref) {
|
||||
const struct scale_factors *const sf =
|
||||
is_intrabc ? &cm->sf_identity : &xd->block_refs[ref]->sf;
|
||||
struct buf_2d *const pre_buf = is_intrabc ? dst_buf : &pd->pre[ref];
|
||||
const MV mv = mi->mv[ref].as_mv;
|
||||
|
||||
uint8_t *pre;
|
||||
SubpelParams subpel_params;
|
||||
calc_subpel_params(xd, sf, mv, plane, pre_x, pre_y, 0, 0, pre_buf, &pre,
|
||||
&subpel_params, bw, bh);
|
||||
|
||||
WarpTypesAllowed warp_types;
|
||||
warp_types.global_warp_allowed = is_global[ref];
|
||||
warp_types.local_warp_allowed = mi->motion_mode == WARPED_CAUSAL;
|
||||
conv_params.ref = ref;
|
||||
|
||||
if (ref && is_masked_compound_type(mi->interinter_comp.type)) {
|
||||
// masked compound type has its own average mechanism
|
||||
conv_params.do_average = 0;
|
||||
av1_make_masked_inter_predictor(
|
||||
pre, pre_buf->stride, dst, dst_buf->stride, &subpel_params, sf, bw,
|
||||
bh, &conv_params, mi->interp_filters, plane, &warp_types,
|
||||
mi_x >> pd->subsampling_x, mi_y >> pd->subsampling_y, ref, xd,
|
||||
cm->allow_warped_motion);
|
||||
} else {
|
||||
conv_params.do_average = ref;
|
||||
av1_make_inter_predictor(
|
||||
pre, pre_buf->stride, dst, dst_buf->stride, &subpel_params, sf, bw,
|
||||
bh, &conv_params, mi->interp_filters, &warp_types,
|
||||
mi_x >> pd->subsampling_x, mi_y >> pd->subsampling_y, plane, ref,
|
||||
mi, build_for_obmc, xd, cm->allow_warped_motion);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static void build_inter_predictors_for_planes(const AV1_COMMON *cm,
|
||||
MACROBLOCKD *xd, BLOCK_SIZE bsize,
|
||||
int mi_row, int mi_col,
|
||||
int plane_from, int plane_to) {
|
||||
int plane;
|
||||
const int mi_x = mi_col * MI_SIZE;
|
||||
const int mi_y = mi_row * MI_SIZE;
|
||||
for (plane = plane_from; plane <= plane_to; ++plane) {
|
||||
const struct macroblockd_plane *pd = &xd->plane[plane];
|
||||
const int bw = pd->width;
|
||||
const int bh = pd->height;
|
||||
|
||||
if (!is_chroma_reference(mi_row, mi_col, bsize, pd->subsampling_x,
|
||||
pd->subsampling_y))
|
||||
continue;
|
||||
|
||||
build_inter_predictors(cm, xd, plane, xd->mi[0], 0, bw, bh, mi_x, mi_y);
|
||||
}
|
||||
}
|
||||
|
||||
void av1_build_inter_predictors_sby(const AV1_COMMON *cm, MACROBLOCKD *xd,
|
||||
int mi_row, int mi_col, BUFFER_SET *ctx,
|
||||
BLOCK_SIZE bsize) {
|
||||
build_inter_predictors_for_planes(cm, xd, bsize, mi_row, mi_col, 0, 0);
|
||||
|
||||
if (is_interintra_pred(xd->mi[0])) {
|
||||
BUFFER_SET default_ctx = { { xd->plane[0].dst.buf, NULL, NULL },
|
||||
{ xd->plane[0].dst.stride, 0, 0 } };
|
||||
if (!ctx) ctx = &default_ctx;
|
||||
av1_build_interintra_predictors_sbp(cm, xd, xd->plane[0].dst.buf,
|
||||
xd->plane[0].dst.stride, ctx, 0, bsize);
|
||||
}
|
||||
}
|
||||
|
||||
void av1_build_inter_predictors_sbuv(const AV1_COMMON *cm, MACROBLOCKD *xd,
|
||||
int mi_row, int mi_col, BUFFER_SET *ctx,
|
||||
BLOCK_SIZE bsize) {
|
||||
build_inter_predictors_for_planes(cm, xd, bsize, mi_row, mi_col, 1,
|
||||
MAX_MB_PLANE - 1);
|
||||
|
||||
if (is_interintra_pred(xd->mi[0])) {
|
||||
BUFFER_SET default_ctx = {
|
||||
{ NULL, xd->plane[1].dst.buf, xd->plane[2].dst.buf },
|
||||
{ 0, xd->plane[1].dst.stride, xd->plane[2].dst.stride }
|
||||
};
|
||||
if (!ctx) ctx = &default_ctx;
|
||||
av1_build_interintra_predictors_sbuv(
|
||||
cm, xd, xd->plane[1].dst.buf, xd->plane[2].dst.buf,
|
||||
xd->plane[1].dst.stride, xd->plane[2].dst.stride, ctx, bsize);
|
||||
}
|
||||
}
|
||||
|
||||
void av1_build_inter_predictors_sb(const AV1_COMMON *cm, MACROBLOCKD *xd,
|
||||
int mi_row, int mi_col, BUFFER_SET *ctx,
|
||||
BLOCK_SIZE bsize) {
|
||||
const int num_planes = av1_num_planes(cm);
|
||||
av1_build_inter_predictors_sby(cm, xd, mi_row, mi_col, ctx, bsize);
|
||||
if (num_planes > 1)
|
||||
av1_build_inter_predictors_sbuv(cm, xd, mi_row, mi_col, ctx, bsize);
|
||||
}
|
||||
|
||||
void av1_setup_dst_planes(struct macroblockd_plane *planes, BLOCK_SIZE bsize,
|
||||
const YV12_BUFFER_CONFIG *src, int mi_row, int mi_col,
|
||||
const int plane_start, const int plane_end) {
|
||||
|
|
@ -1292,63 +934,7 @@ void av1_setup_build_prediction_by_above_pred(
|
|||
|
||||
xd->mb_to_left_edge = 8 * MI_SIZE * (-above_mi_col);
|
||||
xd->mb_to_right_edge = ctxt->mb_to_far_edge +
|
||||
(xd->n8_w - rel_mi_col - above_mi_width) * MI_SIZE * 8;
|
||||
}
|
||||
|
||||
static INLINE void build_prediction_by_above_pred(
|
||||
MACROBLOCKD *xd, int rel_mi_col, uint8_t above_mi_width,
|
||||
MB_MODE_INFO *above_mbmi, void *fun_ctxt, const int num_planes) {
|
||||
struct build_prediction_ctxt *ctxt = (struct build_prediction_ctxt *)fun_ctxt;
|
||||
const int above_mi_col = ctxt->mi_col + rel_mi_col;
|
||||
int mi_x, mi_y;
|
||||
MB_MODE_INFO backup_mbmi = *above_mbmi;
|
||||
|
||||
av1_setup_build_prediction_by_above_pred(xd, rel_mi_col, above_mi_width,
|
||||
above_mbmi, ctxt, num_planes);
|
||||
mi_x = above_mi_col << MI_SIZE_LOG2;
|
||||
mi_y = ctxt->mi_row << MI_SIZE_LOG2;
|
||||
|
||||
const BLOCK_SIZE bsize = xd->mi[0]->sb_type;
|
||||
|
||||
for (int j = 0; j < num_planes; ++j) {
|
||||
const struct macroblockd_plane *pd = &xd->plane[j];
|
||||
int bw = (above_mi_width * MI_SIZE) >> pd->subsampling_x;
|
||||
int bh = clamp(block_size_high[bsize] >> (pd->subsampling_y + 1), 4,
|
||||
block_size_high[BLOCK_64X64] >> (pd->subsampling_y + 1));
|
||||
|
||||
if (av1_skip_u4x4_pred_in_obmc(bsize, pd, 0)) continue;
|
||||
build_inter_predictors(ctxt->cm, xd, j, above_mbmi, 1, bw, bh, mi_x, mi_y);
|
||||
}
|
||||
*above_mbmi = backup_mbmi;
|
||||
}
|
||||
|
||||
void av1_build_prediction_by_above_preds(const AV1_COMMON *cm, MACROBLOCKD *xd,
|
||||
int mi_row, int mi_col,
|
||||
uint8_t *tmp_buf[MAX_MB_PLANE],
|
||||
int tmp_width[MAX_MB_PLANE],
|
||||
int tmp_height[MAX_MB_PLANE],
|
||||
int tmp_stride[MAX_MB_PLANE]) {
|
||||
if (!xd->up_available) return;
|
||||
|
||||
// Adjust mb_to_bottom_edge to have the correct value for the OBMC
|
||||
// prediction block. This is half the height of the original block,
|
||||
// except for 128-wide blocks, where we only use a height of 32.
|
||||
int this_height = xd->n8_h * MI_SIZE;
|
||||
int pred_height = AOMMIN(this_height / 2, 32);
|
||||
xd->mb_to_bottom_edge += (this_height - pred_height) * 8;
|
||||
|
||||
struct build_prediction_ctxt ctxt = { cm, mi_row,
|
||||
mi_col, tmp_buf,
|
||||
tmp_width, tmp_height,
|
||||
tmp_stride, xd->mb_to_right_edge };
|
||||
BLOCK_SIZE bsize = xd->mi[0]->sb_type;
|
||||
foreach_overlappable_nb_above(cm, xd, mi_col,
|
||||
max_neighbor_obmc[mi_size_wide_log2[bsize]],
|
||||
build_prediction_by_above_pred, &ctxt);
|
||||
|
||||
xd->mb_to_left_edge = -((mi_col * MI_SIZE) * 8);
|
||||
xd->mb_to_right_edge = ctxt.mb_to_far_edge;
|
||||
xd->mb_to_bottom_edge -= (this_height - pred_height) * 8;
|
||||
(xd->n4_w - rel_mi_col - above_mi_width) * MI_SIZE * 8;
|
||||
}
|
||||
|
||||
void av1_setup_build_prediction_by_left_pred(MACROBLOCKD *xd, int rel_mi_row,
|
||||
|
|
@ -1386,101 +972,7 @@ void av1_setup_build_prediction_by_left_pred(MACROBLOCKD *xd, int rel_mi_row,
|
|||
xd->mb_to_top_edge = 8 * MI_SIZE * (-left_mi_row);
|
||||
xd->mb_to_bottom_edge =
|
||||
ctxt->mb_to_far_edge +
|
||||
(xd->n8_h - rel_mi_row - left_mi_height) * MI_SIZE * 8;
|
||||
}
|
||||
|
||||
static INLINE void build_prediction_by_left_pred(
|
||||
MACROBLOCKD *xd, int rel_mi_row, uint8_t left_mi_height,
|
||||
MB_MODE_INFO *left_mbmi, void *fun_ctxt, const int num_planes) {
|
||||
struct build_prediction_ctxt *ctxt = (struct build_prediction_ctxt *)fun_ctxt;
|
||||
const int left_mi_row = ctxt->mi_row + rel_mi_row;
|
||||
int mi_x, mi_y;
|
||||
MB_MODE_INFO backup_mbmi = *left_mbmi;
|
||||
|
||||
av1_setup_build_prediction_by_left_pred(xd, rel_mi_row, left_mi_height,
|
||||
left_mbmi, ctxt, num_planes);
|
||||
mi_x = ctxt->mi_col << MI_SIZE_LOG2;
|
||||
mi_y = left_mi_row << MI_SIZE_LOG2;
|
||||
const BLOCK_SIZE bsize = xd->mi[0]->sb_type;
|
||||
|
||||
for (int j = 0; j < num_planes; ++j) {
|
||||
const struct macroblockd_plane *pd = &xd->plane[j];
|
||||
int bw = clamp(block_size_wide[bsize] >> (pd->subsampling_x + 1), 4,
|
||||
block_size_wide[BLOCK_64X64] >> (pd->subsampling_x + 1));
|
||||
int bh = (left_mi_height << MI_SIZE_LOG2) >> pd->subsampling_y;
|
||||
|
||||
if (av1_skip_u4x4_pred_in_obmc(bsize, pd, 1)) continue;
|
||||
build_inter_predictors(ctxt->cm, xd, j, left_mbmi, 1, bw, bh, mi_x, mi_y);
|
||||
}
|
||||
*left_mbmi = backup_mbmi;
|
||||
}
|
||||
|
||||
void av1_build_prediction_by_left_preds(const AV1_COMMON *cm, MACROBLOCKD *xd,
|
||||
int mi_row, int mi_col,
|
||||
uint8_t *tmp_buf[MAX_MB_PLANE],
|
||||
int tmp_width[MAX_MB_PLANE],
|
||||
int tmp_height[MAX_MB_PLANE],
|
||||
int tmp_stride[MAX_MB_PLANE]) {
|
||||
if (!xd->left_available) return;
|
||||
|
||||
// Adjust mb_to_right_edge to have the correct value for the OBMC
|
||||
// prediction block. This is half the width of the original block,
|
||||
// except for 128-wide blocks, where we only use a width of 32.
|
||||
int this_width = xd->n8_w * MI_SIZE;
|
||||
int pred_width = AOMMIN(this_width / 2, 32);
|
||||
xd->mb_to_right_edge += (this_width - pred_width) * 8;
|
||||
|
||||
struct build_prediction_ctxt ctxt = { cm, mi_row,
|
||||
mi_col, tmp_buf,
|
||||
tmp_width, tmp_height,
|
||||
tmp_stride, xd->mb_to_bottom_edge };
|
||||
BLOCK_SIZE bsize = xd->mi[0]->sb_type;
|
||||
foreach_overlappable_nb_left(cm, xd, mi_row,
|
||||
max_neighbor_obmc[mi_size_high_log2[bsize]],
|
||||
build_prediction_by_left_pred, &ctxt);
|
||||
|
||||
xd->mb_to_top_edge = -((mi_row * MI_SIZE) * 8);
|
||||
xd->mb_to_right_edge -= (this_width - pred_width) * 8;
|
||||
xd->mb_to_bottom_edge = ctxt.mb_to_far_edge;
|
||||
}
|
||||
|
||||
void av1_build_obmc_inter_predictors_sb(const AV1_COMMON *cm, MACROBLOCKD *xd,
|
||||
int mi_row, int mi_col) {
|
||||
const int num_planes = av1_num_planes(cm);
|
||||
DECLARE_ALIGNED(16, uint8_t, tmp_buf1[2 * MAX_MB_PLANE * MAX_SB_SQUARE]);
|
||||
DECLARE_ALIGNED(16, uint8_t, tmp_buf2[2 * MAX_MB_PLANE * MAX_SB_SQUARE]);
|
||||
uint8_t *dst_buf1[MAX_MB_PLANE], *dst_buf2[MAX_MB_PLANE];
|
||||
int dst_stride1[MAX_MB_PLANE] = { MAX_SB_SIZE, MAX_SB_SIZE, MAX_SB_SIZE };
|
||||
int dst_stride2[MAX_MB_PLANE] = { MAX_SB_SIZE, MAX_SB_SIZE, MAX_SB_SIZE };
|
||||
int dst_width1[MAX_MB_PLANE] = { MAX_SB_SIZE, MAX_SB_SIZE, MAX_SB_SIZE };
|
||||
int dst_width2[MAX_MB_PLANE] = { MAX_SB_SIZE, MAX_SB_SIZE, MAX_SB_SIZE };
|
||||
int dst_height1[MAX_MB_PLANE] = { MAX_SB_SIZE, MAX_SB_SIZE, MAX_SB_SIZE };
|
||||
int dst_height2[MAX_MB_PLANE] = { MAX_SB_SIZE, MAX_SB_SIZE, MAX_SB_SIZE };
|
||||
|
||||
if (xd->cur_buf->flags & YV12_FLAG_HIGHBITDEPTH) {
|
||||
int len = sizeof(uint16_t);
|
||||
dst_buf1[0] = CONVERT_TO_BYTEPTR(tmp_buf1);
|
||||
dst_buf1[1] = CONVERT_TO_BYTEPTR(tmp_buf1 + MAX_SB_SQUARE * len);
|
||||
dst_buf1[2] = CONVERT_TO_BYTEPTR(tmp_buf1 + MAX_SB_SQUARE * 2 * len);
|
||||
dst_buf2[0] = CONVERT_TO_BYTEPTR(tmp_buf2);
|
||||
dst_buf2[1] = CONVERT_TO_BYTEPTR(tmp_buf2 + MAX_SB_SQUARE * len);
|
||||
dst_buf2[2] = CONVERT_TO_BYTEPTR(tmp_buf2 + MAX_SB_SQUARE * 2 * len);
|
||||
} else {
|
||||
dst_buf1[0] = tmp_buf1;
|
||||
dst_buf1[1] = tmp_buf1 + MAX_SB_SQUARE;
|
||||
dst_buf1[2] = tmp_buf1 + MAX_SB_SQUARE * 2;
|
||||
dst_buf2[0] = tmp_buf2;
|
||||
dst_buf2[1] = tmp_buf2 + MAX_SB_SQUARE;
|
||||
dst_buf2[2] = tmp_buf2 + MAX_SB_SQUARE * 2;
|
||||
}
|
||||
av1_build_prediction_by_above_preds(cm, xd, mi_row, mi_col, dst_buf1,
|
||||
dst_width1, dst_height1, dst_stride1);
|
||||
av1_build_prediction_by_left_preds(cm, xd, mi_row, mi_col, dst_buf2,
|
||||
dst_width2, dst_height2, dst_stride2);
|
||||
av1_setup_dst_planes(xd->plane, xd->mi[0]->sb_type, get_frame_new_buffer(cm),
|
||||
mi_row, mi_col, 0, num_planes);
|
||||
av1_build_obmc_inter_prediction(cm, xd, mi_row, mi_col, dst_buf1, dst_stride1,
|
||||
dst_buf2, dst_stride2);
|
||||
(xd->n4_h - rel_mi_row - left_mi_height) * MI_SIZE * 8;
|
||||
}
|
||||
|
||||
/* clang-format off */
|
||||
|
|
@ -1668,127 +1160,3 @@ void av1_build_interintra_predictors_sbuv(const AV1_COMMON *cm, MACROBLOCKD *xd,
|
|||
av1_build_interintra_predictors_sbp(cm, xd, upred, ustride, ctx, 1, bsize);
|
||||
av1_build_interintra_predictors_sbp(cm, xd, vpred, vstride, ctx, 2, bsize);
|
||||
}
|
||||
|
||||
void av1_build_interintra_predictors(const AV1_COMMON *cm, MACROBLOCKD *xd,
|
||||
uint8_t *ypred, uint8_t *upred,
|
||||
uint8_t *vpred, int ystride, int ustride,
|
||||
int vstride, BUFFER_SET *ctx,
|
||||
BLOCK_SIZE bsize) {
|
||||
av1_build_interintra_predictors_sbp(cm, xd, ypred, ystride, ctx, 0, bsize);
|
||||
av1_build_interintra_predictors_sbuv(cm, xd, upred, vpred, ustride, vstride,
|
||||
ctx, bsize);
|
||||
}
|
||||
|
||||
// Builds the inter-predictor for the single ref case
|
||||
// for use in the encoder to search the wedges efficiently.
|
||||
static void build_inter_predictors_single_buf(MACROBLOCKD *xd, int plane,
|
||||
int bw, int bh, int x, int y,
|
||||
int w, int h, int mi_x, int mi_y,
|
||||
int ref, uint8_t *const ext_dst,
|
||||
int ext_dst_stride,
|
||||
int can_use_previous) {
|
||||
struct macroblockd_plane *const pd = &xd->plane[plane];
|
||||
const MB_MODE_INFO *mi = xd->mi[0];
|
||||
|
||||
const struct scale_factors *const sf = &xd->block_refs[ref]->sf;
|
||||
struct buf_2d *const pre_buf = &pd->pre[ref];
|
||||
uint8_t *const dst = get_buf_by_bd(xd, ext_dst) + ext_dst_stride * y + x;
|
||||
const MV mv = mi->mv[ref].as_mv;
|
||||
|
||||
ConvolveParams conv_params = get_conv_params(ref, 0, plane, xd->bd);
|
||||
WarpTypesAllowed warp_types;
|
||||
const WarpedMotionParams *const wm = &xd->global_motion[mi->ref_frame[ref]];
|
||||
warp_types.global_warp_allowed = is_global_mv_block(mi, wm->wmtype);
|
||||
warp_types.local_warp_allowed = mi->motion_mode == WARPED_CAUSAL;
|
||||
const int pre_x = (mi_x) >> pd->subsampling_x;
|
||||
const int pre_y = (mi_y) >> pd->subsampling_y;
|
||||
uint8_t *pre;
|
||||
SubpelParams subpel_params;
|
||||
calc_subpel_params(xd, sf, mv, plane, pre_x, pre_y, x, y, pre_buf, &pre,
|
||||
&subpel_params, bw, bh);
|
||||
|
||||
av1_make_inter_predictor(pre, pre_buf->stride, dst, ext_dst_stride,
|
||||
&subpel_params, sf, w, h, &conv_params,
|
||||
mi->interp_filters, &warp_types, pre_x + x,
|
||||
pre_y + y, plane, ref, mi, 0, xd, can_use_previous);
|
||||
}
|
||||
|
||||
void av1_build_inter_predictors_for_planes_single_buf(
|
||||
MACROBLOCKD *xd, BLOCK_SIZE bsize, int plane_from, int plane_to, int mi_row,
|
||||
int mi_col, int ref, uint8_t *ext_dst[3], int ext_dst_stride[3],
|
||||
int can_use_previous) {
|
||||
int plane;
|
||||
const int mi_x = mi_col * MI_SIZE;
|
||||
const int mi_y = mi_row * MI_SIZE;
|
||||
for (plane = plane_from; plane <= plane_to; ++plane) {
|
||||
const BLOCK_SIZE plane_bsize = get_plane_block_size(
|
||||
bsize, xd->plane[plane].subsampling_x, xd->plane[plane].subsampling_y);
|
||||
const int bw = block_size_wide[plane_bsize];
|
||||
const int bh = block_size_high[plane_bsize];
|
||||
build_inter_predictors_single_buf(xd, plane, bw, bh, 0, 0, bw, bh, mi_x,
|
||||
mi_y, ref, ext_dst[plane],
|
||||
ext_dst_stride[plane], can_use_previous);
|
||||
}
|
||||
}
|
||||
|
||||
static void build_wedge_inter_predictor_from_buf(
|
||||
MACROBLOCKD *xd, int plane, int x, int y, int w, int h, uint8_t *ext_dst0,
|
||||
int ext_dst_stride0, uint8_t *ext_dst1, int ext_dst_stride1) {
|
||||
MB_MODE_INFO *const mbmi = xd->mi[0];
|
||||
const int is_compound = has_second_ref(mbmi);
|
||||
MACROBLOCKD_PLANE *const pd = &xd->plane[plane];
|
||||
struct buf_2d *const dst_buf = &pd->dst;
|
||||
uint8_t *const dst = dst_buf->buf + dst_buf->stride * y + x;
|
||||
mbmi->interinter_comp.seg_mask = xd->seg_mask;
|
||||
const INTERINTER_COMPOUND_DATA *comp_data = &mbmi->interinter_comp;
|
||||
|
||||
if (is_compound && is_masked_compound_type(comp_data->type)) {
|
||||
if (!plane && comp_data->type == COMPOUND_DIFFWTD) {
|
||||
if (xd->cur_buf->flags & YV12_FLAG_HIGHBITDEPTH)
|
||||
av1_build_compound_diffwtd_mask_highbd(
|
||||
comp_data->seg_mask, comp_data->mask_type,
|
||||
CONVERT_TO_BYTEPTR(ext_dst0), ext_dst_stride0,
|
||||
CONVERT_TO_BYTEPTR(ext_dst1), ext_dst_stride1, h, w, xd->bd);
|
||||
else
|
||||
av1_build_compound_diffwtd_mask(
|
||||
comp_data->seg_mask, comp_data->mask_type, ext_dst0,
|
||||
ext_dst_stride0, ext_dst1, ext_dst_stride1, h, w);
|
||||
}
|
||||
|
||||
if (xd->cur_buf->flags & YV12_FLAG_HIGHBITDEPTH)
|
||||
build_masked_compound_highbd(
|
||||
dst, dst_buf->stride, CONVERT_TO_BYTEPTR(ext_dst0), ext_dst_stride0,
|
||||
CONVERT_TO_BYTEPTR(ext_dst1), ext_dst_stride1, comp_data,
|
||||
mbmi->sb_type, h, w, xd->bd);
|
||||
else
|
||||
build_masked_compound(dst, dst_buf->stride, ext_dst0, ext_dst_stride0,
|
||||
ext_dst1, ext_dst_stride1, comp_data, mbmi->sb_type,
|
||||
h, w);
|
||||
} else {
|
||||
if (xd->cur_buf->flags & YV12_FLAG_HIGHBITDEPTH)
|
||||
aom_highbd_convolve_copy(CONVERT_TO_BYTEPTR(ext_dst0), ext_dst_stride0,
|
||||
dst, dst_buf->stride, NULL, 0, NULL, 0, w, h,
|
||||
xd->bd);
|
||||
else
|
||||
aom_convolve_copy(ext_dst0, ext_dst_stride0, dst, dst_buf->stride, NULL,
|
||||
0, NULL, 0, w, h);
|
||||
}
|
||||
}
|
||||
|
||||
void av1_build_wedge_inter_predictor_from_buf(MACROBLOCKD *xd, BLOCK_SIZE bsize,
|
||||
int plane_from, int plane_to,
|
||||
uint8_t *ext_dst0[3],
|
||||
int ext_dst_stride0[3],
|
||||
uint8_t *ext_dst1[3],
|
||||
int ext_dst_stride1[3]) {
|
||||
int plane;
|
||||
for (plane = plane_from; plane <= plane_to; ++plane) {
|
||||
const BLOCK_SIZE plane_bsize = get_plane_block_size(
|
||||
bsize, xd->plane[plane].subsampling_x, xd->plane[plane].subsampling_y);
|
||||
const int bw = block_size_wide[plane_bsize];
|
||||
const int bh = block_size_high[plane_bsize];
|
||||
build_wedge_inter_predictor_from_buf(
|
||||
xd, plane, 0, 0, bw, bh, ext_dst0[plane], ext_dst_stride0[plane],
|
||||
ext_dst1[plane], ext_dst_stride1[plane]);
|
||||
}
|
||||
}
|
||||
|
|
|
|||
148
third_party/aom/av1/common/reconinter.h
vendored
148
third_party/aom/av1/common/reconinter.h
vendored
|
|
@ -9,8 +9,8 @@
|
|||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#ifndef AV1_COMMON_RECONINTER_H_
|
||||
#define AV1_COMMON_RECONINTER_H_
|
||||
#ifndef AOM_AV1_COMMON_RECONINTER_H_
|
||||
#define AOM_AV1_COMMON_RECONINTER_H_
|
||||
|
||||
#include "av1/common/filter.h"
|
||||
#include "av1/common/onyxc_int.h"
|
||||
|
|
@ -113,40 +113,48 @@ static INLINE void inter_predictor(const uint8_t *src, int src_stride,
|
|||
const SubpelParams *subpel_params,
|
||||
const struct scale_factors *sf, int w, int h,
|
||||
ConvolveParams *conv_params,
|
||||
InterpFilters interp_filters) {
|
||||
InterpFilters interp_filters,
|
||||
int is_intrabc) {
|
||||
assert(conv_params->do_average == 0 || conv_params->do_average == 1);
|
||||
assert(sf);
|
||||
if (has_scale(subpel_params->xs, subpel_params->ys)) {
|
||||
const int is_scaled = has_scale(subpel_params->xs, subpel_params->ys);
|
||||
assert(IMPLIES(is_intrabc, !is_scaled));
|
||||
if (is_scaled) {
|
||||
av1_convolve_2d_facade(src, src_stride, dst, dst_stride, w, h,
|
||||
interp_filters, subpel_params->subpel_x,
|
||||
subpel_params->xs, subpel_params->subpel_y,
|
||||
subpel_params->ys, 1, conv_params, sf);
|
||||
subpel_params->ys, 1, conv_params, sf, is_intrabc);
|
||||
} else {
|
||||
SubpelParams sp = *subpel_params;
|
||||
revert_scale_extra_bits(&sp);
|
||||
av1_convolve_2d_facade(src, src_stride, dst, dst_stride, w, h,
|
||||
interp_filters, sp.subpel_x, sp.xs, sp.subpel_y,
|
||||
sp.ys, 0, conv_params, sf);
|
||||
sp.ys, 0, conv_params, sf, is_intrabc);
|
||||
}
|
||||
}
|
||||
|
||||
static INLINE void highbd_inter_predictor(
|
||||
const uint8_t *src, int src_stride, uint8_t *dst, int dst_stride,
|
||||
const SubpelParams *subpel_params, const struct scale_factors *sf, int w,
|
||||
int h, ConvolveParams *conv_params, InterpFilters interp_filters, int bd) {
|
||||
static INLINE void highbd_inter_predictor(const uint8_t *src, int src_stride,
|
||||
uint8_t *dst, int dst_stride,
|
||||
const SubpelParams *subpel_params,
|
||||
const struct scale_factors *sf, int w,
|
||||
int h, ConvolveParams *conv_params,
|
||||
InterpFilters interp_filters,
|
||||
int is_intrabc, int bd) {
|
||||
assert(conv_params->do_average == 0 || conv_params->do_average == 1);
|
||||
assert(sf);
|
||||
if (has_scale(subpel_params->xs, subpel_params->ys)) {
|
||||
av1_highbd_convolve_2d_facade(src, src_stride, dst, dst_stride, w, h,
|
||||
interp_filters, subpel_params->subpel_x,
|
||||
subpel_params->xs, subpel_params->subpel_y,
|
||||
subpel_params->ys, 1, conv_params, sf, bd);
|
||||
const int is_scaled = has_scale(subpel_params->xs, subpel_params->ys);
|
||||
assert(IMPLIES(is_intrabc, !is_scaled));
|
||||
if (is_scaled) {
|
||||
av1_highbd_convolve_2d_facade(
|
||||
src, src_stride, dst, dst_stride, w, h, interp_filters,
|
||||
subpel_params->subpel_x, subpel_params->xs, subpel_params->subpel_y,
|
||||
subpel_params->ys, 1, conv_params, sf, is_intrabc, bd);
|
||||
} else {
|
||||
SubpelParams sp = *subpel_params;
|
||||
revert_scale_extra_bits(&sp);
|
||||
av1_highbd_convolve_2d_facade(src, src_stride, dst, dst_stride, w, h,
|
||||
interp_filters, sp.subpel_x, sp.xs,
|
||||
sp.subpel_y, sp.ys, 0, conv_params, sf, bd);
|
||||
av1_highbd_convolve_2d_facade(
|
||||
src, src_stride, dst, dst_stride, w, h, interp_filters, sp.subpel_x,
|
||||
sp.xs, sp.subpel_y, sp.ys, 0, conv_params, sf, is_intrabc, bd);
|
||||
}
|
||||
}
|
||||
|
||||
|
|
@ -237,35 +245,6 @@ static INLINE MV clamp_mv_to_umv_border_sb(const MACROBLOCKD *xd,
|
|||
return clamped_mv;
|
||||
}
|
||||
|
||||
void av1_build_inter_predictors_sby(const AV1_COMMON *cm, MACROBLOCKD *xd,
|
||||
int mi_row, int mi_col, BUFFER_SET *ctx,
|
||||
BLOCK_SIZE bsize);
|
||||
|
||||
void av1_build_inter_predictors_sbuv(const AV1_COMMON *cm, MACROBLOCKD *xd,
|
||||
int mi_row, int mi_col, BUFFER_SET *ctx,
|
||||
BLOCK_SIZE bsize);
|
||||
|
||||
void av1_build_inter_predictors_sb(const AV1_COMMON *cm, MACROBLOCKD *xd,
|
||||
int mi_row, int mi_col, BUFFER_SET *ctx,
|
||||
BLOCK_SIZE bsize);
|
||||
|
||||
void av1_build_inter_predictor(const uint8_t *src, int src_stride, uint8_t *dst,
|
||||
int dst_stride, const MV *src_mv,
|
||||
const struct scale_factors *sf, int w, int h,
|
||||
ConvolveParams *conv_params,
|
||||
InterpFilters interp_filters,
|
||||
const WarpTypesAllowed *warp_types, int p_col,
|
||||
int p_row, int plane, int ref,
|
||||
enum mv_precision precision, int x, int y,
|
||||
const MACROBLOCKD *xd, int can_use_previous);
|
||||
|
||||
void av1_highbd_build_inter_predictor(
|
||||
const uint8_t *src, int src_stride, uint8_t *dst, int dst_stride,
|
||||
const MV *mv_q3, const struct scale_factors *sf, int w, int h, int do_avg,
|
||||
InterpFilters interp_filters, const WarpTypesAllowed *warp_types, int p_col,
|
||||
int p_row, int plane, enum mv_precision precision, int x, int y,
|
||||
const MACROBLOCKD *xd, int can_use_previous);
|
||||
|
||||
static INLINE int scaled_buffer_offset(int x_offset, int y_offset, int stride,
|
||||
const struct scale_factors *sf) {
|
||||
const int x =
|
||||
|
|
@ -303,32 +282,6 @@ void av1_setup_pre_planes(MACROBLOCKD *xd, int idx,
|
|||
const YV12_BUFFER_CONFIG *src, int mi_row, int mi_col,
|
||||
const struct scale_factors *sf, const int num_planes);
|
||||
|
||||
// Detect if the block have sub-pixel level motion vectors
|
||||
// per component.
|
||||
#define CHECK_SUBPEL 0
|
||||
static INLINE int has_subpel_mv_component(const MB_MODE_INFO *const mbmi,
|
||||
const MACROBLOCKD *const xd,
|
||||
int dir) {
|
||||
#if CHECK_SUBPEL
|
||||
const BLOCK_SIZE bsize = mbmi->sb_type;
|
||||
int plane;
|
||||
int ref = (dir >> 1);
|
||||
|
||||
if (dir & 0x01) {
|
||||
if (mbmi->mv[ref].as_mv.col & SUBPEL_MASK) return 1;
|
||||
} else {
|
||||
if (mbmi->mv[ref].as_mv.row & SUBPEL_MASK) return 1;
|
||||
}
|
||||
|
||||
return 0;
|
||||
#else
|
||||
(void)mbmi;
|
||||
(void)xd;
|
||||
(void)dir;
|
||||
return 1;
|
||||
#endif
|
||||
}
|
||||
|
||||
static INLINE void set_default_interp_filters(
|
||||
MB_MODE_INFO *const mbmi, InterpFilter frame_interp_filter) {
|
||||
mbmi->interp_filters =
|
||||
|
|
@ -343,21 +296,6 @@ static INLINE int av1_is_interp_needed(const MACROBLOCKD *const xd) {
|
|||
return 1;
|
||||
}
|
||||
|
||||
static INLINE int av1_is_interp_search_needed(const MACROBLOCKD *const xd) {
|
||||
MB_MODE_INFO *const mi = xd->mi[0];
|
||||
const int is_compound = has_second_ref(mi);
|
||||
int ref;
|
||||
for (ref = 0; ref < 1 + is_compound; ++ref) {
|
||||
int row_col;
|
||||
for (row_col = 0; row_col < 2; ++row_col) {
|
||||
const int dir = (ref << 1) + row_col;
|
||||
if (has_subpel_mv_component(mi, xd, dir)) {
|
||||
return 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
void av1_setup_build_prediction_by_above_pred(
|
||||
MACROBLOCKD *xd, int rel_mi_col, uint8_t above_mi_width,
|
||||
MB_MODE_INFO *above_mbmi, struct build_prediction_ctxt *ctxt,
|
||||
|
|
@ -367,18 +305,6 @@ void av1_setup_build_prediction_by_left_pred(MACROBLOCKD *xd, int rel_mi_row,
|
|||
MB_MODE_INFO *left_mbmi,
|
||||
struct build_prediction_ctxt *ctxt,
|
||||
const int num_planes);
|
||||
void av1_build_prediction_by_above_preds(const AV1_COMMON *cm, MACROBLOCKD *xd,
|
||||
int mi_row, int mi_col,
|
||||
uint8_t *tmp_buf[MAX_MB_PLANE],
|
||||
int tmp_width[MAX_MB_PLANE],
|
||||
int tmp_height[MAX_MB_PLANE],
|
||||
int tmp_stride[MAX_MB_PLANE]);
|
||||
void av1_build_prediction_by_left_preds(const AV1_COMMON *cm, MACROBLOCKD *xd,
|
||||
int mi_row, int mi_col,
|
||||
uint8_t *tmp_buf[MAX_MB_PLANE],
|
||||
int tmp_width[MAX_MB_PLANE],
|
||||
int tmp_height[MAX_MB_PLANE],
|
||||
int tmp_stride[MAX_MB_PLANE]);
|
||||
void av1_build_obmc_inter_prediction(const AV1_COMMON *cm, MACROBLOCKD *xd,
|
||||
int mi_row, int mi_col,
|
||||
uint8_t *above[MAX_MB_PLANE],
|
||||
|
|
@ -389,8 +315,6 @@ void av1_build_obmc_inter_prediction(const AV1_COMMON *cm, MACROBLOCKD *xd,
|
|||
const uint8_t *av1_get_obmc_mask(int length);
|
||||
void av1_count_overlappable_neighbors(const AV1_COMMON *cm, MACROBLOCKD *xd,
|
||||
int mi_row, int mi_col);
|
||||
void av1_build_obmc_inter_predictors_sb(const AV1_COMMON *cm, MACROBLOCKD *xd,
|
||||
int mi_row, int mi_col);
|
||||
|
||||
#define MASK_MASTER_SIZE ((MAX_WEDGE_SIZE) << 1)
|
||||
#define MASK_MASTER_STRIDE (MASK_MASTER_SIZE)
|
||||
|
|
@ -406,12 +330,6 @@ static INLINE const uint8_t *av1_get_contiguous_soft_mask(int wedge_index,
|
|||
const uint8_t *av1_get_compound_type_mask(
|
||||
const INTERINTER_COMPOUND_DATA *const comp_data, BLOCK_SIZE sb_type);
|
||||
|
||||
void av1_build_interintra_predictors(const AV1_COMMON *cm, MACROBLOCKD *xd,
|
||||
uint8_t *ypred, uint8_t *upred,
|
||||
uint8_t *vpred, int ystride, int ustride,
|
||||
int vstride, BUFFER_SET *ctx,
|
||||
BLOCK_SIZE bsize);
|
||||
|
||||
// build interintra_predictors for one plane
|
||||
void av1_build_interintra_predictors_sbp(const AV1_COMMON *cm, MACROBLOCKD *xd,
|
||||
uint8_t *pred, int stride,
|
||||
|
|
@ -431,18 +349,6 @@ void av1_combine_interintra(MACROBLOCKD *xd, BLOCK_SIZE bsize, int plane,
|
|||
const uint8_t *inter_pred, int inter_stride,
|
||||
const uint8_t *intra_pred, int intra_stride);
|
||||
|
||||
// Encoder only
|
||||
void av1_build_inter_predictors_for_planes_single_buf(
|
||||
MACROBLOCKD *xd, BLOCK_SIZE bsize, int plane_from, int plane_to, int mi_row,
|
||||
int mi_col, int ref, uint8_t *ext_dst[3], int ext_dst_stride[3],
|
||||
int can_use_previous);
|
||||
void av1_build_wedge_inter_predictor_from_buf(MACROBLOCKD *xd, BLOCK_SIZE bsize,
|
||||
int plane_from, int plane_to,
|
||||
uint8_t *ext_dst0[3],
|
||||
int ext_dst_stride0[3],
|
||||
uint8_t *ext_dst1[3],
|
||||
int ext_dst_stride1[3]);
|
||||
|
||||
void av1_jnt_comp_weight_assign(const AV1_COMMON *cm, const MB_MODE_INFO *mbmi,
|
||||
int order_idx, int *fwd_offset, int *bck_offset,
|
||||
int *use_jnt_comp_avg, int is_compound);
|
||||
|
|
@ -456,4 +362,4 @@ int av1_allow_warp(const MB_MODE_INFO *const mbmi,
|
|||
} // extern "C"
|
||||
#endif
|
||||
|
||||
#endif // AV1_COMMON_RECONINTER_H_
|
||||
#endif // AOM_AV1_COMMON_RECONINTER_H_
|
||||
|
|
|
|||
6
third_party/aom/av1/common/reconintra.h
vendored
6
third_party/aom/av1/common/reconintra.h
vendored
|
|
@ -9,8 +9,8 @@
|
|||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#ifndef AV1_COMMON_RECONINTRA_H_
|
||||
#define AV1_COMMON_RECONINTRA_H_
|
||||
#ifndef AOM_AV1_COMMON_RECONINTRA_H_
|
||||
#define AOM_AV1_COMMON_RECONINTRA_H_
|
||||
|
||||
#include <stdlib.h>
|
||||
|
||||
|
|
@ -116,4 +116,4 @@ static INLINE int av1_use_intra_edge_upsample(int bs0, int bs1, int delta,
|
|||
#ifdef __cplusplus
|
||||
} // extern "C"
|
||||
#endif
|
||||
#endif // AV1_COMMON_RECONINTRA_H_
|
||||
#endif // AOM_AV1_COMMON_RECONINTRA_H_
|
||||
|
|
|
|||
39
third_party/aom/av1/common/resize.c
vendored
39
third_party/aom/av1/common/resize.c
vendored
|
|
@ -170,42 +170,6 @@ static const InterpKernel filteredinterp_filters875[(1 << RS_SUBPEL_BITS)] = {
|
|||
{ -1, 3, -9, 17, 112, 10, -7, 3 }, { -1, 3, -8, 15, 112, 12, -7, 2 },
|
||||
};
|
||||
|
||||
// Filters for interpolation (full-band) - no filtering for integer pixels
|
||||
static const InterpKernel filteredinterp_filters1000[(1 << RS_SUBPEL_BITS)] = {
|
||||
{ 0, 0, 0, 128, 0, 0, 0, 0 }, { 0, 0, -1, 128, 2, -1, 0, 0 },
|
||||
{ 0, 1, -3, 127, 4, -2, 1, 0 }, { 0, 1, -4, 127, 6, -3, 1, 0 },
|
||||
{ 0, 2, -6, 126, 8, -3, 1, 0 }, { 0, 2, -7, 125, 11, -4, 1, 0 },
|
||||
{ -1, 2, -8, 125, 13, -5, 2, 0 }, { -1, 3, -9, 124, 15, -6, 2, 0 },
|
||||
{ -1, 3, -10, 123, 18, -6, 2, -1 }, { -1, 3, -11, 122, 20, -7, 3, -1 },
|
||||
{ -1, 4, -12, 121, 22, -8, 3, -1 }, { -1, 4, -13, 120, 25, -9, 3, -1 },
|
||||
{ -1, 4, -14, 118, 28, -9, 3, -1 }, { -1, 4, -15, 117, 30, -10, 4, -1 },
|
||||
{ -1, 5, -16, 116, 32, -11, 4, -1 }, { -1, 5, -16, 114, 35, -12, 4, -1 },
|
||||
{ -1, 5, -17, 112, 38, -12, 4, -1 }, { -1, 5, -18, 111, 40, -13, 5, -1 },
|
||||
{ -1, 5, -18, 109, 43, -14, 5, -1 }, { -1, 6, -19, 107, 45, -14, 5, -1 },
|
||||
{ -1, 6, -19, 105, 48, -15, 5, -1 }, { -1, 6, -19, 103, 51, -16, 5, -1 },
|
||||
{ -1, 6, -20, 101, 53, -16, 6, -1 }, { -1, 6, -20, 99, 56, -17, 6, -1 },
|
||||
{ -1, 6, -20, 97, 58, -17, 6, -1 }, { -1, 6, -20, 95, 61, -18, 6, -1 },
|
||||
{ -2, 7, -20, 93, 64, -18, 6, -2 }, { -2, 7, -20, 91, 66, -19, 6, -1 },
|
||||
{ -2, 7, -20, 88, 69, -19, 6, -1 }, { -2, 7, -20, 86, 71, -19, 6, -1 },
|
||||
{ -2, 7, -20, 84, 74, -20, 7, -2 }, { -2, 7, -20, 81, 76, -20, 7, -1 },
|
||||
{ -2, 7, -20, 79, 79, -20, 7, -2 }, { -1, 7, -20, 76, 81, -20, 7, -2 },
|
||||
{ -2, 7, -20, 74, 84, -20, 7, -2 }, { -1, 6, -19, 71, 86, -20, 7, -2 },
|
||||
{ -1, 6, -19, 69, 88, -20, 7, -2 }, { -1, 6, -19, 66, 91, -20, 7, -2 },
|
||||
{ -2, 6, -18, 64, 93, -20, 7, -2 }, { -1, 6, -18, 61, 95, -20, 6, -1 },
|
||||
{ -1, 6, -17, 58, 97, -20, 6, -1 }, { -1, 6, -17, 56, 99, -20, 6, -1 },
|
||||
{ -1, 6, -16, 53, 101, -20, 6, -1 }, { -1, 5, -16, 51, 103, -19, 6, -1 },
|
||||
{ -1, 5, -15, 48, 105, -19, 6, -1 }, { -1, 5, -14, 45, 107, -19, 6, -1 },
|
||||
{ -1, 5, -14, 43, 109, -18, 5, -1 }, { -1, 5, -13, 40, 111, -18, 5, -1 },
|
||||
{ -1, 4, -12, 38, 112, -17, 5, -1 }, { -1, 4, -12, 35, 114, -16, 5, -1 },
|
||||
{ -1, 4, -11, 32, 116, -16, 5, -1 }, { -1, 4, -10, 30, 117, -15, 4, -1 },
|
||||
{ -1, 3, -9, 28, 118, -14, 4, -1 }, { -1, 3, -9, 25, 120, -13, 4, -1 },
|
||||
{ -1, 3, -8, 22, 121, -12, 4, -1 }, { -1, 3, -7, 20, 122, -11, 3, -1 },
|
||||
{ -1, 2, -6, 18, 123, -10, 3, -1 }, { 0, 2, -6, 15, 124, -9, 3, -1 },
|
||||
{ 0, 2, -5, 13, 125, -8, 2, -1 }, { 0, 1, -4, 11, 125, -7, 2, 0 },
|
||||
{ 0, 1, -3, 8, 126, -6, 2, 0 }, { 0, 1, -3, 6, 127, -4, 1, 0 },
|
||||
{ 0, 1, -2, 4, 127, -3, 1, 0 }, { 0, 0, -1, 2, 128, -1, 0, 0 },
|
||||
};
|
||||
|
||||
const int16_t av1_resize_filter_normative[(
|
||||
1 << RS_SUBPEL_BITS)][UPSCALE_NORMATIVE_TAPS] = {
|
||||
#if UPSCALE_NORMATIVE_TAPS == 8
|
||||
|
|
@ -246,6 +210,9 @@ const int16_t av1_resize_filter_normative[(
|
|||
#endif // UPSCALE_NORMATIVE_TAPS == 8
|
||||
};
|
||||
|
||||
// Filters for interpolation (full-band) - no filtering for integer pixels
|
||||
#define filteredinterp_filters1000 av1_resize_filter_normative
|
||||
|
||||
// Filters for factor of 2 downsampling.
|
||||
static const int16_t av1_down2_symeven_half_filter[] = { 56, 12, -3, -1 };
|
||||
static const int16_t av1_down2_symodd_half_filter[] = { 64, 35, 0, -3 };
|
||||
|
|
|
|||
6
third_party/aom/av1/common/resize.h
vendored
6
third_party/aom/av1/common/resize.h
vendored
|
|
@ -9,8 +9,8 @@
|
|||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#ifndef AV1_ENCODER_RESIZE_H_
|
||||
#define AV1_ENCODER_RESIZE_H_
|
||||
#ifndef AOM_AV1_COMMON_RESIZE_H_
|
||||
#define AOM_AV1_COMMON_RESIZE_H_
|
||||
|
||||
#include <stdio.h>
|
||||
#include "aom/aom_integer.h"
|
||||
|
|
@ -109,4 +109,4 @@ int32_t av1_get_upscale_convolve_step(int in_length, int out_length);
|
|||
} // extern "C"
|
||||
#endif
|
||||
|
||||
#endif // AV1_ENCODER_RESIZE_H_
|
||||
#endif // AOM_AV1_COMMON_RESIZE_H_
|
||||
|
|
|
|||
131
third_party/aom/av1/common/restoration.c
vendored
131
third_party/aom/av1/common/restoration.c
vendored
|
|
@ -661,9 +661,10 @@ const int32_t one_by_x[MAX_NELEM] = {
|
|||
293, 273, 256, 241, 228, 216, 205, 195, 186, 178, 171, 164,
|
||||
};
|
||||
|
||||
static void selfguided_restoration_fast_internal(
|
||||
int32_t *dgd, int width, int height, int dgd_stride, int32_t *dst,
|
||||
int dst_stride, int bit_depth, int sgr_params_idx, int radius_idx) {
|
||||
static void calculate_intermediate_result(int32_t *dgd, int width, int height,
|
||||
int dgd_stride, int bit_depth,
|
||||
int sgr_params_idx, int radius_idx,
|
||||
int pass, int32_t *A, int32_t *B) {
|
||||
const sgr_params_type *const params = &sgr_params[sgr_params_idx];
|
||||
const int r = params->r[radius_idx];
|
||||
const int width_ext = width + 2 * SGRPROJ_BORDER_HORZ;
|
||||
|
|
@ -673,10 +674,7 @@ static void selfguided_restoration_fast_internal(
|
|||
// We also align the stride to a multiple of 16 bytes, for consistency
|
||||
// with the SIMD version of this function.
|
||||
int buf_stride = ((width_ext + 3) & ~3) + 16;
|
||||
int32_t A_[RESTORATION_PROC_UNIT_PELS];
|
||||
int32_t B_[RESTORATION_PROC_UNIT_PELS];
|
||||
int32_t *A = A_;
|
||||
int32_t *B = B_;
|
||||
const int step = pass == 0 ? 1 : 2;
|
||||
int i, j;
|
||||
|
||||
assert(r <= MAX_RADIUS && "Need MAX_RADIUS >= r");
|
||||
|
|
@ -691,7 +689,7 @@ static void selfguided_restoration_fast_internal(
|
|||
B += SGRPROJ_BORDER_VERT * buf_stride + SGRPROJ_BORDER_HORZ;
|
||||
// Calculate the eventual A[] and B[] arrays. Include a 1-pixel border - ie,
|
||||
// for a 64x64 processing unit, we calculate 66x66 pixels of A[] and B[].
|
||||
for (i = -1; i < height + 1; i += 2) {
|
||||
for (i = -1; i < height + 1; i += step) {
|
||||
for (j = -1; j < width + 1; ++j) {
|
||||
const int k = i * buf_stride + j;
|
||||
const int n = (2 * r + 1) * (2 * r + 1);
|
||||
|
|
@ -754,7 +752,31 @@ static void selfguided_restoration_fast_internal(
|
|||
SGRPROJ_RECIP_BITS);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static void selfguided_restoration_fast_internal(
|
||||
int32_t *dgd, int width, int height, int dgd_stride, int32_t *dst,
|
||||
int dst_stride, int bit_depth, int sgr_params_idx, int radius_idx) {
|
||||
const sgr_params_type *const params = &sgr_params[sgr_params_idx];
|
||||
const int r = params->r[radius_idx];
|
||||
const int width_ext = width + 2 * SGRPROJ_BORDER_HORZ;
|
||||
// Adjusting the stride of A and B here appears to avoid bad cache effects,
|
||||
// leading to a significant speed improvement.
|
||||
// We also align the stride to a multiple of 16 bytes, for consistency
|
||||
// with the SIMD version of this function.
|
||||
int buf_stride = ((width_ext + 3) & ~3) + 16;
|
||||
int32_t A_[RESTORATION_PROC_UNIT_PELS];
|
||||
int32_t B_[RESTORATION_PROC_UNIT_PELS];
|
||||
int32_t *A = A_;
|
||||
int32_t *B = B_;
|
||||
int i, j;
|
||||
calculate_intermediate_result(dgd, width, height, dgd_stride, bit_depth,
|
||||
sgr_params_idx, radius_idx, 1, A, B);
|
||||
A += SGRPROJ_BORDER_VERT * buf_stride + SGRPROJ_BORDER_HORZ;
|
||||
B += SGRPROJ_BORDER_VERT * buf_stride + SGRPROJ_BORDER_HORZ;
|
||||
|
||||
// Use the A[] and B[] arrays to calculate the filtered image
|
||||
(void)r;
|
||||
assert(r == 2);
|
||||
for (i = 0; i < height; ++i) {
|
||||
if (!(i & 1)) { // even row
|
||||
|
|
@ -796,10 +818,7 @@ static void selfguided_restoration_internal(int32_t *dgd, int width, int height,
|
|||
int dst_stride, int bit_depth,
|
||||
int sgr_params_idx,
|
||||
int radius_idx) {
|
||||
const sgr_params_type *const params = &sgr_params[sgr_params_idx];
|
||||
const int r = params->r[radius_idx];
|
||||
const int width_ext = width + 2 * SGRPROJ_BORDER_HORZ;
|
||||
const int height_ext = height + 2 * SGRPROJ_BORDER_VERT;
|
||||
// Adjusting the stride of A and B here appears to avoid bad cache effects,
|
||||
// leading to a significant speed improvement.
|
||||
// We also align the stride to a multiple of 16 bytes, for consistency
|
||||
|
|
@ -810,82 +829,11 @@ static void selfguided_restoration_internal(int32_t *dgd, int width, int height,
|
|||
int32_t *A = A_;
|
||||
int32_t *B = B_;
|
||||
int i, j;
|
||||
|
||||
assert(r <= MAX_RADIUS && "Need MAX_RADIUS >= r");
|
||||
assert(r <= SGRPROJ_BORDER_VERT - 1 && r <= SGRPROJ_BORDER_HORZ - 1 &&
|
||||
"Need SGRPROJ_BORDER_* >= r+1");
|
||||
|
||||
boxsum(dgd - dgd_stride * SGRPROJ_BORDER_VERT - SGRPROJ_BORDER_HORZ,
|
||||
width_ext, height_ext, dgd_stride, r, 0, B, buf_stride);
|
||||
boxsum(dgd - dgd_stride * SGRPROJ_BORDER_VERT - SGRPROJ_BORDER_HORZ,
|
||||
width_ext, height_ext, dgd_stride, r, 1, A, buf_stride);
|
||||
calculate_intermediate_result(dgd, width, height, dgd_stride, bit_depth,
|
||||
sgr_params_idx, radius_idx, 0, A, B);
|
||||
A += SGRPROJ_BORDER_VERT * buf_stride + SGRPROJ_BORDER_HORZ;
|
||||
B += SGRPROJ_BORDER_VERT * buf_stride + SGRPROJ_BORDER_HORZ;
|
||||
// Calculate the eventual A[] and B[] arrays. Include a 1-pixel border - ie,
|
||||
// for a 64x64 processing unit, we calculate 66x66 pixels of A[] and B[].
|
||||
for (i = -1; i < height + 1; ++i) {
|
||||
for (j = -1; j < width + 1; ++j) {
|
||||
const int k = i * buf_stride + j;
|
||||
const int n = (2 * r + 1) * (2 * r + 1);
|
||||
|
||||
// a < 2^16 * n < 2^22 regardless of bit depth
|
||||
uint32_t a = ROUND_POWER_OF_TWO(A[k], 2 * (bit_depth - 8));
|
||||
// b < 2^8 * n < 2^14 regardless of bit depth
|
||||
uint32_t b = ROUND_POWER_OF_TWO(B[k], bit_depth - 8);
|
||||
|
||||
// Each term in calculating p = a * n - b * b is < 2^16 * n^2 < 2^28,
|
||||
// and p itself satisfies p < 2^14 * n^2 < 2^26.
|
||||
// This bound on p is due to:
|
||||
// https://en.wikipedia.org/wiki/Popoviciu's_inequality_on_variances
|
||||
//
|
||||
// Note: Sometimes, in high bit depth, we can end up with a*n < b*b.
|
||||
// This is an artefact of rounding, and can only happen if all pixels
|
||||
// are (almost) identical, so in this case we saturate to p=0.
|
||||
uint32_t p = (a * n < b * b) ? 0 : a * n - b * b;
|
||||
|
||||
const uint32_t s = params->s[radius_idx];
|
||||
|
||||
// p * s < (2^14 * n^2) * round(2^20 / n^2 eps) < 2^34 / eps < 2^32
|
||||
// as long as eps >= 4. So p * s fits into a uint32_t, and z < 2^12
|
||||
// (this holds even after accounting for the rounding in s)
|
||||
const uint32_t z = ROUND_POWER_OF_TWO(p * s, SGRPROJ_MTABLE_BITS);
|
||||
|
||||
// Note: We have to be quite careful about the value of A[k].
|
||||
// This is used as a blend factor between individual pixel values and the
|
||||
// local mean. So it logically has a range of [0, 256], including both
|
||||
// endpoints.
|
||||
//
|
||||
// This is a pain for hardware, as we'd like something which can be stored
|
||||
// in exactly 8 bits.
|
||||
// Further, in the calculation of B[k] below, if z == 0 and r == 2,
|
||||
// then A[k] "should be" 0. But then we can end up setting B[k] to a value
|
||||
// slightly above 2^(8 + bit depth), due to rounding in the value of
|
||||
// one_by_x[25-1].
|
||||
//
|
||||
// Thus we saturate so that, when z == 0, A[k] is set to 1 instead of 0.
|
||||
// This fixes the above issues (256 - A[k] fits in a uint8, and we can't
|
||||
// overflow), without significantly affecting the final result: z == 0
|
||||
// implies that the image is essentially "flat", so the local mean and
|
||||
// individual pixel values are very similar.
|
||||
//
|
||||
// Note that saturating on the other side, ie. requring A[k] <= 255,
|
||||
// would be a bad idea, as that corresponds to the case where the image
|
||||
// is very variable, when we want to preserve the local pixel value as
|
||||
// much as possible.
|
||||
A[k] = x_by_xplus1[AOMMIN(z, 255)]; // in range [1, 256]
|
||||
|
||||
// SGRPROJ_SGR - A[k] < 2^8 (from above), B[k] < 2^(bit_depth) * n,
|
||||
// one_by_x[n - 1] = round(2^12 / n)
|
||||
// => the product here is < 2^(20 + bit_depth) <= 2^32,
|
||||
// and B[k] is set to a value < 2^(8 + bit depth)
|
||||
// This holds even with the rounding in one_by_x and in the overall
|
||||
// result, as long as SGRPROJ_SGR - A[k] is strictly less than 2^8.
|
||||
B[k] = (int32_t)ROUND_POWER_OF_TWO((uint32_t)(SGRPROJ_SGR - A[k]) *
|
||||
(uint32_t)B[k] *
|
||||
(uint32_t)one_by_x[n - 1],
|
||||
SGRPROJ_RECIP_BITS);
|
||||
}
|
||||
}
|
||||
// Use the A[] and B[] arrays to calculate the filtered image
|
||||
for (i = 0; i < height; ++i) {
|
||||
for (j = 0; j < width; ++j) {
|
||||
|
|
@ -911,10 +859,10 @@ static void selfguided_restoration_internal(int32_t *dgd, int width, int height,
|
|||
}
|
||||
}
|
||||
|
||||
void av1_selfguided_restoration_c(const uint8_t *dgd8, int width, int height,
|
||||
int dgd_stride, int32_t *flt0, int32_t *flt1,
|
||||
int flt_stride, int sgr_params_idx,
|
||||
int bit_depth, int highbd) {
|
||||
int av1_selfguided_restoration_c(const uint8_t *dgd8, int width, int height,
|
||||
int dgd_stride, int32_t *flt0, int32_t *flt1,
|
||||
int flt_stride, int sgr_params_idx,
|
||||
int bit_depth, int highbd) {
|
||||
int32_t dgd32_[RESTORATION_PROC_UNIT_PELS];
|
||||
const int dgd32_stride = width + 2 * SGRPROJ_BORDER_HORZ;
|
||||
int32_t *dgd32 =
|
||||
|
|
@ -948,6 +896,7 @@ void av1_selfguided_restoration_c(const uint8_t *dgd8, int width, int height,
|
|||
if (params->r[1] > 0)
|
||||
selfguided_restoration_internal(dgd32, width, height, dgd32_stride, flt1,
|
||||
flt_stride, bit_depth, sgr_params_idx, 1);
|
||||
return 0;
|
||||
}
|
||||
|
||||
void apply_selfguided_restoration_c(const uint8_t *dat8, int width, int height,
|
||||
|
|
@ -959,8 +908,10 @@ void apply_selfguided_restoration_c(const uint8_t *dat8, int width, int height,
|
|||
int32_t *flt1 = flt0 + RESTORATION_UNITPELS_MAX;
|
||||
assert(width * height <= RESTORATION_UNITPELS_MAX);
|
||||
|
||||
av1_selfguided_restoration_c(dat8, width, height, stride, flt0, flt1, width,
|
||||
eps, bit_depth, highbd);
|
||||
const int ret = av1_selfguided_restoration_c(
|
||||
dat8, width, height, stride, flt0, flt1, width, eps, bit_depth, highbd);
|
||||
(void)ret;
|
||||
assert(!ret);
|
||||
const sgr_params_type *const params = &sgr_params[eps];
|
||||
int xq[2];
|
||||
decode_xq(xqd, xq, params);
|
||||
|
|
|
|||
7
third_party/aom/av1/common/restoration.h
vendored
7
third_party/aom/av1/common/restoration.h
vendored
|
|
@ -9,8 +9,8 @@
|
|||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#ifndef AV1_COMMON_RESTORATION_H_
|
||||
#define AV1_COMMON_RESTORATION_H_
|
||||
#ifndef AOM_AV1_COMMON_RESTORATION_H_
|
||||
#define AOM_AV1_COMMON_RESTORATION_H_
|
||||
|
||||
#include "aom_ports/mem.h"
|
||||
#include "config/aom_config.h"
|
||||
|
|
@ -120,6 +120,7 @@ extern "C" {
|
|||
// If WIENER_WIN_CHROMA == WIENER_WIN - 2, that implies 5x5 filters are used for
|
||||
// chroma. To use 7x7 for chroma set WIENER_WIN_CHROMA to WIENER_WIN.
|
||||
#define WIENER_WIN_CHROMA (WIENER_WIN - 2)
|
||||
#define WIENER_WIN2_CHROMA ((WIENER_WIN_CHROMA) * (WIENER_WIN_CHROMA))
|
||||
|
||||
#define WIENER_FILT_PREC_BITS 7
|
||||
#define WIENER_FILT_STEP (1 << WIENER_FILT_PREC_BITS)
|
||||
|
|
@ -373,4 +374,4 @@ void av1_lr_sync_write_dummy(void *const lr_sync, int r, int c,
|
|||
} // extern "C"
|
||||
#endif
|
||||
|
||||
#endif // AV1_COMMON_RESTORATION_H_
|
||||
#endif // AOM_AV1_COMMON_RESTORATION_H_
|
||||
|
|
|
|||
7
third_party/aom/av1/common/scale.h
vendored
7
third_party/aom/av1/common/scale.h
vendored
|
|
@ -9,12 +9,11 @@
|
|||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#ifndef AV1_COMMON_SCALE_H_
|
||||
#define AV1_COMMON_SCALE_H_
|
||||
#ifndef AOM_AV1_COMMON_SCALE_H_
|
||||
#define AOM_AV1_COMMON_SCALE_H_
|
||||
|
||||
#include "av1/common/convolve.h"
|
||||
#include "av1/common/mv.h"
|
||||
#include "aom_dsp/aom_convolve.h"
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
|
|
@ -65,4 +64,4 @@ static INLINE int valid_ref_frame_size(int ref_width, int ref_height,
|
|||
} // extern "C"
|
||||
#endif
|
||||
|
||||
#endif // AV1_COMMON_SCALE_H_
|
||||
#endif // AOM_AV1_COMMON_SCALE_H_
|
||||
|
|
|
|||
6
third_party/aom/av1/common/scan.h
vendored
6
third_party/aom/av1/common/scan.h
vendored
|
|
@ -9,8 +9,8 @@
|
|||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#ifndef AV1_COMMON_SCAN_H_
|
||||
#define AV1_COMMON_SCAN_H_
|
||||
#ifndef AOM_AV1_COMMON_SCAN_H_
|
||||
#define AOM_AV1_COMMON_SCAN_H_
|
||||
|
||||
#include "aom/aom_integer.h"
|
||||
#include "aom_ports/mem.h"
|
||||
|
|
@ -52,4 +52,4 @@ static INLINE const SCAN_ORDER *get_scan(TX_SIZE tx_size, TX_TYPE tx_type) {
|
|||
} // extern "C"
|
||||
#endif
|
||||
|
||||
#endif // AV1_COMMON_SCAN_H_
|
||||
#endif // AOM_AV1_COMMON_SCAN_H_
|
||||
|
|
|
|||
6
third_party/aom/av1/common/seg_common.h
vendored
6
third_party/aom/av1/common/seg_common.h
vendored
|
|
@ -9,8 +9,8 @@
|
|||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#ifndef AV1_COMMON_SEG_COMMON_H_
|
||||
#define AV1_COMMON_SEG_COMMON_H_
|
||||
#ifndef AOM_AV1_COMMON_SEG_COMMON_H_
|
||||
#define AOM_AV1_COMMON_SEG_COMMON_H_
|
||||
|
||||
#include "aom_dsp/prob.h"
|
||||
|
||||
|
|
@ -101,4 +101,4 @@ static INLINE int get_segdata(const struct segmentation *seg, int segment_id,
|
|||
} // extern "C"
|
||||
#endif
|
||||
|
||||
#endif // AV1_COMMON_SEG_COMMON_H_
|
||||
#endif // AOM_AV1_COMMON_SEG_COMMON_H_
|
||||
|
|
|
|||
18
third_party/aom/av1/common/thread_common.c
vendored
18
third_party/aom/av1/common/thread_common.c
vendored
|
|
@ -304,8 +304,9 @@ static INLINE void thread_loop_filter_rows(
|
|||
}
|
||||
|
||||
// Row-based multi-threaded loopfilter hook
|
||||
static int loop_filter_row_worker(AV1LfSync *const lf_sync,
|
||||
LFWorkerData *const lf_data) {
|
||||
static int loop_filter_row_worker(void *arg1, void *arg2) {
|
||||
AV1LfSync *const lf_sync = (AV1LfSync *)arg1;
|
||||
LFWorkerData *const lf_data = (LFWorkerData *)arg2;
|
||||
thread_loop_filter_rows(lf_data->frame_buffer, lf_data->cm, lf_data->planes,
|
||||
lf_data->xd, lf_sync);
|
||||
return 1;
|
||||
|
|
@ -342,7 +343,7 @@ static void loop_filter_rows_mt(YV12_BUFFER_CONFIG *frame, AV1_COMMON *cm,
|
|||
AVxWorker *const worker = &workers[i];
|
||||
LFWorkerData *const lf_data = &lf_sync->lfdata[i];
|
||||
|
||||
worker->hook = (AVxWorkerHook)loop_filter_row_worker;
|
||||
worker->hook = loop_filter_row_worker;
|
||||
worker->data1 = lf_sync;
|
||||
worker->data2 = lf_data;
|
||||
|
||||
|
|
@ -649,8 +650,9 @@ AV1LrMTInfo *get_lr_job_info(AV1LrSync *lr_sync) {
|
|||
}
|
||||
|
||||
// Implement row loop restoration for each thread.
|
||||
static int loop_restoration_row_worker(AV1LrSync *const lr_sync,
|
||||
LRWorkerData *lrworkerdata) {
|
||||
static int loop_restoration_row_worker(void *arg1, void *arg2) {
|
||||
AV1LrSync *const lr_sync = (AV1LrSync *)arg1;
|
||||
LRWorkerData *lrworkerdata = (LRWorkerData *)arg2;
|
||||
AV1LrStruct *lr_ctxt = (AV1LrStruct *)lrworkerdata->lr_ctxt;
|
||||
FilterFrameCtxt *ctxt = lr_ctxt->ctxt;
|
||||
int lr_unit_row;
|
||||
|
|
@ -714,10 +716,12 @@ static void foreach_rest_unit_in_planes_mt(AV1LrStruct *lr_ctxt,
|
|||
int num_rows_lr = 0;
|
||||
|
||||
for (int plane = 0; plane < num_planes; plane++) {
|
||||
if (cm->rst_info[plane].frame_restoration_type == RESTORE_NONE) continue;
|
||||
|
||||
const AV1PixelRect tile_rect = ctxt[plane].tile_rect;
|
||||
const int max_tile_h = tile_rect.bottom - tile_rect.top;
|
||||
|
||||
const int unit_size = cm->seq_params.sb_size == BLOCK_128X128 ? 128 : 64;
|
||||
const int unit_size = cm->rst_info[plane].restoration_unit_size;
|
||||
|
||||
num_rows_lr =
|
||||
AOMMAX(num_rows_lr, av1_lr_count_units_in_tile(unit_size, max_tile_h));
|
||||
|
|
@ -746,7 +750,7 @@ static void foreach_rest_unit_in_planes_mt(AV1LrStruct *lr_ctxt,
|
|||
for (i = 0; i < num_workers; ++i) {
|
||||
AVxWorker *const worker = &workers[i];
|
||||
lr_sync->lrworkerdata[i].lr_ctxt = (void *)lr_ctxt;
|
||||
worker->hook = (AVxWorkerHook)loop_restoration_row_worker;
|
||||
worker->hook = loop_restoration_row_worker;
|
||||
worker->data1 = lr_sync;
|
||||
worker->data2 = &lr_sync->lrworkerdata[i];
|
||||
|
||||
|
|
|
|||
6
third_party/aom/av1/common/thread_common.h
vendored
6
third_party/aom/av1/common/thread_common.h
vendored
|
|
@ -9,8 +9,8 @@
|
|||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#ifndef AV1_COMMON_LOOPFILTER_THREAD_H_
|
||||
#define AV1_COMMON_LOOPFILTER_THREAD_H_
|
||||
#ifndef AOM_AV1_COMMON_THREAD_COMMON_H_
|
||||
#define AOM_AV1_COMMON_THREAD_COMMON_H_
|
||||
|
||||
#include "config/aom_config.h"
|
||||
|
||||
|
|
@ -116,4 +116,4 @@ void av1_loop_restoration_dealloc(AV1LrSync *lr_sync, int num_workers);
|
|||
} // extern "C"
|
||||
#endif
|
||||
|
||||
#endif // AV1_COMMON_LOOPFILTER_THREAD_H_
|
||||
#endif // AOM_AV1_COMMON_THREAD_COMMON_H_
|
||||
|
|
|
|||
16
third_party/aom/av1/common/tile_common.c
vendored
16
third_party/aom/av1/common/tile_common.c
vendored
|
|
@ -127,6 +127,22 @@ void av1_tile_set_col(TileInfo *tile, const AV1_COMMON *cm, int col) {
|
|||
assert(tile->mi_col_end > tile->mi_col_start);
|
||||
}
|
||||
|
||||
int av1_get_sb_rows_in_tile(AV1_COMMON *cm, TileInfo tile) {
|
||||
int mi_rows_aligned_to_sb = ALIGN_POWER_OF_TWO(
|
||||
tile.mi_row_end - tile.mi_row_start, cm->seq_params.mib_size_log2);
|
||||
int sb_rows = mi_rows_aligned_to_sb >> cm->seq_params.mib_size_log2;
|
||||
|
||||
return sb_rows;
|
||||
}
|
||||
|
||||
int av1_get_sb_cols_in_tile(AV1_COMMON *cm, TileInfo tile) {
|
||||
int mi_cols_aligned_to_sb = ALIGN_POWER_OF_TWO(
|
||||
tile.mi_col_end - tile.mi_col_start, cm->seq_params.mib_size_log2);
|
||||
int sb_cols = mi_cols_aligned_to_sb >> cm->seq_params.mib_size_log2;
|
||||
|
||||
return sb_cols;
|
||||
}
|
||||
|
||||
int get_tile_size(int mi_frame_size, int log2_tile_num, int *ntiles) {
|
||||
// Round the frame up to a whole number of max superblocks
|
||||
mi_frame_size = ALIGN_POWER_OF_TWO(mi_frame_size, MAX_MIB_SIZE_LOG2);
|
||||
|
|
|
|||
9
third_party/aom/av1/common/tile_common.h
vendored
9
third_party/aom/av1/common/tile_common.h
vendored
|
|
@ -9,8 +9,8 @@
|
|||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#ifndef AV1_COMMON_TILE_COMMON_H_
|
||||
#define AV1_COMMON_TILE_COMMON_H_
|
||||
#ifndef AOM_AV1_COMMON_TILE_COMMON_H_
|
||||
#define AOM_AV1_COMMON_TILE_COMMON_H_
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
|
|
@ -44,6 +44,9 @@ void av1_get_tile_n_bits(int mi_cols, int *min_log2_tile_cols,
|
|||
// tiles horizontally or vertically in the frame.
|
||||
int get_tile_size(int mi_frame_size, int log2_tile_num, int *ntiles);
|
||||
|
||||
int av1_get_sb_rows_in_tile(struct AV1Common *cm, TileInfo tile);
|
||||
int av1_get_sb_cols_in_tile(struct AV1Common *cm, TileInfo tile);
|
||||
|
||||
typedef struct {
|
||||
int left, top, right, bottom;
|
||||
} AV1PixelRect;
|
||||
|
|
@ -66,4 +69,4 @@ void av1_calculate_tile_rows(struct AV1Common *const cm);
|
|||
} // extern "C"
|
||||
#endif
|
||||
|
||||
#endif // AV1_COMMON_TILE_COMMON_H_
|
||||
#endif // AOM_AV1_COMMON_TILE_COMMON_H_
|
||||
|
|
|
|||
6
third_party/aom/av1/common/timing.h
vendored
6
third_party/aom/av1/common/timing.h
vendored
|
|
@ -9,8 +9,8 @@
|
|||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#ifndef AOM_TIMING_H_
|
||||
#define AOM_TIMING_H_
|
||||
#ifndef AOM_AV1_COMMON_TIMING_H_
|
||||
#define AOM_AV1_COMMON_TIMING_H_
|
||||
|
||||
#include "aom/aom_integer.h"
|
||||
#include "av1/common/enums.h"
|
||||
|
|
@ -56,4 +56,4 @@ void set_resource_availability_parameters(
|
|||
int64_t max_level_bitrate(BITSTREAM_PROFILE seq_profile, int seq_level_idx,
|
||||
int seq_tier);
|
||||
|
||||
#endif // AOM_TIMING_H_
|
||||
#endif // AOM_AV1_COMMON_TIMING_H_
|
||||
|
|
|
|||
5
third_party/aom/av1/common/token_cdfs.h
vendored
5
third_party/aom/av1/common/token_cdfs.h
vendored
|
|
@ -9,6 +9,9 @@
|
|||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#ifndef AOM_AV1_COMMON_TOKEN_CDFS_H_
|
||||
#define AOM_AV1_COMMON_TOKEN_CDFS_H_
|
||||
|
||||
#include "config/aom_config.h"
|
||||
|
||||
#include "av1/common/entropy.h"
|
||||
|
|
@ -3548,3 +3551,5 @@ static const aom_cdf_prob av1_default_coeff_base_eob_multi_cdfs
|
|||
{ AOM_CDF3(10923, 21845) },
|
||||
{ AOM_CDF3(10923, 21845) },
|
||||
{ AOM_CDF3(10923, 21845) } } } } };
|
||||
|
||||
#endif // AOM_AV1_COMMON_TOKEN_CDFS_H_
|
||||
|
|
|
|||
243
third_party/aom/av1/common/txb_common.h
vendored
243
third_party/aom/av1/common/txb_common.h
vendored
|
|
@ -9,8 +9,8 @@
|
|||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#ifndef AV1_COMMON_TXB_COMMON_H_
|
||||
#define AV1_COMMON_TXB_COMMON_H_
|
||||
#ifndef AOM_AV1_COMMON_TXB_COMMON_H_
|
||||
#define AOM_AV1_COMMON_TXB_COMMON_H_
|
||||
|
||||
extern const int16_t k_eob_group_start[12];
|
||||
extern const int16_t k_eob_offset_bits[12];
|
||||
|
|
@ -34,24 +34,6 @@ static const int base_level_count_to_index[13] = {
|
|||
0, 0, 1, 1, 1, 2, 2, 2, 2, 3, 3, 3, 3,
|
||||
};
|
||||
|
||||
// Note: TX_PAD_2D is dependent to this offset table.
|
||||
static const int base_ref_offset[BASE_CONTEXT_POSITION_NUM][2] = {
|
||||
/* clang-format off*/
|
||||
{ -2, 0 }, { -1, -1 }, { -1, 0 }, { -1, 1 }, { 0, -2 }, { 0, -1 }, { 0, 1 },
|
||||
{ 0, 2 }, { 1, -1 }, { 1, 0 }, { 1, 1 }, { 2, 0 }
|
||||
/* clang-format on*/
|
||||
};
|
||||
|
||||
#define CONTEXT_MAG_POSITION_NUM 3
|
||||
static const int mag_ref_offset_with_txclass[3][CONTEXT_MAG_POSITION_NUM][2] = {
|
||||
{ { 0, 1 }, { 1, 0 }, { 1, 1 } },
|
||||
{ { 0, 1 }, { 1, 0 }, { 0, 2 } },
|
||||
{ { 0, 1 }, { 1, 0 }, { 2, 0 } }
|
||||
};
|
||||
static const int mag_ref_offset[CONTEXT_MAG_POSITION_NUM][2] = {
|
||||
{ 0, 1 }, { 1, 0 }, { 1, 1 }
|
||||
};
|
||||
|
||||
static const TX_CLASS tx_type_to_class[TX_TYPES] = {
|
||||
TX_CLASS_2D, // DCT_DCT
|
||||
TX_CLASS_2D, // ADST_DCT
|
||||
|
|
@ -71,61 +53,6 @@ static const TX_CLASS tx_type_to_class[TX_TYPES] = {
|
|||
TX_CLASS_HORIZ, // H_FLIPADST
|
||||
};
|
||||
|
||||
static const int8_t eob_to_pos_small[33] = {
|
||||
0, 1, 2, // 0-2
|
||||
3, 3, // 3-4
|
||||
4, 4, 4, 4, // 5-8
|
||||
5, 5, 5, 5, 5, 5, 5, 5, // 9-16
|
||||
6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6 // 17-32
|
||||
};
|
||||
|
||||
static const int8_t eob_to_pos_large[17] = {
|
||||
6, // place holder
|
||||
7, // 33-64
|
||||
8, 8, // 65-128
|
||||
9, 9, 9, 9, // 129-256
|
||||
10, 10, 10, 10, 10, 10, 10, 10, // 257-512
|
||||
11 // 513-
|
||||
};
|
||||
|
||||
static INLINE int get_eob_pos_token(const int eob, int *const extra) {
|
||||
int t;
|
||||
|
||||
if (eob < 33) {
|
||||
t = eob_to_pos_small[eob];
|
||||
} else {
|
||||
const int e = AOMMIN((eob - 1) >> 5, 16);
|
||||
t = eob_to_pos_large[e];
|
||||
}
|
||||
|
||||
*extra = eob - k_eob_group_start[t];
|
||||
|
||||
return t;
|
||||
}
|
||||
|
||||
static INLINE int av1_get_eob_pos_ctx(const TX_TYPE tx_type,
|
||||
const int eob_token) {
|
||||
static const int8_t tx_type_to_offset[TX_TYPES] = {
|
||||
-1, // DCT_DCT
|
||||
-1, // ADST_DCT
|
||||
-1, // DCT_ADST
|
||||
-1, // ADST_ADST
|
||||
-1, // FLIPADST_DCT
|
||||
-1, // DCT_FLIPADST
|
||||
-1, // FLIPADST_FLIPADST
|
||||
-1, // ADST_FLIPADST
|
||||
-1, // FLIPADST_ADST
|
||||
-1, // IDTX
|
||||
10, // V_DCT
|
||||
10, // H_DCT
|
||||
10, // V_ADST
|
||||
10, // H_ADST
|
||||
10, // V_FLIPADST
|
||||
10, // H_FLIPADST
|
||||
};
|
||||
return eob_token + tx_type_to_offset[tx_type];
|
||||
}
|
||||
|
||||
static INLINE int get_txb_bwl(TX_SIZE tx_size) {
|
||||
tx_size = av1_get_adjusted_tx_size(tx_size);
|
||||
return tx_size_wide_log2[tx_size];
|
||||
|
|
@ -141,36 +68,6 @@ static INLINE int get_txb_high(TX_SIZE tx_size) {
|
|||
return tx_size_high[tx_size];
|
||||
}
|
||||
|
||||
static INLINE void get_base_count_mag(int *mag, int *count,
|
||||
const tran_low_t *tcoeffs, int bwl,
|
||||
int height, int row, int col) {
|
||||
mag[0] = 0;
|
||||
mag[1] = 0;
|
||||
for (int i = 0; i < NUM_BASE_LEVELS; ++i) count[i] = 0;
|
||||
for (int idx = 0; idx < BASE_CONTEXT_POSITION_NUM; ++idx) {
|
||||
const int ref_row = row + base_ref_offset[idx][0];
|
||||
const int ref_col = col + base_ref_offset[idx][1];
|
||||
if (ref_row < 0 || ref_col < 0 || ref_row >= height ||
|
||||
ref_col >= (1 << bwl))
|
||||
continue;
|
||||
const int pos = (ref_row << bwl) + ref_col;
|
||||
tran_low_t abs_coeff = abs(tcoeffs[pos]);
|
||||
// count
|
||||
for (int i = 0; i < NUM_BASE_LEVELS; ++i) {
|
||||
count[i] += abs_coeff > i;
|
||||
}
|
||||
// mag
|
||||
if (base_ref_offset[idx][0] >= 0 && base_ref_offset[idx][1] >= 0) {
|
||||
if (abs_coeff > mag[0]) {
|
||||
mag[0] = abs_coeff;
|
||||
mag[1] = 1;
|
||||
} else if (abs_coeff == mag[0]) {
|
||||
++mag[1];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static INLINE uint8_t *set_levels(uint8_t *const levels_buf, const int width) {
|
||||
return levels_buf + TX_PAD_TOP * (width + TX_PAD_HOR);
|
||||
}
|
||||
|
|
@ -179,30 +76,6 @@ static INLINE int get_padded_idx(const int idx, const int bwl) {
|
|||
return idx + ((idx >> bwl) << TX_PAD_HOR_LOG2);
|
||||
}
|
||||
|
||||
static INLINE int get_level_count(const uint8_t *const levels, const int stride,
|
||||
const int row, const int col, const int level,
|
||||
const int (*nb_offset)[2], const int nb_num) {
|
||||
int count = 0;
|
||||
|
||||
for (int idx = 0; idx < nb_num; ++idx) {
|
||||
const int ref_row = row + nb_offset[idx][0];
|
||||
const int ref_col = col + nb_offset[idx][1];
|
||||
const int pos = ref_row * stride + ref_col;
|
||||
count += levels[pos] > level;
|
||||
}
|
||||
return count;
|
||||
}
|
||||
|
||||
static INLINE void get_level_mag(const uint8_t *const levels, const int stride,
|
||||
const int row, const int col, int *const mag) {
|
||||
for (int idx = 0; idx < CONTEXT_MAG_POSITION_NUM; ++idx) {
|
||||
const int ref_row = row + mag_ref_offset[idx][0];
|
||||
const int ref_col = col + mag_ref_offset[idx][1];
|
||||
const int pos = ref_row * stride + ref_col;
|
||||
mag[idx] = levels[pos];
|
||||
}
|
||||
}
|
||||
|
||||
static INLINE int get_base_ctx_from_count_mag(int row, int col, int count,
|
||||
int sig_mag) {
|
||||
const int ctx = base_level_count_to_index[count];
|
||||
|
|
@ -267,84 +140,6 @@ static INLINE int get_base_ctx_from_count_mag(int row, int col, int count,
|
|||
return ctx_idx;
|
||||
}
|
||||
|
||||
static INLINE int get_base_ctx(const uint8_t *const levels,
|
||||
const int c, // raster order
|
||||
const int bwl, const int level_minus_1,
|
||||
const int count) {
|
||||
const int row = c >> bwl;
|
||||
const int col = c - (row << bwl);
|
||||
const int stride = (1 << bwl) + TX_PAD_HOR;
|
||||
int mag_count = 0;
|
||||
int nb_mag[3] = { 0 };
|
||||
|
||||
get_level_mag(levels, stride, row, col, nb_mag);
|
||||
|
||||
for (int idx = 0; idx < 3; ++idx)
|
||||
mag_count += nb_mag[idx] > (level_minus_1 + 1);
|
||||
const int ctx_idx =
|
||||
get_base_ctx_from_count_mag(row, col, count, AOMMIN(2, mag_count));
|
||||
return ctx_idx;
|
||||
}
|
||||
|
||||
#define BR_CONTEXT_POSITION_NUM 8 // Base range coefficient context
|
||||
// Note: TX_PAD_2D is dependent to this offset table.
|
||||
static const int br_ref_offset[BR_CONTEXT_POSITION_NUM][2] = {
|
||||
/* clang-format off*/
|
||||
{ -1, -1 }, { -1, 0 }, { -1, 1 }, { 0, -1 },
|
||||
{ 0, 1 }, { 1, -1 }, { 1, 0 }, { 1, 1 },
|
||||
/* clang-format on*/
|
||||
};
|
||||
|
||||
static const int br_level_map[9] = {
|
||||
0, 0, 1, 1, 2, 2, 3, 3, 3,
|
||||
};
|
||||
|
||||
// Note: If BR_MAG_OFFSET changes, the calculation of offset in
|
||||
// get_br_ctx_from_count_mag() must be updated.
|
||||
#define BR_MAG_OFFSET 1
|
||||
// TODO(angiebird): optimize this function by using a table to map from
|
||||
// count/mag to ctx
|
||||
|
||||
static INLINE int get_br_count_mag(int *mag, const tran_low_t *tcoeffs, int bwl,
|
||||
int height, int row, int col, int level) {
|
||||
mag[0] = 0;
|
||||
mag[1] = 0;
|
||||
int count = 0;
|
||||
for (int idx = 0; idx < BR_CONTEXT_POSITION_NUM; ++idx) {
|
||||
const int ref_row = row + br_ref_offset[idx][0];
|
||||
const int ref_col = col + br_ref_offset[idx][1];
|
||||
if (ref_row < 0 || ref_col < 0 || ref_row >= height ||
|
||||
ref_col >= (1 << bwl))
|
||||
continue;
|
||||
const int pos = (ref_row << bwl) + ref_col;
|
||||
tran_low_t abs_coeff = abs(tcoeffs[pos]);
|
||||
count += abs_coeff > level;
|
||||
if (br_ref_offset[idx][0] >= 0 && br_ref_offset[idx][1] >= 0) {
|
||||
if (abs_coeff > mag[0]) {
|
||||
mag[0] = abs_coeff;
|
||||
mag[1] = 1;
|
||||
} else if (abs_coeff == mag[0]) {
|
||||
++mag[1];
|
||||
}
|
||||
}
|
||||
}
|
||||
return count;
|
||||
}
|
||||
|
||||
static INLINE int get_br_ctx_from_count_mag(const int row, const int col,
|
||||
const int count, const int mag) {
|
||||
// DC: 0 - 1
|
||||
// Top row: 2 - 4
|
||||
// Left column: 5 - 7
|
||||
// others: 8 - 11
|
||||
static const int offset_pos[2][2] = { { 8, 5 }, { 2, 0 } };
|
||||
const int mag_clamp = AOMMIN(mag, 6);
|
||||
const int offset = mag_clamp >> 1;
|
||||
const int ctx =
|
||||
br_level_map[count] + offset * BR_TMP_OFFSET + offset_pos[!row][!col];
|
||||
return ctx;
|
||||
}
|
||||
|
||||
static INLINE int get_br_ctx_2d(const uint8_t *const levels,
|
||||
const int c, // raster order
|
||||
const int bwl) {
|
||||
|
|
@ -396,38 +191,6 @@ static AOM_FORCE_INLINE int get_br_ctx(const uint8_t *const levels,
|
|||
return mag + 14;
|
||||
}
|
||||
|
||||
#define SIG_REF_OFFSET_NUM 5
|
||||
|
||||
// Note: TX_PAD_2D is dependent to these offset tables.
|
||||
static const int sig_ref_offset[SIG_REF_OFFSET_NUM][2] = {
|
||||
{ 0, 1 }, { 1, 0 }, { 1, 1 }, { 0, 2 }, { 2, 0 }
|
||||
// , { 1, 2 }, { 2, 1 },
|
||||
};
|
||||
|
||||
static const int sig_ref_offset_vert[SIG_REF_OFFSET_NUM][2] = {
|
||||
{ 1, 0 }, { 2, 0 }, { 0, 1 }, { 3, 0 }, { 4, 0 }
|
||||
// , { 1, 1 }, { 2, 1 },
|
||||
};
|
||||
|
||||
static const int sig_ref_offset_horiz[SIG_REF_OFFSET_NUM][2] = {
|
||||
{ 0, 1 }, { 0, 2 }, { 1, 0 }, { 0, 3 }, { 0, 4 }
|
||||
// , { 1, 1 }, { 1, 2 },
|
||||
};
|
||||
|
||||
#define SIG_REF_DIFF_OFFSET_NUM 3
|
||||
|
||||
static const int sig_ref_diff_offset[SIG_REF_DIFF_OFFSET_NUM][2] = {
|
||||
{ 1, 1 }, { 0, 2 }, { 2, 0 }
|
||||
};
|
||||
|
||||
static const int sig_ref_diff_offset_vert[SIG_REF_DIFF_OFFSET_NUM][2] = {
|
||||
{ 2, 0 }, { 3, 0 }, { 4, 0 }
|
||||
};
|
||||
|
||||
static const int sig_ref_diff_offset_horiz[SIG_REF_DIFF_OFFSET_NUM][2] = {
|
||||
{ 0, 2 }, { 0, 3 }, { 0, 4 }
|
||||
};
|
||||
|
||||
static const uint8_t clip_max3[256] = {
|
||||
0, 1, 2, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3,
|
||||
3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3,
|
||||
|
|
@ -658,4 +421,4 @@ static INLINE void get_txb_ctx(const BLOCK_SIZE plane_bsize,
|
|||
|
||||
void av1_init_lv_map(AV1_COMMON *cm);
|
||||
|
||||
#endif // AV1_COMMON_TXB_COMMON_H_
|
||||
#endif // AOM_AV1_COMMON_TXB_COMMON_H_
|
||||
|
|
|
|||
4
third_party/aom/av1/common/warped_motion.c
vendored
4
third_party/aom/av1/common/warped_motion.c
vendored
|
|
@ -562,7 +562,7 @@ static int64_t highbd_warp_error(
|
|||
const int error_bsize_h = AOMMIN(p_height, WARP_ERROR_BLOCK);
|
||||
uint16_t tmp[WARP_ERROR_BLOCK * WARP_ERROR_BLOCK];
|
||||
|
||||
ConvolveParams conv_params = get_conv_params(0, 0, 0, bd);
|
||||
ConvolveParams conv_params = get_conv_params(0, 0, bd);
|
||||
conv_params.use_jnt_comp_avg = 0;
|
||||
for (int i = p_row; i < p_row + p_height; i += WARP_ERROR_BLOCK) {
|
||||
for (int j = p_col; j < p_col + p_width; j += WARP_ERROR_BLOCK) {
|
||||
|
|
@ -845,7 +845,7 @@ static int64_t warp_error(WarpedMotionParams *wm, const uint8_t *const ref,
|
|||
int error_bsize_w = AOMMIN(p_width, WARP_ERROR_BLOCK);
|
||||
int error_bsize_h = AOMMIN(p_height, WARP_ERROR_BLOCK);
|
||||
uint8_t tmp[WARP_ERROR_BLOCK * WARP_ERROR_BLOCK];
|
||||
ConvolveParams conv_params = get_conv_params(0, 0, 0, 8);
|
||||
ConvolveParams conv_params = get_conv_params(0, 0, 8);
|
||||
conv_params.use_jnt_comp_avg = 0;
|
||||
|
||||
for (int i = p_row; i < p_row + p_height; i += WARP_ERROR_BLOCK) {
|
||||
|
|
|
|||
6
third_party/aom/av1/common/warped_motion.h
vendored
6
third_party/aom/av1/common/warped_motion.h
vendored
|
|
@ -9,8 +9,8 @@
|
|||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#ifndef AV1_COMMON_WARPED_MOTION_H_
|
||||
#define AV1_COMMON_WARPED_MOTION_H_
|
||||
#ifndef AOM_AV1_COMMON_WARPED_MOTION_H_
|
||||
#define AOM_AV1_COMMON_WARPED_MOTION_H_
|
||||
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
|
|
@ -92,4 +92,4 @@ int find_projection(int np, int *pts1, int *pts2, BLOCK_SIZE bsize, int mvy,
|
|||
int mi_col);
|
||||
|
||||
int get_shear_params(WarpedMotionParams *wm);
|
||||
#endif // AV1_COMMON_WARPED_MOTION_H_
|
||||
#endif // AOM_AV1_COMMON_WARPED_MOTION_H_
|
||||
|
|
|
|||
|
|
@ -14,7 +14,6 @@
|
|||
|
||||
#include "config/aom_dsp_rtcd.h"
|
||||
|
||||
#include "aom_dsp/aom_convolve.h"
|
||||
#include "aom_dsp/aom_dsp_common.h"
|
||||
#include "aom_dsp/aom_filter.h"
|
||||
#include "av1/common/convolve.h"
|
||||
|
|
|
|||
|
|
@ -18,6 +18,12 @@
|
|||
#include "av1/common/x86/av1_inv_txfm_avx2.h"
|
||||
#include "av1/common/x86/av1_inv_txfm_ssse3.h"
|
||||
|
||||
// TODO(venkatsanampudi@ittiam.com): move this to header file
|
||||
|
||||
// Sqrt2, Sqrt2^2, Sqrt2^3, Sqrt2^4, Sqrt2^5
|
||||
static int32_t NewSqrt2list[TX_SIZES] = { 5793, 2 * 4096, 2 * 5793, 4 * 4096,
|
||||
4 * 5793 };
|
||||
|
||||
static INLINE void idct16_stage5_avx2(__m256i *x1, const int32_t *cospi,
|
||||
const __m256i _r, int8_t cos_bit) {
|
||||
const __m256i cospi_m32_p32 = pair_set_w16_epi16(-cospi[32], cospi[32]);
|
||||
|
|
|
|||
|
|
@ -8,8 +8,8 @@
|
|||
* Media Patent License 1.0 was not distributed with this source code in the
|
||||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
#ifndef AV1_COMMON_X86_AV1_INV_TXFM_AVX2_H_
|
||||
#define AV1_COMMON_X86_AV1_INV_TXFM_AVX2_H_
|
||||
#ifndef AOM_AV1_COMMON_X86_AV1_INV_TXFM_AVX2_H_
|
||||
#define AOM_AV1_COMMON_X86_AV1_INV_TXFM_AVX2_H_
|
||||
|
||||
#include <immintrin.h>
|
||||
|
||||
|
|
@ -68,4 +68,4 @@ void av1_lowbd_inv_txfm2d_add_avx2(const int32_t *input, uint8_t *output,
|
|||
}
|
||||
#endif
|
||||
|
||||
#endif // AV1_COMMON_X86_AV1_INV_TXFM_AVX2_H_
|
||||
#endif // AOM_AV1_COMMON_X86_AV1_INV_TXFM_AVX2_H_
|
||||
|
|
|
|||
|
|
@ -16,6 +16,12 @@
|
|||
#include "av1/common/x86/av1_inv_txfm_ssse3.h"
|
||||
#include "av1/common/x86/av1_txfm_sse2.h"
|
||||
|
||||
// TODO(venkatsanampudi@ittiam.com): move this to header file
|
||||
|
||||
// Sqrt2, Sqrt2^2, Sqrt2^3, Sqrt2^4, Sqrt2^5
|
||||
static int32_t NewSqrt2list[TX_SIZES] = { 5793, 2 * 4096, 2 * 5793, 4 * 4096,
|
||||
4 * 5793 };
|
||||
|
||||
// TODO(binpengsmail@gmail.com): replace some for loop with do {} while
|
||||
|
||||
static void idct4_new_sse2(const __m128i *input, __m128i *output,
|
||||
|
|
|
|||
|
|
@ -8,8 +8,8 @@
|
|||
* Media Patent License 1.0 was not distributed with this source code in the
|
||||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
#ifndef AV1_COMMON_X86_AV1_INV_TXFM_SSSE3_H_
|
||||
#define AV1_COMMON_X86_AV1_INV_TXFM_SSSE3_H_
|
||||
#ifndef AOM_AV1_COMMON_X86_AV1_INV_TXFM_SSSE3_H_
|
||||
#define AOM_AV1_COMMON_X86_AV1_INV_TXFM_SSSE3_H_
|
||||
|
||||
#include <emmintrin.h> // SSE2
|
||||
#include <tmmintrin.h> // SSSE3
|
||||
|
|
@ -94,10 +94,6 @@ static const ITX_TYPE_1D hitx_1d_tab[TX_TYPES] = {
|
|||
IIDENTITY_1D, IADST_1D, IIDENTITY_1D, IFLIPADST_1D,
|
||||
};
|
||||
|
||||
// Sqrt2, Sqrt2^2, Sqrt2^3, Sqrt2^4, Sqrt2^5
|
||||
static int32_t NewSqrt2list[TX_SIZES] = { 5793, 2 * 4096, 2 * 5793, 4 * 4096,
|
||||
4 * 5793 };
|
||||
|
||||
DECLARE_ALIGNED(16, static const int16_t, av1_eob_to_eobxy_8x8_default[8]) = {
|
||||
0x0707, 0x0707, 0x0707, 0x0707, 0x0707, 0x0707, 0x0707, 0x0707,
|
||||
};
|
||||
|
|
@ -233,4 +229,4 @@ void av1_lowbd_inv_txfm2d_add_ssse3(const int32_t *input, uint8_t *output,
|
|||
} // extern "C"
|
||||
#endif
|
||||
|
||||
#endif // AV1_COMMON_X86_AV1_INV_TXFM_SSSE3_H_
|
||||
#endif // AOM_AV1_COMMON_X86_AV1_INV_TXFM_SSSE3_H_
|
||||
|
|
|
|||
|
|
@ -8,8 +8,8 @@
|
|||
* Media Patent License 1.0 was not distributed with this source code in the
|
||||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
#ifndef AV1_COMMON_X86_AV1_TXFM_SSE2_H_
|
||||
#define AV1_COMMON_X86_AV1_TXFM_SSE2_H_
|
||||
#ifndef AOM_AV1_COMMON_X86_AV1_TXFM_SSE2_H_
|
||||
#define AOM_AV1_COMMON_X86_AV1_TXFM_SSE2_H_
|
||||
|
||||
#include <emmintrin.h> // SSE2
|
||||
|
||||
|
|
@ -314,4 +314,4 @@ typedef struct {
|
|||
#ifdef __cplusplus
|
||||
}
|
||||
#endif // __cplusplus
|
||||
#endif // AV1_COMMON_X86_AV1_TXFM_SSE2_H_
|
||||
#endif // AOM_AV1_COMMON_X86_AV1_TXFM_SSE2_H_
|
||||
|
|
|
|||
11
third_party/aom/av1/common/x86/av1_txfm_sse4.h
vendored
11
third_party/aom/av1/common/x86/av1_txfm_sse4.h
vendored
|
|
@ -9,8 +9,8 @@
|
|||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#ifndef AV1_TXFM_SSE4_H_
|
||||
#define AV1_TXFM_SSE4_H_
|
||||
#ifndef AOM_AV1_COMMON_X86_AV1_TXFM_SSE4_H_
|
||||
#define AOM_AV1_COMMON_X86_AV1_TXFM_SSE4_H_
|
||||
|
||||
#include <smmintrin.h>
|
||||
|
||||
|
|
@ -45,8 +45,9 @@ static INLINE void av1_round_shift_array_32_sse4_1(__m128i *input,
|
|||
static INLINE void av1_round_shift_rect_array_32_sse4_1(__m128i *input,
|
||||
__m128i *output,
|
||||
const int size,
|
||||
const int bit) {
|
||||
const __m128i sqrt2 = _mm_set1_epi32(NewSqrt2);
|
||||
const int bit,
|
||||
const int val) {
|
||||
const __m128i sqrt2 = _mm_set1_epi32(val);
|
||||
if (bit > 0) {
|
||||
int i;
|
||||
for (i = 0; i < size; i++) {
|
||||
|
|
@ -68,4 +69,4 @@ static INLINE void av1_round_shift_rect_array_32_sse4_1(__m128i *input,
|
|||
}
|
||||
#endif
|
||||
|
||||
#endif // AV1_TXFM_SSE4_H_
|
||||
#endif // AOM_AV1_COMMON_X86_AV1_TXFM_SSE4_H_
|
||||
|
|
|
|||
5
third_party/aom/av1/common/x86/cfl_simd.h
vendored
5
third_party/aom/av1/common/x86/cfl_simd.h
vendored
|
|
@ -9,6 +9,9 @@
|
|||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#ifndef AOM_AV1_COMMON_X86_CFL_SIMD_H_
|
||||
#define AOM_AV1_COMMON_X86_CFL_SIMD_H_
|
||||
|
||||
#include "av1/common/blockd.h"
|
||||
|
||||
// SSSE3 version is optimal for with == 4, we reuse them in AVX2
|
||||
|
|
@ -236,3 +239,5 @@ void predict_hbd_16x16_ssse3(const int16_t *pred_buf_q3, uint16_t *dst,
|
|||
int dst_stride, int alpha_q3, int bd);
|
||||
void predict_hbd_16x32_ssse3(const int16_t *pred_buf_q3, uint16_t *dst,
|
||||
int dst_stride, int alpha_q3, int bd);
|
||||
|
||||
#endif // AOM_AV1_COMMON_X86_CFL_SIMD_H_
|
||||
|
|
|
|||
|
|
@ -11,10 +11,8 @@
|
|||
|
||||
#include <immintrin.h>
|
||||
|
||||
#include "config/aom_dsp_rtcd.h"
|
||||
#include "config/av1_rtcd.h"
|
||||
|
||||
#include "aom_dsp/aom_convolve.h"
|
||||
#include "aom_dsp/x86/convolve_avx2.h"
|
||||
#include "aom_dsp/x86/convolve_common_intrin.h"
|
||||
#include "aom_dsp/aom_dsp_common.h"
|
||||
|
|
|
|||
|
|
@ -11,9 +11,8 @@
|
|||
|
||||
#include <emmintrin.h>
|
||||
|
||||
#include "config/aom_dsp_rtcd.h"
|
||||
#include "config/av1_rtcd.h"
|
||||
|
||||
#include "aom_dsp/aom_convolve.h"
|
||||
#include "aom_dsp/aom_dsp_common.h"
|
||||
#include "aom_dsp/aom_filter.h"
|
||||
#include "aom_dsp/x86/convolve_sse2.h"
|
||||
|
|
|
|||
11
third_party/aom/av1/common/x86/convolve_sse2.c
vendored
11
third_party/aom/av1/common/x86/convolve_sse2.c
vendored
|
|
@ -11,9 +11,8 @@
|
|||
|
||||
#include <emmintrin.h>
|
||||
|
||||
#include "config/aom_dsp_rtcd.h"
|
||||
#include "config/av1_rtcd.h"
|
||||
|
||||
#include "aom_dsp/aom_convolve.h"
|
||||
#include "aom_dsp/aom_dsp_common.h"
|
||||
#include "aom_dsp/aom_filter.h"
|
||||
#include "aom_dsp/x86/convolve_common_intrin.h"
|
||||
|
|
@ -76,8 +75,8 @@ static INLINE __m128i convolve_hi_y(const __m128i *const s,
|
|||
return convolve(ss, coeffs);
|
||||
}
|
||||
|
||||
void av1_convolve_y_sr_sse2(const uint8_t *src, int src_stride,
|
||||
const uint8_t *dst, int dst_stride, int w, int h,
|
||||
void av1_convolve_y_sr_sse2(const uint8_t *src, int src_stride, uint8_t *dst,
|
||||
int dst_stride, int w, int h,
|
||||
const InterpFilterParams *filter_params_x,
|
||||
const InterpFilterParams *filter_params_y,
|
||||
const int subpel_x_q4, const int subpel_y_q4,
|
||||
|
|
@ -237,8 +236,8 @@ void av1_convolve_y_sr_sse2(const uint8_t *src, int src_stride,
|
|||
}
|
||||
}
|
||||
|
||||
void av1_convolve_x_sr_sse2(const uint8_t *src, int src_stride,
|
||||
const uint8_t *dst, int dst_stride, int w, int h,
|
||||
void av1_convolve_x_sr_sse2(const uint8_t *src, int src_stride, uint8_t *dst,
|
||||
int dst_stride, int w, int h,
|
||||
const InterpFilterParams *filter_params_x,
|
||||
const InterpFilterParams *filter_params_y,
|
||||
const int subpel_x_q4, const int subpel_y_q4,
|
||||
|
|
|
|||
|
|
@ -14,7 +14,6 @@
|
|||
|
||||
#include "config/aom_dsp_rtcd.h"
|
||||
|
||||
#include "aom_dsp/aom_convolve.h"
|
||||
#include "aom_dsp/x86/convolve_avx2.h"
|
||||
#include "aom_dsp/x86/synonyms.h"
|
||||
#include "aom_dsp/aom_dsp_common.h"
|
||||
|
|
|
|||
|
|
@ -15,7 +15,6 @@
|
|||
|
||||
#include "config/aom_dsp_rtcd.h"
|
||||
|
||||
#include "aom_dsp/aom_convolve.h"
|
||||
#include "aom_dsp/aom_dsp_common.h"
|
||||
#include "aom_dsp/aom_filter.h"
|
||||
#include "aom_dsp/x86/convolve_sse2.h"
|
||||
|
|
|
|||
|
|
@ -14,7 +14,6 @@
|
|||
|
||||
#include "config/aom_dsp_rtcd.h"
|
||||
|
||||
#include "aom_dsp/aom_convolve.h"
|
||||
#include "aom_dsp/aom_dsp_common.h"
|
||||
#include "aom_dsp/aom_filter.h"
|
||||
#include "aom_dsp/x86/convolve_sse2.h"
|
||||
|
|
|
|||
1133
third_party/aom/av1/common/x86/highbd_inv_txfm_avx2.c
vendored
1133
third_party/aom/av1/common/x86/highbd_inv_txfm_avx2.c
vendored
File diff suppressed because it is too large
Load diff
3965
third_party/aom/av1/common/x86/highbd_inv_txfm_sse4.c
vendored
3965
third_party/aom/av1/common/x86/highbd_inv_txfm_sse4.c
vendored
File diff suppressed because it is too large
Load diff
|
|
@ -14,7 +14,6 @@
|
|||
|
||||
#include "config/aom_dsp_rtcd.h"
|
||||
|
||||
#include "aom_dsp/aom_convolve.h"
|
||||
#include "aom_dsp/x86/convolve_avx2.h"
|
||||
#include "aom_dsp/x86/convolve_common_intrin.h"
|
||||
#include "aom_dsp/x86/convolve_sse4_1.h"
|
||||
|
|
|
|||
|
|
@ -9,8 +9,8 @@
|
|||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#ifndef _HIGHBD_TXFM_UTILITY_SSE4_H
|
||||
#define _HIGHBD_TXFM_UTILITY_SSE4_H
|
||||
#ifndef AOM_AV1_COMMON_X86_HIGHBD_TXFM_UTILITY_SSE4_H_
|
||||
#define AOM_AV1_COMMON_X86_HIGHBD_TXFM_UTILITY_SSE4_H_
|
||||
|
||||
#include <smmintrin.h> /* SSE4.1 */
|
||||
|
||||
|
|
@ -75,6 +75,17 @@ static INLINE void transpose_16x16(const __m128i *in, __m128i *out) {
|
|||
out[63]);
|
||||
}
|
||||
|
||||
static INLINE void transpose_32x32(const __m128i *input, __m128i *output) {
|
||||
for (int j = 0; j < 8; j++) {
|
||||
for (int i = 0; i < 8; i++) {
|
||||
TRANSPOSE_4X4(input[i * 32 + j + 0], input[i * 32 + j + 8],
|
||||
input[i * 32 + j + 16], input[i * 32 + j + 24],
|
||||
output[j * 32 + i + 0], output[j * 32 + i + 8],
|
||||
output[j * 32 + i + 16], output[j * 32 + i + 24]);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Note:
|
||||
// rounding = 1 << (bit - 1)
|
||||
static INLINE __m128i half_btf_sse4_1(const __m128i *w0, const __m128i *n0,
|
||||
|
|
@ -100,4 +111,15 @@ static INLINE __m128i half_btf_0_sse4_1(const __m128i *w0, const __m128i *n0,
|
|||
return x;
|
||||
}
|
||||
|
||||
#endif // _HIGHBD_TXFM_UTILITY_SSE4_H
|
||||
typedef void (*transform_1d_sse4_1)(__m128i *in, __m128i *out, int bit,
|
||||
int do_cols, int bd, int out_shift);
|
||||
|
||||
typedef void (*fwd_transform_1d_sse4_1)(__m128i *in, __m128i *out, int bit,
|
||||
const int num_cols);
|
||||
|
||||
void av1_highbd_inv_txfm2d_add_universe_sse4_1(const int32_t *input,
|
||||
uint8_t *output, int stride,
|
||||
TX_TYPE tx_type, TX_SIZE tx_size,
|
||||
int eob, const int bd);
|
||||
|
||||
#endif // AOM_AV1_COMMON_X86_HIGHBD_TXFM_UTILITY_SSE4_H_
|
||||
|
|
|
|||
|
|
@ -19,10 +19,21 @@ static const uint8_t warp_highbd_arrange_bytes[16] = {
|
|||
0, 2, 4, 6, 8, 10, 12, 14, 1, 3, 5, 7, 9, 11, 13, 15
|
||||
};
|
||||
|
||||
static INLINE void horizontal_filter(__m128i src, __m128i src2, __m128i *tmp,
|
||||
int sx, int alpha, int k,
|
||||
const int offset_bits_horiz,
|
||||
const int reduce_bits_horiz) {
|
||||
static const uint8_t highbd_shuffle_alpha0_mask0[16] = {
|
||||
0, 1, 2, 3, 0, 1, 2, 3, 0, 1, 2, 3, 0, 1, 2, 3
|
||||
};
|
||||
static const uint8_t highbd_shuffle_alpha0_mask1[16] = {
|
||||
4, 5, 6, 7, 4, 5, 6, 7, 4, 5, 6, 7, 4, 5, 6, 7
|
||||
};
|
||||
static const uint8_t highbd_shuffle_alpha0_mask2[16] = {
|
||||
8, 9, 10, 11, 8, 9, 10, 11, 8, 9, 10, 11, 8, 9, 10, 11
|
||||
};
|
||||
static const uint8_t highbd_shuffle_alpha0_mask3[16] = {
|
||||
12, 13, 14, 15, 12, 13, 14, 15, 12, 13, 14, 15, 12, 13, 14, 15
|
||||
};
|
||||
|
||||
static INLINE void highbd_prepare_horizontal_filter_coeff(int alpha, int sx,
|
||||
__m128i *coeff) {
|
||||
// Filter even-index pixels
|
||||
const __m128i tmp_0 = _mm_loadu_si128(
|
||||
(__m128i *)(warped_filter + ((sx + 0 * alpha) >> WARPEDDIFF_PREC_BITS)));
|
||||
|
|
@ -43,27 +54,13 @@ static INLINE void horizontal_filter(__m128i src, __m128i src2, __m128i *tmp,
|
|||
const __m128i tmp_14 = _mm_unpackhi_epi32(tmp_4, tmp_6);
|
||||
|
||||
// coeffs 0 1 0 1 0 1 0 1 for pixels 0, 2, 4, 6
|
||||
const __m128i coeff_0 = _mm_unpacklo_epi64(tmp_8, tmp_10);
|
||||
coeff[0] = _mm_unpacklo_epi64(tmp_8, tmp_10);
|
||||
// coeffs 2 3 2 3 2 3 2 3 for pixels 0, 2, 4, 6
|
||||
const __m128i coeff_2 = _mm_unpackhi_epi64(tmp_8, tmp_10);
|
||||
coeff[2] = _mm_unpackhi_epi64(tmp_8, tmp_10);
|
||||
// coeffs 4 5 4 5 4 5 4 5 for pixels 0, 2, 4, 6
|
||||
const __m128i coeff_4 = _mm_unpacklo_epi64(tmp_12, tmp_14);
|
||||
coeff[4] = _mm_unpacklo_epi64(tmp_12, tmp_14);
|
||||
// coeffs 6 7 6 7 6 7 6 7 for pixels 0, 2, 4, 6
|
||||
const __m128i coeff_6 = _mm_unpackhi_epi64(tmp_12, tmp_14);
|
||||
|
||||
const __m128i round_const = _mm_set1_epi32((1 << offset_bits_horiz) +
|
||||
((1 << reduce_bits_horiz) >> 1));
|
||||
|
||||
// Calculate filtered results
|
||||
const __m128i res_0 = _mm_madd_epi16(src, coeff_0);
|
||||
const __m128i res_2 = _mm_madd_epi16(_mm_alignr_epi8(src2, src, 4), coeff_2);
|
||||
const __m128i res_4 = _mm_madd_epi16(_mm_alignr_epi8(src2, src, 8), coeff_4);
|
||||
const __m128i res_6 = _mm_madd_epi16(_mm_alignr_epi8(src2, src, 12), coeff_6);
|
||||
|
||||
__m128i res_even =
|
||||
_mm_add_epi32(_mm_add_epi32(res_0, res_4), _mm_add_epi32(res_2, res_6));
|
||||
res_even = _mm_sra_epi32(_mm_add_epi32(res_even, round_const),
|
||||
_mm_cvtsi32_si128(reduce_bits_horiz));
|
||||
coeff[6] = _mm_unpackhi_epi64(tmp_12, tmp_14);
|
||||
|
||||
// Filter odd-index pixels
|
||||
const __m128i tmp_1 = _mm_loadu_si128(
|
||||
|
|
@ -80,15 +77,63 @@ static INLINE void horizontal_filter(__m128i src, __m128i src2, __m128i *tmp,
|
|||
const __m128i tmp_13 = _mm_unpackhi_epi32(tmp_1, tmp_3);
|
||||
const __m128i tmp_15 = _mm_unpackhi_epi32(tmp_5, tmp_7);
|
||||
|
||||
const __m128i coeff_1 = _mm_unpacklo_epi64(tmp_9, tmp_11);
|
||||
const __m128i coeff_3 = _mm_unpackhi_epi64(tmp_9, tmp_11);
|
||||
const __m128i coeff_5 = _mm_unpacklo_epi64(tmp_13, tmp_15);
|
||||
const __m128i coeff_7 = _mm_unpackhi_epi64(tmp_13, tmp_15);
|
||||
coeff[1] = _mm_unpacklo_epi64(tmp_9, tmp_11);
|
||||
coeff[3] = _mm_unpackhi_epi64(tmp_9, tmp_11);
|
||||
coeff[5] = _mm_unpacklo_epi64(tmp_13, tmp_15);
|
||||
coeff[7] = _mm_unpackhi_epi64(tmp_13, tmp_15);
|
||||
}
|
||||
|
||||
const __m128i res_1 = _mm_madd_epi16(_mm_alignr_epi8(src2, src, 2), coeff_1);
|
||||
const __m128i res_3 = _mm_madd_epi16(_mm_alignr_epi8(src2, src, 6), coeff_3);
|
||||
const __m128i res_5 = _mm_madd_epi16(_mm_alignr_epi8(src2, src, 10), coeff_5);
|
||||
const __m128i res_7 = _mm_madd_epi16(_mm_alignr_epi8(src2, src, 14), coeff_7);
|
||||
static INLINE void highbd_prepare_horizontal_filter_coeff_alpha0(
|
||||
int sx, __m128i *coeff) {
|
||||
// Filter coeff
|
||||
const __m128i tmp_0 = _mm_loadu_si128(
|
||||
(__m128i *)(warped_filter + (sx >> WARPEDDIFF_PREC_BITS)));
|
||||
|
||||
coeff[0] = _mm_shuffle_epi8(
|
||||
tmp_0, _mm_loadu_si128((__m128i *)highbd_shuffle_alpha0_mask0));
|
||||
coeff[2] = _mm_shuffle_epi8(
|
||||
tmp_0, _mm_loadu_si128((__m128i *)highbd_shuffle_alpha0_mask1));
|
||||
coeff[4] = _mm_shuffle_epi8(
|
||||
tmp_0, _mm_loadu_si128((__m128i *)highbd_shuffle_alpha0_mask2));
|
||||
coeff[6] = _mm_shuffle_epi8(
|
||||
tmp_0, _mm_loadu_si128((__m128i *)highbd_shuffle_alpha0_mask3));
|
||||
|
||||
coeff[1] = coeff[0];
|
||||
coeff[3] = coeff[2];
|
||||
coeff[5] = coeff[4];
|
||||
coeff[7] = coeff[6];
|
||||
}
|
||||
|
||||
static INLINE void highbd_filter_src_pixels(
|
||||
const __m128i *src, const __m128i *src2, __m128i *tmp, __m128i *coeff,
|
||||
const int offset_bits_horiz, const int reduce_bits_horiz, int k) {
|
||||
const __m128i src_1 = *src;
|
||||
const __m128i src2_1 = *src2;
|
||||
|
||||
const __m128i round_const = _mm_set1_epi32((1 << offset_bits_horiz) +
|
||||
((1 << reduce_bits_horiz) >> 1));
|
||||
|
||||
const __m128i res_0 = _mm_madd_epi16(src_1, coeff[0]);
|
||||
const __m128i res_2 =
|
||||
_mm_madd_epi16(_mm_alignr_epi8(src2_1, src_1, 4), coeff[2]);
|
||||
const __m128i res_4 =
|
||||
_mm_madd_epi16(_mm_alignr_epi8(src2_1, src_1, 8), coeff[4]);
|
||||
const __m128i res_6 =
|
||||
_mm_madd_epi16(_mm_alignr_epi8(src2_1, src_1, 12), coeff[6]);
|
||||
|
||||
__m128i res_even =
|
||||
_mm_add_epi32(_mm_add_epi32(res_0, res_4), _mm_add_epi32(res_2, res_6));
|
||||
res_even = _mm_sra_epi32(_mm_add_epi32(res_even, round_const),
|
||||
_mm_cvtsi32_si128(reduce_bits_horiz));
|
||||
|
||||
const __m128i res_1 =
|
||||
_mm_madd_epi16(_mm_alignr_epi8(src2_1, src_1, 2), coeff[1]);
|
||||
const __m128i res_3 =
|
||||
_mm_madd_epi16(_mm_alignr_epi8(src2_1, src_1, 6), coeff[3]);
|
||||
const __m128i res_5 =
|
||||
_mm_madd_epi16(_mm_alignr_epi8(src2_1, src_1, 10), coeff[5]);
|
||||
const __m128i res_7 =
|
||||
_mm_madd_epi16(_mm_alignr_epi8(src2_1, src_1, 14), coeff[7]);
|
||||
|
||||
__m128i res_odd =
|
||||
_mm_add_epi32(_mm_add_epi32(res_1, res_5), _mm_add_epi32(res_3, res_7));
|
||||
|
|
@ -101,6 +146,145 @@ static INLINE void horizontal_filter(__m128i src, __m128i src2, __m128i *tmp,
|
|||
tmp[k + 7] = _mm_packs_epi32(res_even, res_odd);
|
||||
}
|
||||
|
||||
static INLINE void highbd_horiz_filter(const __m128i *src, const __m128i *src2,
|
||||
__m128i *tmp, int sx, int alpha, int k,
|
||||
const int offset_bits_horiz,
|
||||
const int reduce_bits_horiz) {
|
||||
__m128i coeff[8];
|
||||
highbd_prepare_horizontal_filter_coeff(alpha, sx, coeff);
|
||||
highbd_filter_src_pixels(src, src2, tmp, coeff, offset_bits_horiz,
|
||||
reduce_bits_horiz, k);
|
||||
}
|
||||
|
||||
static INLINE void highbd_warp_horizontal_filter_alpha0_beta0(
|
||||
const uint16_t *ref, __m128i *tmp, int stride, int32_t ix4, int32_t iy4,
|
||||
int32_t sx4, int alpha, int beta, int p_height, int height, int i,
|
||||
const int offset_bits_horiz, const int reduce_bits_horiz) {
|
||||
(void)beta;
|
||||
(void)alpha;
|
||||
int k;
|
||||
|
||||
__m128i coeff[8];
|
||||
highbd_prepare_horizontal_filter_coeff_alpha0(sx4, coeff);
|
||||
|
||||
for (k = -7; k < AOMMIN(8, p_height - i); ++k) {
|
||||
int iy = iy4 + k;
|
||||
if (iy < 0)
|
||||
iy = 0;
|
||||
else if (iy > height - 1)
|
||||
iy = height - 1;
|
||||
|
||||
// Load source pixels
|
||||
const __m128i src =
|
||||
_mm_loadu_si128((__m128i *)(ref + iy * stride + ix4 - 7));
|
||||
const __m128i src2 =
|
||||
_mm_loadu_si128((__m128i *)(ref + iy * stride + ix4 + 1));
|
||||
highbd_filter_src_pixels(&src, &src2, tmp, coeff, offset_bits_horiz,
|
||||
reduce_bits_horiz, k);
|
||||
}
|
||||
}
|
||||
|
||||
static INLINE void highbd_warp_horizontal_filter_alpha0(
|
||||
const uint16_t *ref, __m128i *tmp, int stride, int32_t ix4, int32_t iy4,
|
||||
int32_t sx4, int alpha, int beta, int p_height, int height, int i,
|
||||
const int offset_bits_horiz, const int reduce_bits_horiz) {
|
||||
(void)alpha;
|
||||
int k;
|
||||
for (k = -7; k < AOMMIN(8, p_height - i); ++k) {
|
||||
int iy = iy4 + k;
|
||||
if (iy < 0)
|
||||
iy = 0;
|
||||
else if (iy > height - 1)
|
||||
iy = height - 1;
|
||||
int sx = sx4 + beta * (k + 4);
|
||||
|
||||
// Load source pixels
|
||||
const __m128i src =
|
||||
_mm_loadu_si128((__m128i *)(ref + iy * stride + ix4 - 7));
|
||||
const __m128i src2 =
|
||||
_mm_loadu_si128((__m128i *)(ref + iy * stride + ix4 + 1));
|
||||
|
||||
__m128i coeff[8];
|
||||
highbd_prepare_horizontal_filter_coeff_alpha0(sx, coeff);
|
||||
highbd_filter_src_pixels(&src, &src2, tmp, coeff, offset_bits_horiz,
|
||||
reduce_bits_horiz, k);
|
||||
}
|
||||
}
|
||||
|
||||
static INLINE void highbd_warp_horizontal_filter_beta0(
|
||||
const uint16_t *ref, __m128i *tmp, int stride, int32_t ix4, int32_t iy4,
|
||||
int32_t sx4, int alpha, int beta, int p_height, int height, int i,
|
||||
const int offset_bits_horiz, const int reduce_bits_horiz) {
|
||||
(void)beta;
|
||||
int k;
|
||||
__m128i coeff[8];
|
||||
highbd_prepare_horizontal_filter_coeff(alpha, sx4, coeff);
|
||||
|
||||
for (k = -7; k < AOMMIN(8, p_height - i); ++k) {
|
||||
int iy = iy4 + k;
|
||||
if (iy < 0)
|
||||
iy = 0;
|
||||
else if (iy > height - 1)
|
||||
iy = height - 1;
|
||||
|
||||
// Load source pixels
|
||||
const __m128i src =
|
||||
_mm_loadu_si128((__m128i *)(ref + iy * stride + ix4 - 7));
|
||||
const __m128i src2 =
|
||||
_mm_loadu_si128((__m128i *)(ref + iy * stride + ix4 + 1));
|
||||
highbd_filter_src_pixels(&src, &src2, tmp, coeff, offset_bits_horiz,
|
||||
reduce_bits_horiz, k);
|
||||
}
|
||||
}
|
||||
|
||||
static INLINE void highbd_warp_horizontal_filter(
|
||||
const uint16_t *ref, __m128i *tmp, int stride, int32_t ix4, int32_t iy4,
|
||||
int32_t sx4, int alpha, int beta, int p_height, int height, int i,
|
||||
const int offset_bits_horiz, const int reduce_bits_horiz) {
|
||||
int k;
|
||||
for (k = -7; k < AOMMIN(8, p_height - i); ++k) {
|
||||
int iy = iy4 + k;
|
||||
if (iy < 0)
|
||||
iy = 0;
|
||||
else if (iy > height - 1)
|
||||
iy = height - 1;
|
||||
int sx = sx4 + beta * (k + 4);
|
||||
|
||||
// Load source pixels
|
||||
const __m128i src =
|
||||
_mm_loadu_si128((__m128i *)(ref + iy * stride + ix4 - 7));
|
||||
const __m128i src2 =
|
||||
_mm_loadu_si128((__m128i *)(ref + iy * stride + ix4 + 1));
|
||||
|
||||
highbd_horiz_filter(&src, &src2, tmp, sx, alpha, k, offset_bits_horiz,
|
||||
reduce_bits_horiz);
|
||||
}
|
||||
}
|
||||
|
||||
static INLINE void highbd_prepare_warp_horizontal_filter(
|
||||
const uint16_t *ref, __m128i *tmp, int stride, int32_t ix4, int32_t iy4,
|
||||
int32_t sx4, int alpha, int beta, int p_height, int height, int i,
|
||||
const int offset_bits_horiz, const int reduce_bits_horiz) {
|
||||
if (alpha == 0 && beta == 0)
|
||||
highbd_warp_horizontal_filter_alpha0_beta0(
|
||||
ref, tmp, stride, ix4, iy4, sx4, alpha, beta, p_height, height, i,
|
||||
offset_bits_horiz, reduce_bits_horiz);
|
||||
|
||||
else if (alpha == 0 && beta != 0)
|
||||
highbd_warp_horizontal_filter_alpha0(ref, tmp, stride, ix4, iy4, sx4, alpha,
|
||||
beta, p_height, height, i,
|
||||
offset_bits_horiz, reduce_bits_horiz);
|
||||
|
||||
else if (alpha != 0 && beta == 0)
|
||||
highbd_warp_horizontal_filter_beta0(ref, tmp, stride, ix4, iy4, sx4, alpha,
|
||||
beta, p_height, height, i,
|
||||
offset_bits_horiz, reduce_bits_horiz);
|
||||
else
|
||||
highbd_warp_horizontal_filter(ref, tmp, stride, ix4, iy4, sx4, alpha, beta,
|
||||
p_height, height, i, offset_bits_horiz,
|
||||
reduce_bits_horiz);
|
||||
}
|
||||
|
||||
void av1_highbd_warp_affine_sse4_1(const int32_t *mat, const uint16_t *ref,
|
||||
int width, int height, int stride,
|
||||
uint16_t *pred, int p_col, int p_row,
|
||||
|
|
@ -247,27 +431,13 @@ void av1_highbd_warp_affine_sse4_1(const int32_t *mat, const uint16_t *ref,
|
|||
const __m128i src_padded = _mm_unpacklo_epi8(src_lo, src_hi);
|
||||
const __m128i src2_padded = _mm_unpackhi_epi8(src_lo, src_hi);
|
||||
|
||||
horizontal_filter(src_padded, src2_padded, tmp, sx, alpha, k,
|
||||
offset_bits_horiz, reduce_bits_horiz);
|
||||
highbd_horiz_filter(&src_padded, &src2_padded, tmp, sx, alpha, k,
|
||||
offset_bits_horiz, reduce_bits_horiz);
|
||||
}
|
||||
} else {
|
||||
for (k = -7; k < AOMMIN(8, p_height - i); ++k) {
|
||||
int iy = iy4 + k;
|
||||
if (iy < 0)
|
||||
iy = 0;
|
||||
else if (iy > height - 1)
|
||||
iy = height - 1;
|
||||
int sx = sx4 + beta * (k + 4);
|
||||
|
||||
// Load source pixels
|
||||
const __m128i src =
|
||||
_mm_loadu_si128((__m128i *)(ref + iy * stride + ix4 - 7));
|
||||
const __m128i src2 =
|
||||
_mm_loadu_si128((__m128i *)(ref + iy * stride + ix4 + 1));
|
||||
|
||||
horizontal_filter(src, src2, tmp, sx, alpha, k, offset_bits_horiz,
|
||||
reduce_bits_horiz);
|
||||
}
|
||||
highbd_prepare_warp_horizontal_filter(
|
||||
ref, tmp, stride, ix4, iy4, sx4, alpha, beta, p_height, height, i,
|
||||
offset_bits_horiz, reduce_bits_horiz);
|
||||
}
|
||||
|
||||
// Vertical filter
|
||||
|
|
|
|||
209
third_party/aom/av1/common/x86/jnt_convolve_avx2.c
vendored
209
third_party/aom/av1/common/x86/jnt_convolve_avx2.c
vendored
|
|
@ -13,7 +13,6 @@
|
|||
|
||||
#include "config/aom_dsp_rtcd.h"
|
||||
|
||||
#include "aom_dsp/aom_convolve.h"
|
||||
#include "aom_dsp/x86/convolve_avx2.h"
|
||||
#include "aom_dsp/x86/convolve_common_intrin.h"
|
||||
#include "aom_dsp/x86/convolve_sse4_1.h"
|
||||
|
|
@ -21,6 +20,21 @@
|
|||
#include "aom_dsp/aom_filter.h"
|
||||
#include "av1/common/convolve.h"
|
||||
|
||||
static INLINE __m256i unpack_weights_avx2(ConvolveParams *conv_params) {
|
||||
const int w0 = conv_params->fwd_offset;
|
||||
const int w1 = conv_params->bck_offset;
|
||||
const __m256i wt0 = _mm256_set1_epi16(w0);
|
||||
const __m256i wt1 = _mm256_set1_epi16(w1);
|
||||
const __m256i wt = _mm256_unpacklo_epi16(wt0, wt1);
|
||||
return wt;
|
||||
}
|
||||
|
||||
static INLINE __m256i load_line2_avx2(const void *a, const void *b) {
|
||||
return _mm256_permute2x128_si256(
|
||||
_mm256_castsi128_si256(_mm_loadu_si128((__m128i *)a)),
|
||||
_mm256_castsi128_si256(_mm_loadu_si128((__m128i *)b)), 0x20);
|
||||
}
|
||||
|
||||
void av1_jnt_convolve_x_avx2(const uint8_t *src, int src_stride, uint8_t *dst0,
|
||||
int dst_stride0, int w, int h,
|
||||
const InterpFilterParams *filter_params_x,
|
||||
|
|
@ -34,11 +48,7 @@ void av1_jnt_convolve_x_avx2(const uint8_t *src, int src_stride, uint8_t *dst0,
|
|||
const int fo_horiz = filter_params_x->taps / 2 - 1;
|
||||
const uint8_t *const src_ptr = src - fo_horiz;
|
||||
const int bits = FILTER_BITS - conv_params->round_1;
|
||||
const int w0 = conv_params->fwd_offset;
|
||||
const int w1 = conv_params->bck_offset;
|
||||
const __m256i wt0 = _mm256_set1_epi16(w0);
|
||||
const __m256i wt1 = _mm256_set1_epi16(w1);
|
||||
const __m256i wt = _mm256_unpacklo_epi16(wt0, wt1);
|
||||
const __m256i wt = unpack_weights_avx2(conv_params);
|
||||
const int do_average = conv_params->do_average;
|
||||
const int use_jnt_comp_avg = conv_params->use_jnt_comp_avg;
|
||||
const int offset_0 =
|
||||
|
|
@ -68,13 +78,11 @@ void av1_jnt_convolve_x_avx2(const uint8_t *src, int src_stride, uint8_t *dst0,
|
|||
(void)subpel_y_q4;
|
||||
|
||||
for (i = 0; i < h; i += 2) {
|
||||
const uint8_t *src_data = src_ptr + i * src_stride;
|
||||
CONV_BUF_TYPE *dst_data = dst + i * dst_stride;
|
||||
for (j = 0; j < w; j += 8) {
|
||||
const __m256i data = _mm256_permute2x128_si256(
|
||||
_mm256_castsi128_si256(
|
||||
_mm_loadu_si128((__m128i *)(&src_ptr[i * src_stride + j]))),
|
||||
_mm256_castsi128_si256(_mm_loadu_si128(
|
||||
(__m128i *)(&src_ptr[i * src_stride + j + src_stride]))),
|
||||
0x20);
|
||||
const __m256i data =
|
||||
load_line2_avx2(&src_data[j], &src_data[j + src_stride]);
|
||||
|
||||
__m256i res = convolve_lowbd_x(data, coeffs, filt);
|
||||
|
||||
|
|
@ -86,13 +94,8 @@ void av1_jnt_convolve_x_avx2(const uint8_t *src, int src_stride, uint8_t *dst0,
|
|||
|
||||
// Accumulate values into the destination buffer
|
||||
if (do_average) {
|
||||
const __m256i data_ref_0 = _mm256_permute2x128_si256(
|
||||
_mm256_castsi128_si256(
|
||||
_mm_loadu_si128((__m128i *)(&dst[i * dst_stride + j]))),
|
||||
_mm256_castsi128_si256(_mm_loadu_si128(
|
||||
(__m128i *)(&dst[i * dst_stride + j + dst_stride]))),
|
||||
0x20);
|
||||
|
||||
const __m256i data_ref_0 =
|
||||
load_line2_avx2(&dst_data[j], &dst_data[j + dst_stride]);
|
||||
const __m256i comp_avg_res =
|
||||
comp_avg(&data_ref_0, &res_unsigned, &wt, use_jnt_comp_avg);
|
||||
|
||||
|
|
@ -141,11 +144,7 @@ void av1_jnt_convolve_y_avx2(const uint8_t *src, int src_stride, uint8_t *dst0,
|
|||
const __m256i round_const =
|
||||
_mm256_set1_epi32((1 << conv_params->round_1) >> 1);
|
||||
const __m128i round_shift = _mm_cvtsi32_si128(conv_params->round_1);
|
||||
const int w0 = conv_params->fwd_offset;
|
||||
const int w1 = conv_params->bck_offset;
|
||||
const __m256i wt0 = _mm256_set1_epi16(w0);
|
||||
const __m256i wt1 = _mm256_set1_epi16(w1);
|
||||
const __m256i wt = _mm256_unpacklo_epi16(wt0, wt1);
|
||||
const __m256i wt = unpack_weights_avx2(conv_params);
|
||||
const int do_average = conv_params->do_average;
|
||||
const int use_jnt_comp_avg = conv_params->use_jnt_comp_avg;
|
||||
const int offset_0 =
|
||||
|
|
@ -172,72 +171,35 @@ void av1_jnt_convolve_y_avx2(const uint8_t *src, int src_stride, uint8_t *dst0,
|
|||
for (j = 0; j < w; j += 16) {
|
||||
const uint8_t *data = &src_ptr[j];
|
||||
__m256i src6;
|
||||
|
||||
// Load lines a and b. Line a to lower 128, line b to upper 128
|
||||
const __m256i src_01a = _mm256_permute2x128_si256(
|
||||
_mm256_castsi128_si256(
|
||||
_mm_loadu_si128((__m128i *)(data + 0 * src_stride))),
|
||||
_mm256_castsi128_si256(
|
||||
_mm_loadu_si128((__m128i *)(data + 1 * src_stride))),
|
||||
0x20);
|
||||
|
||||
const __m256i src_12a = _mm256_permute2x128_si256(
|
||||
_mm256_castsi128_si256(
|
||||
_mm_loadu_si128((__m128i *)(data + 1 * src_stride))),
|
||||
_mm256_castsi128_si256(
|
||||
_mm_loadu_si128((__m128i *)(data + 2 * src_stride))),
|
||||
0x20);
|
||||
|
||||
const __m256i src_23a = _mm256_permute2x128_si256(
|
||||
_mm256_castsi128_si256(
|
||||
_mm_loadu_si128((__m128i *)(data + 2 * src_stride))),
|
||||
_mm256_castsi128_si256(
|
||||
_mm_loadu_si128((__m128i *)(data + 3 * src_stride))),
|
||||
0x20);
|
||||
|
||||
const __m256i src_34a = _mm256_permute2x128_si256(
|
||||
_mm256_castsi128_si256(
|
||||
_mm_loadu_si128((__m128i *)(data + 3 * src_stride))),
|
||||
_mm256_castsi128_si256(
|
||||
_mm_loadu_si128((__m128i *)(data + 4 * src_stride))),
|
||||
0x20);
|
||||
|
||||
const __m256i src_45a = _mm256_permute2x128_si256(
|
||||
_mm256_castsi128_si256(
|
||||
_mm_loadu_si128((__m128i *)(data + 4 * src_stride))),
|
||||
_mm256_castsi128_si256(
|
||||
_mm_loadu_si128((__m128i *)(data + 5 * src_stride))),
|
||||
0x20);
|
||||
|
||||
src6 = _mm256_castsi128_si256(
|
||||
_mm_loadu_si128((__m128i *)(data + 6 * src_stride)));
|
||||
const __m256i src_56a = _mm256_permute2x128_si256(
|
||||
_mm256_castsi128_si256(
|
||||
_mm_loadu_si128((__m128i *)(data + 5 * src_stride))),
|
||||
src6, 0x20);
|
||||
|
||||
s[0] = _mm256_unpacklo_epi8(src_01a, src_12a);
|
||||
s[1] = _mm256_unpacklo_epi8(src_23a, src_34a);
|
||||
s[2] = _mm256_unpacklo_epi8(src_45a, src_56a);
|
||||
|
||||
s[4] = _mm256_unpackhi_epi8(src_01a, src_12a);
|
||||
s[5] = _mm256_unpackhi_epi8(src_23a, src_34a);
|
||||
s[6] = _mm256_unpackhi_epi8(src_45a, src_56a);
|
||||
{
|
||||
__m256i src_ab[7];
|
||||
__m256i src_a[7];
|
||||
src_a[0] = _mm256_castsi128_si256(_mm_loadu_si128((__m128i *)data));
|
||||
for (int kk = 0; kk < 6; ++kk) {
|
||||
data += src_stride;
|
||||
src_a[kk + 1] =
|
||||
_mm256_castsi128_si256(_mm_loadu_si128((__m128i *)data));
|
||||
src_ab[kk] = _mm256_permute2x128_si256(src_a[kk], src_a[kk + 1], 0x20);
|
||||
}
|
||||
src6 = src_a[6];
|
||||
s[0] = _mm256_unpacklo_epi8(src_ab[0], src_ab[1]);
|
||||
s[1] = _mm256_unpacklo_epi8(src_ab[2], src_ab[3]);
|
||||
s[2] = _mm256_unpacklo_epi8(src_ab[4], src_ab[5]);
|
||||
s[4] = _mm256_unpackhi_epi8(src_ab[0], src_ab[1]);
|
||||
s[5] = _mm256_unpackhi_epi8(src_ab[2], src_ab[3]);
|
||||
s[6] = _mm256_unpackhi_epi8(src_ab[4], src_ab[5]);
|
||||
}
|
||||
|
||||
for (i = 0; i < h; i += 2) {
|
||||
data = &src_ptr[i * src_stride + j];
|
||||
const __m256i src_67a = _mm256_permute2x128_si256(
|
||||
src6,
|
||||
_mm256_castsi128_si256(
|
||||
_mm_loadu_si128((__m128i *)(data + 7 * src_stride))),
|
||||
0x20);
|
||||
data = &src_ptr[(i + 7) * src_stride + j];
|
||||
const __m256i src7 =
|
||||
_mm256_castsi128_si256(_mm_loadu_si128((__m128i *)data));
|
||||
const __m256i src_67a = _mm256_permute2x128_si256(src6, src7, 0x20);
|
||||
|
||||
src6 = _mm256_castsi128_si256(
|
||||
_mm_loadu_si128((__m128i *)(data + 8 * src_stride)));
|
||||
const __m256i src_78a = _mm256_permute2x128_si256(
|
||||
_mm256_castsi128_si256(
|
||||
_mm_loadu_si128((__m128i *)(data + 7 * src_stride))),
|
||||
src6, 0x20);
|
||||
_mm_loadu_si128((__m128i *)(data + src_stride)));
|
||||
const __m256i src_78a = _mm256_permute2x128_si256(src7, src6, 0x20);
|
||||
|
||||
s[3] = _mm256_unpacklo_epi8(src_67a, src_78a);
|
||||
s[7] = _mm256_unpackhi_epi8(src_67a, src_78a);
|
||||
|
|
@ -266,13 +228,8 @@ void av1_jnt_convolve_y_avx2(const uint8_t *src, int src_stride, uint8_t *dst0,
|
|||
|
||||
if (w - j < 16) {
|
||||
if (do_average) {
|
||||
const __m256i data_ref_0 = _mm256_permute2x128_si256(
|
||||
_mm256_castsi128_si256(
|
||||
_mm_loadu_si128((__m128i *)(&dst[i * dst_stride + j]))),
|
||||
_mm256_castsi128_si256(_mm_loadu_si128(
|
||||
(__m128i *)(&dst[i * dst_stride + j + dst_stride]))),
|
||||
0x20);
|
||||
|
||||
const __m256i data_ref_0 = load_line2_avx2(
|
||||
&dst[i * dst_stride + j], &dst[i * dst_stride + j + dst_stride]);
|
||||
const __m256i comp_avg_res =
|
||||
comp_avg(&data_ref_0, &res_lo_unsigned, &wt, use_jnt_comp_avg);
|
||||
|
||||
|
|
@ -325,19 +282,12 @@ void av1_jnt_convolve_y_avx2(const uint8_t *src, int src_stride, uint8_t *dst0,
|
|||
_mm256_add_epi16(res_hi_round, offset_const_2);
|
||||
|
||||
if (do_average) {
|
||||
const __m256i data_ref_0_lo = _mm256_permute2x128_si256(
|
||||
_mm256_castsi128_si256(
|
||||
_mm_loadu_si128((__m128i *)(&dst[i * dst_stride + j]))),
|
||||
_mm256_castsi128_si256(_mm_loadu_si128(
|
||||
(__m128i *)(&dst[i * dst_stride + j + dst_stride]))),
|
||||
0x20);
|
||||
const __m256i data_ref_0_lo = load_line2_avx2(
|
||||
&dst[i * dst_stride + j], &dst[i * dst_stride + j + dst_stride]);
|
||||
|
||||
const __m256i data_ref_0_hi = _mm256_permute2x128_si256(
|
||||
_mm256_castsi128_si256(
|
||||
_mm_loadu_si128((__m128i *)(&dst[i * dst_stride + j + 8]))),
|
||||
_mm256_castsi128_si256(_mm_loadu_si128(
|
||||
(__m128i *)(&dst[i * dst_stride + j + 8 + dst_stride]))),
|
||||
0x20);
|
||||
const __m256i data_ref_0_hi =
|
||||
load_line2_avx2(&dst[i * dst_stride + j + 8],
|
||||
&dst[i * dst_stride + j + 8 + dst_stride]);
|
||||
|
||||
const __m256i comp_avg_res_lo =
|
||||
comp_avg(&data_ref_0_lo, &res_lo_unsigned, &wt, use_jnt_comp_avg);
|
||||
|
|
@ -404,11 +354,7 @@ void av1_jnt_convolve_2d_avx2(const uint8_t *src, int src_stride, uint8_t *dst0,
|
|||
const int fo_vert = filter_params_y->taps / 2 - 1;
|
||||
const int fo_horiz = filter_params_x->taps / 2 - 1;
|
||||
const uint8_t *const src_ptr = src - fo_vert * src_stride - fo_horiz;
|
||||
const int w0 = conv_params->fwd_offset;
|
||||
const int w1 = conv_params->bck_offset;
|
||||
const __m256i wt0 = _mm256_set1_epi16(w0);
|
||||
const __m256i wt1 = _mm256_set1_epi16(w1);
|
||||
const __m256i wt = _mm256_unpacklo_epi16(wt0, wt1);
|
||||
const __m256i wt = unpack_weights_avx2(conv_params);
|
||||
const int do_average = conv_params->do_average;
|
||||
const int use_jnt_comp_avg = conv_params->use_jnt_comp_avg;
|
||||
const int offset_0 =
|
||||
|
|
@ -442,15 +388,14 @@ void av1_jnt_convolve_2d_avx2(const uint8_t *src, int src_stride, uint8_t *dst0,
|
|||
for (j = 0; j < w; j += 8) {
|
||||
/* Horizontal filter */
|
||||
{
|
||||
const uint8_t *src_h = src_ptr + j;
|
||||
for (i = 0; i < im_h; i += 2) {
|
||||
__m256i data = _mm256_castsi128_si256(
|
||||
_mm_loadu_si128((__m128i *)&src_ptr[(i * src_stride) + j]));
|
||||
__m256i data =
|
||||
_mm256_castsi128_si256(_mm_loadu_si128((__m128i *)src_h));
|
||||
if (i + 1 < im_h)
|
||||
data = _mm256_inserti128_si256(
|
||||
data,
|
||||
_mm_loadu_si128(
|
||||
(__m128i *)&src_ptr[(i * src_stride) + j + src_stride]),
|
||||
1);
|
||||
data, _mm_loadu_si128((__m128i *)(src_h + src_stride)), 1);
|
||||
src_h += (src_stride << 1);
|
||||
__m256i res = convolve_lowbd_x(data, coeffs_x, filt);
|
||||
|
||||
res = _mm256_sra_epi16(_mm256_add_epi16(res, round_const_h),
|
||||
|
|
@ -500,13 +445,9 @@ void av1_jnt_convolve_2d_avx2(const uint8_t *src, int src_stride, uint8_t *dst0,
|
|||
const __m256i res_unsigned = _mm256_add_epi16(res_16b, offset_const);
|
||||
|
||||
if (do_average) {
|
||||
const __m256i data_ref_0 = _mm256_permute2x128_si256(
|
||||
_mm256_castsi128_si256(
|
||||
_mm_loadu_si128((__m128i *)(&dst[i * dst_stride + j]))),
|
||||
_mm256_castsi128_si256(_mm_loadu_si128(
|
||||
(__m128i *)(&dst[i * dst_stride + j + dst_stride]))),
|
||||
0x20);
|
||||
|
||||
const __m256i data_ref_0 =
|
||||
load_line2_avx2(&dst[i * dst_stride + j],
|
||||
&dst[i * dst_stride + j + dst_stride]);
|
||||
const __m256i comp_avg_res =
|
||||
comp_avg(&data_ref_0, &res_unsigned, &wt, use_jnt_comp_avg);
|
||||
|
||||
|
|
@ -534,12 +475,9 @@ void av1_jnt_convolve_2d_avx2(const uint8_t *src, int src_stride, uint8_t *dst0,
|
|||
const __m256i res_unsigned = _mm256_add_epi16(res_16b, offset_const);
|
||||
|
||||
if (do_average) {
|
||||
const __m256i data_ref_0 = _mm256_permute2x128_si256(
|
||||
_mm256_castsi128_si256(
|
||||
_mm_loadu_si128((__m128i *)(&dst[i * dst_stride + j]))),
|
||||
_mm256_castsi128_si256(_mm_loadu_si128(
|
||||
(__m128i *)(&dst[i * dst_stride + j + dst_stride]))),
|
||||
0x20);
|
||||
const __m256i data_ref_0 =
|
||||
load_line2_avx2(&dst[i * dst_stride + j],
|
||||
&dst[i * dst_stride + j + dst_stride]);
|
||||
|
||||
const __m256i comp_avg_res =
|
||||
comp_avg(&data_ref_0, &res_unsigned, &wt, use_jnt_comp_avg);
|
||||
|
|
@ -598,11 +536,7 @@ void av1_jnt_convolve_2d_copy_avx2(const uint8_t *src, int src_stride,
|
|||
const __m128i left_shift = _mm_cvtsi32_si128(bits);
|
||||
const int do_average = conv_params->do_average;
|
||||
const int use_jnt_comp_avg = conv_params->use_jnt_comp_avg;
|
||||
const int w0 = conv_params->fwd_offset;
|
||||
const int w1 = conv_params->bck_offset;
|
||||
const __m256i wt0 = _mm256_set1_epi16(w0);
|
||||
const __m256i wt1 = _mm256_set1_epi16(w1);
|
||||
const __m256i wt = _mm256_unpacklo_epi16(wt0, wt1);
|
||||
const __m256i wt = unpack_weights_avx2(conv_params);
|
||||
const __m256i zero = _mm256_setzero_si256();
|
||||
|
||||
const int offset_0 =
|
||||
|
|
@ -663,13 +597,8 @@ void av1_jnt_convolve_2d_copy_avx2(const uint8_t *src, int src_stride,
|
|||
|
||||
// Accumulate values into the destination buffer
|
||||
if (do_average) {
|
||||
const __m256i data_ref_0 = _mm256_permute2x128_si256(
|
||||
_mm256_castsi128_si256(
|
||||
_mm_loadu_si128((__m128i *)(&dst[i * dst_stride + j]))),
|
||||
_mm256_castsi128_si256(_mm_loadu_si128(
|
||||
(__m128i *)(&dst[i * dst_stride + j + dst_stride]))),
|
||||
0x20);
|
||||
|
||||
const __m256i data_ref_0 = load_line2_avx2(
|
||||
&dst[i * dst_stride + j], &dst[i * dst_stride + j + dst_stride]);
|
||||
const __m256i comp_avg_res =
|
||||
comp_avg(&data_ref_0, &res_unsigned, &wt, use_jnt_comp_avg);
|
||||
|
||||
|
|
|
|||
496
third_party/aom/av1/common/x86/reconinter_avx2.c
vendored
496
third_party/aom/av1/common/x86/reconinter_avx2.c
vendored
|
|
@ -16,8 +16,504 @@
|
|||
#include "aom/aom_integer.h"
|
||||
#include "aom_dsp/blend.h"
|
||||
#include "aom_dsp/x86/synonyms.h"
|
||||
#include "aom_dsp/x86/synonyms_avx2.h"
|
||||
#include "av1/common/blockd.h"
|
||||
|
||||
static INLINE __m256i calc_mask_avx2(const __m256i mask_base, const __m256i s0,
|
||||
const __m256i s1) {
|
||||
const __m256i diff = _mm256_abs_epi16(_mm256_sub_epi16(s0, s1));
|
||||
return _mm256_abs_epi16(
|
||||
_mm256_add_epi16(mask_base, _mm256_srli_epi16(diff, 4)));
|
||||
// clamp(diff, 0, 64) can be skiped for diff is always in the range ( 38, 54)
|
||||
}
|
||||
void av1_build_compound_diffwtd_mask_avx2(uint8_t *mask,
|
||||
DIFFWTD_MASK_TYPE mask_type,
|
||||
const uint8_t *src0, int stride0,
|
||||
const uint8_t *src1, int stride1,
|
||||
int h, int w) {
|
||||
const int mb = (mask_type == DIFFWTD_38_INV) ? AOM_BLEND_A64_MAX_ALPHA : 0;
|
||||
const __m256i y_mask_base = _mm256_set1_epi16(38 - mb);
|
||||
int i = 0;
|
||||
if (4 == w) {
|
||||
do {
|
||||
const __m128i s0A = xx_loadl_32(src0);
|
||||
const __m128i s0B = xx_loadl_32(src0 + stride0);
|
||||
const __m128i s0C = xx_loadl_32(src0 + stride0 * 2);
|
||||
const __m128i s0D = xx_loadl_32(src0 + stride0 * 3);
|
||||
const __m128i s0AB = _mm_unpacklo_epi32(s0A, s0B);
|
||||
const __m128i s0CD = _mm_unpacklo_epi32(s0C, s0D);
|
||||
const __m128i s0ABCD = _mm_unpacklo_epi64(s0AB, s0CD);
|
||||
const __m256i s0ABCD_w = _mm256_cvtepu8_epi16(s0ABCD);
|
||||
|
||||
const __m128i s1A = xx_loadl_32(src1);
|
||||
const __m128i s1B = xx_loadl_32(src1 + stride1);
|
||||
const __m128i s1C = xx_loadl_32(src1 + stride1 * 2);
|
||||
const __m128i s1D = xx_loadl_32(src1 + stride1 * 3);
|
||||
const __m128i s1AB = _mm_unpacklo_epi32(s1A, s1B);
|
||||
const __m128i s1CD = _mm_unpacklo_epi32(s1C, s1D);
|
||||
const __m128i s1ABCD = _mm_unpacklo_epi64(s1AB, s1CD);
|
||||
const __m256i s1ABCD_w = _mm256_cvtepu8_epi16(s1ABCD);
|
||||
const __m256i m16 = calc_mask_avx2(y_mask_base, s0ABCD_w, s1ABCD_w);
|
||||
const __m256i m8 = _mm256_packus_epi16(m16, _mm256_setzero_si256());
|
||||
const __m128i x_m8 =
|
||||
_mm256_castsi256_si128(_mm256_permute4x64_epi64(m8, 0xd8));
|
||||
xx_storeu_128(mask, x_m8);
|
||||
src0 += (stride0 << 2);
|
||||
src1 += (stride1 << 2);
|
||||
mask += 16;
|
||||
i += 4;
|
||||
} while (i < h);
|
||||
} else if (8 == w) {
|
||||
do {
|
||||
const __m128i s0A = xx_loadl_64(src0);
|
||||
const __m128i s0B = xx_loadl_64(src0 + stride0);
|
||||
const __m128i s0C = xx_loadl_64(src0 + stride0 * 2);
|
||||
const __m128i s0D = xx_loadl_64(src0 + stride0 * 3);
|
||||
const __m256i s0AC_w = _mm256_cvtepu8_epi16(_mm_unpacklo_epi64(s0A, s0C));
|
||||
const __m256i s0BD_w = _mm256_cvtepu8_epi16(_mm_unpacklo_epi64(s0B, s0D));
|
||||
const __m128i s1A = xx_loadl_64(src1);
|
||||
const __m128i s1B = xx_loadl_64(src1 + stride1);
|
||||
const __m128i s1C = xx_loadl_64(src1 + stride1 * 2);
|
||||
const __m128i s1D = xx_loadl_64(src1 + stride1 * 3);
|
||||
const __m256i s1AB_w = _mm256_cvtepu8_epi16(_mm_unpacklo_epi64(s1A, s1C));
|
||||
const __m256i s1CD_w = _mm256_cvtepu8_epi16(_mm_unpacklo_epi64(s1B, s1D));
|
||||
const __m256i m16AC = calc_mask_avx2(y_mask_base, s0AC_w, s1AB_w);
|
||||
const __m256i m16BD = calc_mask_avx2(y_mask_base, s0BD_w, s1CD_w);
|
||||
const __m256i m8 = _mm256_packus_epi16(m16AC, m16BD);
|
||||
yy_storeu_256(mask, m8);
|
||||
src0 += stride0 << 2;
|
||||
src1 += stride1 << 2;
|
||||
mask += 32;
|
||||
i += 4;
|
||||
} while (i < h);
|
||||
} else if (16 == w) {
|
||||
do {
|
||||
const __m128i s0A = xx_load_128(src0);
|
||||
const __m128i s0B = xx_load_128(src0 + stride0);
|
||||
const __m128i s1A = xx_load_128(src1);
|
||||
const __m128i s1B = xx_load_128(src1 + stride1);
|
||||
const __m256i s0AL = _mm256_cvtepu8_epi16(s0A);
|
||||
const __m256i s0BL = _mm256_cvtepu8_epi16(s0B);
|
||||
const __m256i s1AL = _mm256_cvtepu8_epi16(s1A);
|
||||
const __m256i s1BL = _mm256_cvtepu8_epi16(s1B);
|
||||
|
||||
const __m256i m16AL = calc_mask_avx2(y_mask_base, s0AL, s1AL);
|
||||
const __m256i m16BL = calc_mask_avx2(y_mask_base, s0BL, s1BL);
|
||||
|
||||
const __m256i m8 =
|
||||
_mm256_permute4x64_epi64(_mm256_packus_epi16(m16AL, m16BL), 0xd8);
|
||||
yy_storeu_256(mask, m8);
|
||||
src0 += stride0 << 1;
|
||||
src1 += stride1 << 1;
|
||||
mask += 32;
|
||||
i += 2;
|
||||
} while (i < h);
|
||||
} else {
|
||||
do {
|
||||
int j = 0;
|
||||
do {
|
||||
const __m256i s0 = yy_loadu_256(src0 + j);
|
||||
const __m256i s1 = yy_loadu_256(src1 + j);
|
||||
const __m256i s0L = _mm256_cvtepu8_epi16(_mm256_castsi256_si128(s0));
|
||||
const __m256i s1L = _mm256_cvtepu8_epi16(_mm256_castsi256_si128(s1));
|
||||
const __m256i s0H =
|
||||
_mm256_cvtepu8_epi16(_mm256_extracti128_si256(s0, 1));
|
||||
const __m256i s1H =
|
||||
_mm256_cvtepu8_epi16(_mm256_extracti128_si256(s1, 1));
|
||||
const __m256i m16L = calc_mask_avx2(y_mask_base, s0L, s1L);
|
||||
const __m256i m16H = calc_mask_avx2(y_mask_base, s0H, s1H);
|
||||
const __m256i m8 =
|
||||
_mm256_permute4x64_epi64(_mm256_packus_epi16(m16L, m16H), 0xd8);
|
||||
yy_storeu_256(mask + j, m8);
|
||||
j += 32;
|
||||
} while (j < w);
|
||||
src0 += stride0;
|
||||
src1 += stride1;
|
||||
mask += w;
|
||||
i += 1;
|
||||
} while (i < h);
|
||||
}
|
||||
}
|
||||
|
||||
static INLINE __m256i calc_mask_d16_avx2(const __m256i *data_src0,
|
||||
const __m256i *data_src1,
|
||||
const __m256i *round_const,
|
||||
const __m256i *mask_base_16,
|
||||
const __m256i *clip_diff, int round) {
|
||||
const __m256i diffa = _mm256_subs_epu16(*data_src0, *data_src1);
|
||||
const __m256i diffb = _mm256_subs_epu16(*data_src1, *data_src0);
|
||||
const __m256i diff = _mm256_max_epu16(diffa, diffb);
|
||||
const __m256i diff_round =
|
||||
_mm256_srli_epi16(_mm256_adds_epu16(diff, *round_const), round);
|
||||
const __m256i diff_factor = _mm256_srli_epi16(diff_round, DIFF_FACTOR_LOG2);
|
||||
const __m256i diff_mask = _mm256_adds_epi16(diff_factor, *mask_base_16);
|
||||
const __m256i diff_clamp = _mm256_min_epi16(diff_mask, *clip_diff);
|
||||
return diff_clamp;
|
||||
}
|
||||
|
||||
static INLINE __m256i calc_mask_d16_inv_avx2(const __m256i *data_src0,
|
||||
const __m256i *data_src1,
|
||||
const __m256i *round_const,
|
||||
const __m256i *mask_base_16,
|
||||
const __m256i *clip_diff,
|
||||
int round) {
|
||||
const __m256i diffa = _mm256_subs_epu16(*data_src0, *data_src1);
|
||||
const __m256i diffb = _mm256_subs_epu16(*data_src1, *data_src0);
|
||||
const __m256i diff = _mm256_max_epu16(diffa, diffb);
|
||||
const __m256i diff_round =
|
||||
_mm256_srli_epi16(_mm256_adds_epu16(diff, *round_const), round);
|
||||
const __m256i diff_factor = _mm256_srli_epi16(diff_round, DIFF_FACTOR_LOG2);
|
||||
const __m256i diff_mask = _mm256_adds_epi16(diff_factor, *mask_base_16);
|
||||
const __m256i diff_clamp = _mm256_min_epi16(diff_mask, *clip_diff);
|
||||
const __m256i diff_const_16 = _mm256_sub_epi16(*clip_diff, diff_clamp);
|
||||
return diff_const_16;
|
||||
}
|
||||
|
||||
static INLINE void build_compound_diffwtd_mask_d16_avx2(
|
||||
uint8_t *mask, const CONV_BUF_TYPE *src0, int src0_stride,
|
||||
const CONV_BUF_TYPE *src1, int src1_stride, int h, int w, int shift) {
|
||||
const int mask_base = 38;
|
||||
const __m256i _r = _mm256_set1_epi16((1 << shift) >> 1);
|
||||
const __m256i y38 = _mm256_set1_epi16(mask_base);
|
||||
const __m256i y64 = _mm256_set1_epi16(AOM_BLEND_A64_MAX_ALPHA);
|
||||
int i = 0;
|
||||
if (w == 4) {
|
||||
do {
|
||||
const __m128i s0A = xx_loadl_64(src0);
|
||||
const __m128i s0B = xx_loadl_64(src0 + src0_stride);
|
||||
const __m128i s0C = xx_loadl_64(src0 + src0_stride * 2);
|
||||
const __m128i s0D = xx_loadl_64(src0 + src0_stride * 3);
|
||||
const __m128i s1A = xx_loadl_64(src1);
|
||||
const __m128i s1B = xx_loadl_64(src1 + src1_stride);
|
||||
const __m128i s1C = xx_loadl_64(src1 + src1_stride * 2);
|
||||
const __m128i s1D = xx_loadl_64(src1 + src1_stride * 3);
|
||||
const __m256i s0 = yy_set_m128i(_mm_unpacklo_epi64(s0C, s0D),
|
||||
_mm_unpacklo_epi64(s0A, s0B));
|
||||
const __m256i s1 = yy_set_m128i(_mm_unpacklo_epi64(s1C, s1D),
|
||||
_mm_unpacklo_epi64(s1A, s1B));
|
||||
const __m256i m16 = calc_mask_d16_avx2(&s0, &s1, &_r, &y38, &y64, shift);
|
||||
const __m256i m8 = _mm256_packus_epi16(m16, _mm256_setzero_si256());
|
||||
xx_storeu_128(mask,
|
||||
_mm256_castsi256_si128(_mm256_permute4x64_epi64(m8, 0xd8)));
|
||||
src0 += src0_stride << 2;
|
||||
src1 += src1_stride << 2;
|
||||
mask += 16;
|
||||
i += 4;
|
||||
} while (i < h);
|
||||
} else if (w == 8) {
|
||||
do {
|
||||
const __m256i s0AB = yy_loadu2_128(src0 + src0_stride, src0);
|
||||
const __m256i s0CD =
|
||||
yy_loadu2_128(src0 + src0_stride * 3, src0 + src0_stride * 2);
|
||||
const __m256i s1AB = yy_loadu2_128(src1 + src1_stride, src1);
|
||||
const __m256i s1CD =
|
||||
yy_loadu2_128(src1 + src1_stride * 3, src1 + src1_stride * 2);
|
||||
const __m256i m16AB =
|
||||
calc_mask_d16_avx2(&s0AB, &s1AB, &_r, &y38, &y64, shift);
|
||||
const __m256i m16CD =
|
||||
calc_mask_d16_avx2(&s0CD, &s1CD, &_r, &y38, &y64, shift);
|
||||
const __m256i m8 = _mm256_packus_epi16(m16AB, m16CD);
|
||||
yy_storeu_256(mask, _mm256_permute4x64_epi64(m8, 0xd8));
|
||||
src0 += src0_stride << 2;
|
||||
src1 += src1_stride << 2;
|
||||
mask += 32;
|
||||
i += 4;
|
||||
} while (i < h);
|
||||
} else if (w == 16) {
|
||||
do {
|
||||
const __m256i s0A = yy_loadu_256(src0);
|
||||
const __m256i s0B = yy_loadu_256(src0 + src0_stride);
|
||||
const __m256i s1A = yy_loadu_256(src1);
|
||||
const __m256i s1B = yy_loadu_256(src1 + src1_stride);
|
||||
const __m256i m16A =
|
||||
calc_mask_d16_avx2(&s0A, &s1A, &_r, &y38, &y64, shift);
|
||||
const __m256i m16B =
|
||||
calc_mask_d16_avx2(&s0B, &s1B, &_r, &y38, &y64, shift);
|
||||
const __m256i m8 = _mm256_packus_epi16(m16A, m16B);
|
||||
yy_storeu_256(mask, _mm256_permute4x64_epi64(m8, 0xd8));
|
||||
src0 += src0_stride << 1;
|
||||
src1 += src1_stride << 1;
|
||||
mask += 32;
|
||||
i += 2;
|
||||
} while (i < h);
|
||||
} else if (w == 32) {
|
||||
do {
|
||||
const __m256i s0A = yy_loadu_256(src0);
|
||||
const __m256i s0B = yy_loadu_256(src0 + 16);
|
||||
const __m256i s1A = yy_loadu_256(src1);
|
||||
const __m256i s1B = yy_loadu_256(src1 + 16);
|
||||
const __m256i m16A =
|
||||
calc_mask_d16_avx2(&s0A, &s1A, &_r, &y38, &y64, shift);
|
||||
const __m256i m16B =
|
||||
calc_mask_d16_avx2(&s0B, &s1B, &_r, &y38, &y64, shift);
|
||||
const __m256i m8 = _mm256_packus_epi16(m16A, m16B);
|
||||
yy_storeu_256(mask, _mm256_permute4x64_epi64(m8, 0xd8));
|
||||
src0 += src0_stride;
|
||||
src1 += src1_stride;
|
||||
mask += 32;
|
||||
i += 1;
|
||||
} while (i < h);
|
||||
} else if (w == 64) {
|
||||
do {
|
||||
const __m256i s0A = yy_loadu_256(src0);
|
||||
const __m256i s0B = yy_loadu_256(src0 + 16);
|
||||
const __m256i s0C = yy_loadu_256(src0 + 32);
|
||||
const __m256i s0D = yy_loadu_256(src0 + 48);
|
||||
const __m256i s1A = yy_loadu_256(src1);
|
||||
const __m256i s1B = yy_loadu_256(src1 + 16);
|
||||
const __m256i s1C = yy_loadu_256(src1 + 32);
|
||||
const __m256i s1D = yy_loadu_256(src1 + 48);
|
||||
const __m256i m16A =
|
||||
calc_mask_d16_avx2(&s0A, &s1A, &_r, &y38, &y64, shift);
|
||||
const __m256i m16B =
|
||||
calc_mask_d16_avx2(&s0B, &s1B, &_r, &y38, &y64, shift);
|
||||
const __m256i m16C =
|
||||
calc_mask_d16_avx2(&s0C, &s1C, &_r, &y38, &y64, shift);
|
||||
const __m256i m16D =
|
||||
calc_mask_d16_avx2(&s0D, &s1D, &_r, &y38, &y64, shift);
|
||||
const __m256i m8AB = _mm256_packus_epi16(m16A, m16B);
|
||||
const __m256i m8CD = _mm256_packus_epi16(m16C, m16D);
|
||||
yy_storeu_256(mask, _mm256_permute4x64_epi64(m8AB, 0xd8));
|
||||
yy_storeu_256(mask + 32, _mm256_permute4x64_epi64(m8CD, 0xd8));
|
||||
src0 += src0_stride;
|
||||
src1 += src1_stride;
|
||||
mask += 64;
|
||||
i += 1;
|
||||
} while (i < h);
|
||||
} else {
|
||||
do {
|
||||
const __m256i s0A = yy_loadu_256(src0);
|
||||
const __m256i s0B = yy_loadu_256(src0 + 16);
|
||||
const __m256i s0C = yy_loadu_256(src0 + 32);
|
||||
const __m256i s0D = yy_loadu_256(src0 + 48);
|
||||
const __m256i s0E = yy_loadu_256(src0 + 64);
|
||||
const __m256i s0F = yy_loadu_256(src0 + 80);
|
||||
const __m256i s0G = yy_loadu_256(src0 + 96);
|
||||
const __m256i s0H = yy_loadu_256(src0 + 112);
|
||||
const __m256i s1A = yy_loadu_256(src1);
|
||||
const __m256i s1B = yy_loadu_256(src1 + 16);
|
||||
const __m256i s1C = yy_loadu_256(src1 + 32);
|
||||
const __m256i s1D = yy_loadu_256(src1 + 48);
|
||||
const __m256i s1E = yy_loadu_256(src1 + 64);
|
||||
const __m256i s1F = yy_loadu_256(src1 + 80);
|
||||
const __m256i s1G = yy_loadu_256(src1 + 96);
|
||||
const __m256i s1H = yy_loadu_256(src1 + 112);
|
||||
const __m256i m16A =
|
||||
calc_mask_d16_avx2(&s0A, &s1A, &_r, &y38, &y64, shift);
|
||||
const __m256i m16B =
|
||||
calc_mask_d16_avx2(&s0B, &s1B, &_r, &y38, &y64, shift);
|
||||
const __m256i m16C =
|
||||
calc_mask_d16_avx2(&s0C, &s1C, &_r, &y38, &y64, shift);
|
||||
const __m256i m16D =
|
||||
calc_mask_d16_avx2(&s0D, &s1D, &_r, &y38, &y64, shift);
|
||||
const __m256i m16E =
|
||||
calc_mask_d16_avx2(&s0E, &s1E, &_r, &y38, &y64, shift);
|
||||
const __m256i m16F =
|
||||
calc_mask_d16_avx2(&s0F, &s1F, &_r, &y38, &y64, shift);
|
||||
const __m256i m16G =
|
||||
calc_mask_d16_avx2(&s0G, &s1G, &_r, &y38, &y64, shift);
|
||||
const __m256i m16H =
|
||||
calc_mask_d16_avx2(&s0H, &s1H, &_r, &y38, &y64, shift);
|
||||
const __m256i m8AB = _mm256_packus_epi16(m16A, m16B);
|
||||
const __m256i m8CD = _mm256_packus_epi16(m16C, m16D);
|
||||
const __m256i m8EF = _mm256_packus_epi16(m16E, m16F);
|
||||
const __m256i m8GH = _mm256_packus_epi16(m16G, m16H);
|
||||
yy_storeu_256(mask, _mm256_permute4x64_epi64(m8AB, 0xd8));
|
||||
yy_storeu_256(mask + 32, _mm256_permute4x64_epi64(m8CD, 0xd8));
|
||||
yy_storeu_256(mask + 64, _mm256_permute4x64_epi64(m8EF, 0xd8));
|
||||
yy_storeu_256(mask + 96, _mm256_permute4x64_epi64(m8GH, 0xd8));
|
||||
src0 += src0_stride;
|
||||
src1 += src1_stride;
|
||||
mask += 128;
|
||||
i += 1;
|
||||
} while (i < h);
|
||||
}
|
||||
}
|
||||
|
||||
static INLINE void build_compound_diffwtd_mask_d16_inv_avx2(
|
||||
uint8_t *mask, const CONV_BUF_TYPE *src0, int src0_stride,
|
||||
const CONV_BUF_TYPE *src1, int src1_stride, int h, int w, int shift) {
|
||||
const int mask_base = 38;
|
||||
const __m256i _r = _mm256_set1_epi16((1 << shift) >> 1);
|
||||
const __m256i y38 = _mm256_set1_epi16(mask_base);
|
||||
const __m256i y64 = _mm256_set1_epi16(AOM_BLEND_A64_MAX_ALPHA);
|
||||
int i = 0;
|
||||
if (w == 4) {
|
||||
do {
|
||||
const __m128i s0A = xx_loadl_64(src0);
|
||||
const __m128i s0B = xx_loadl_64(src0 + src0_stride);
|
||||
const __m128i s0C = xx_loadl_64(src0 + src0_stride * 2);
|
||||
const __m128i s0D = xx_loadl_64(src0 + src0_stride * 3);
|
||||
const __m128i s1A = xx_loadl_64(src1);
|
||||
const __m128i s1B = xx_loadl_64(src1 + src1_stride);
|
||||
const __m128i s1C = xx_loadl_64(src1 + src1_stride * 2);
|
||||
const __m128i s1D = xx_loadl_64(src1 + src1_stride * 3);
|
||||
const __m256i s0 = yy_set_m128i(_mm_unpacklo_epi64(s0C, s0D),
|
||||
_mm_unpacklo_epi64(s0A, s0B));
|
||||
const __m256i s1 = yy_set_m128i(_mm_unpacklo_epi64(s1C, s1D),
|
||||
_mm_unpacklo_epi64(s1A, s1B));
|
||||
const __m256i m16 =
|
||||
calc_mask_d16_inv_avx2(&s0, &s1, &_r, &y38, &y64, shift);
|
||||
const __m256i m8 = _mm256_packus_epi16(m16, _mm256_setzero_si256());
|
||||
xx_storeu_128(mask,
|
||||
_mm256_castsi256_si128(_mm256_permute4x64_epi64(m8, 0xd8)));
|
||||
src0 += src0_stride << 2;
|
||||
src1 += src1_stride << 2;
|
||||
mask += 16;
|
||||
i += 4;
|
||||
} while (i < h);
|
||||
} else if (w == 8) {
|
||||
do {
|
||||
const __m256i s0AB = yy_loadu2_128(src0 + src0_stride, src0);
|
||||
const __m256i s0CD =
|
||||
yy_loadu2_128(src0 + src0_stride * 3, src0 + src0_stride * 2);
|
||||
const __m256i s1AB = yy_loadu2_128(src1 + src1_stride, src1);
|
||||
const __m256i s1CD =
|
||||
yy_loadu2_128(src1 + src1_stride * 3, src1 + src1_stride * 2);
|
||||
const __m256i m16AB =
|
||||
calc_mask_d16_inv_avx2(&s0AB, &s1AB, &_r, &y38, &y64, shift);
|
||||
const __m256i m16CD =
|
||||
calc_mask_d16_inv_avx2(&s0CD, &s1CD, &_r, &y38, &y64, shift);
|
||||
const __m256i m8 = _mm256_packus_epi16(m16AB, m16CD);
|
||||
yy_storeu_256(mask, _mm256_permute4x64_epi64(m8, 0xd8));
|
||||
src0 += src0_stride << 2;
|
||||
src1 += src1_stride << 2;
|
||||
mask += 32;
|
||||
i += 4;
|
||||
} while (i < h);
|
||||
} else if (w == 16) {
|
||||
do {
|
||||
const __m256i s0A = yy_loadu_256(src0);
|
||||
const __m256i s0B = yy_loadu_256(src0 + src0_stride);
|
||||
const __m256i s1A = yy_loadu_256(src1);
|
||||
const __m256i s1B = yy_loadu_256(src1 + src1_stride);
|
||||
const __m256i m16A =
|
||||
calc_mask_d16_inv_avx2(&s0A, &s1A, &_r, &y38, &y64, shift);
|
||||
const __m256i m16B =
|
||||
calc_mask_d16_inv_avx2(&s0B, &s1B, &_r, &y38, &y64, shift);
|
||||
const __m256i m8 = _mm256_packus_epi16(m16A, m16B);
|
||||
yy_storeu_256(mask, _mm256_permute4x64_epi64(m8, 0xd8));
|
||||
src0 += src0_stride << 1;
|
||||
src1 += src1_stride << 1;
|
||||
mask += 32;
|
||||
i += 2;
|
||||
} while (i < h);
|
||||
} else if (w == 32) {
|
||||
do {
|
||||
const __m256i s0A = yy_loadu_256(src0);
|
||||
const __m256i s0B = yy_loadu_256(src0 + 16);
|
||||
const __m256i s1A = yy_loadu_256(src1);
|
||||
const __m256i s1B = yy_loadu_256(src1 + 16);
|
||||
const __m256i m16A =
|
||||
calc_mask_d16_inv_avx2(&s0A, &s1A, &_r, &y38, &y64, shift);
|
||||
const __m256i m16B =
|
||||
calc_mask_d16_inv_avx2(&s0B, &s1B, &_r, &y38, &y64, shift);
|
||||
const __m256i m8 = _mm256_packus_epi16(m16A, m16B);
|
||||
yy_storeu_256(mask, _mm256_permute4x64_epi64(m8, 0xd8));
|
||||
src0 += src0_stride;
|
||||
src1 += src1_stride;
|
||||
mask += 32;
|
||||
i += 1;
|
||||
} while (i < h);
|
||||
} else if (w == 64) {
|
||||
do {
|
||||
const __m256i s0A = yy_loadu_256(src0);
|
||||
const __m256i s0B = yy_loadu_256(src0 + 16);
|
||||
const __m256i s0C = yy_loadu_256(src0 + 32);
|
||||
const __m256i s0D = yy_loadu_256(src0 + 48);
|
||||
const __m256i s1A = yy_loadu_256(src1);
|
||||
const __m256i s1B = yy_loadu_256(src1 + 16);
|
||||
const __m256i s1C = yy_loadu_256(src1 + 32);
|
||||
const __m256i s1D = yy_loadu_256(src1 + 48);
|
||||
const __m256i m16A =
|
||||
calc_mask_d16_inv_avx2(&s0A, &s1A, &_r, &y38, &y64, shift);
|
||||
const __m256i m16B =
|
||||
calc_mask_d16_inv_avx2(&s0B, &s1B, &_r, &y38, &y64, shift);
|
||||
const __m256i m16C =
|
||||
calc_mask_d16_inv_avx2(&s0C, &s1C, &_r, &y38, &y64, shift);
|
||||
const __m256i m16D =
|
||||
calc_mask_d16_inv_avx2(&s0D, &s1D, &_r, &y38, &y64, shift);
|
||||
const __m256i m8AB = _mm256_packus_epi16(m16A, m16B);
|
||||
const __m256i m8CD = _mm256_packus_epi16(m16C, m16D);
|
||||
yy_storeu_256(mask, _mm256_permute4x64_epi64(m8AB, 0xd8));
|
||||
yy_storeu_256(mask + 32, _mm256_permute4x64_epi64(m8CD, 0xd8));
|
||||
src0 += src0_stride;
|
||||
src1 += src1_stride;
|
||||
mask += 64;
|
||||
i += 1;
|
||||
} while (i < h);
|
||||
} else {
|
||||
do {
|
||||
const __m256i s0A = yy_loadu_256(src0);
|
||||
const __m256i s0B = yy_loadu_256(src0 + 16);
|
||||
const __m256i s0C = yy_loadu_256(src0 + 32);
|
||||
const __m256i s0D = yy_loadu_256(src0 + 48);
|
||||
const __m256i s0E = yy_loadu_256(src0 + 64);
|
||||
const __m256i s0F = yy_loadu_256(src0 + 80);
|
||||
const __m256i s0G = yy_loadu_256(src0 + 96);
|
||||
const __m256i s0H = yy_loadu_256(src0 + 112);
|
||||
const __m256i s1A = yy_loadu_256(src1);
|
||||
const __m256i s1B = yy_loadu_256(src1 + 16);
|
||||
const __m256i s1C = yy_loadu_256(src1 + 32);
|
||||
const __m256i s1D = yy_loadu_256(src1 + 48);
|
||||
const __m256i s1E = yy_loadu_256(src1 + 64);
|
||||
const __m256i s1F = yy_loadu_256(src1 + 80);
|
||||
const __m256i s1G = yy_loadu_256(src1 + 96);
|
||||
const __m256i s1H = yy_loadu_256(src1 + 112);
|
||||
const __m256i m16A =
|
||||
calc_mask_d16_inv_avx2(&s0A, &s1A, &_r, &y38, &y64, shift);
|
||||
const __m256i m16B =
|
||||
calc_mask_d16_inv_avx2(&s0B, &s1B, &_r, &y38, &y64, shift);
|
||||
const __m256i m16C =
|
||||
calc_mask_d16_inv_avx2(&s0C, &s1C, &_r, &y38, &y64, shift);
|
||||
const __m256i m16D =
|
||||
calc_mask_d16_inv_avx2(&s0D, &s1D, &_r, &y38, &y64, shift);
|
||||
const __m256i m16E =
|
||||
calc_mask_d16_inv_avx2(&s0E, &s1E, &_r, &y38, &y64, shift);
|
||||
const __m256i m16F =
|
||||
calc_mask_d16_inv_avx2(&s0F, &s1F, &_r, &y38, &y64, shift);
|
||||
const __m256i m16G =
|
||||
calc_mask_d16_inv_avx2(&s0G, &s1G, &_r, &y38, &y64, shift);
|
||||
const __m256i m16H =
|
||||
calc_mask_d16_inv_avx2(&s0H, &s1H, &_r, &y38, &y64, shift);
|
||||
const __m256i m8AB = _mm256_packus_epi16(m16A, m16B);
|
||||
const __m256i m8CD = _mm256_packus_epi16(m16C, m16D);
|
||||
const __m256i m8EF = _mm256_packus_epi16(m16E, m16F);
|
||||
const __m256i m8GH = _mm256_packus_epi16(m16G, m16H);
|
||||
yy_storeu_256(mask, _mm256_permute4x64_epi64(m8AB, 0xd8));
|
||||
yy_storeu_256(mask + 32, _mm256_permute4x64_epi64(m8CD, 0xd8));
|
||||
yy_storeu_256(mask + 64, _mm256_permute4x64_epi64(m8EF, 0xd8));
|
||||
yy_storeu_256(mask + 96, _mm256_permute4x64_epi64(m8GH, 0xd8));
|
||||
src0 += src0_stride;
|
||||
src1 += src1_stride;
|
||||
mask += 128;
|
||||
i += 1;
|
||||
} while (i < h);
|
||||
}
|
||||
}
|
||||
|
||||
void av1_build_compound_diffwtd_mask_d16_avx2(
|
||||
uint8_t *mask, DIFFWTD_MASK_TYPE mask_type, const CONV_BUF_TYPE *src0,
|
||||
int src0_stride, const CONV_BUF_TYPE *src1, int src1_stride, int h, int w,
|
||||
ConvolveParams *conv_params, int bd) {
|
||||
const int shift =
|
||||
2 * FILTER_BITS - conv_params->round_0 - conv_params->round_1 + (bd - 8);
|
||||
// When rounding constant is added, there is a possibility of overflow.
|
||||
// However that much precision is not required. Code should very well work for
|
||||
// other values of DIFF_FACTOR_LOG2 and AOM_BLEND_A64_MAX_ALPHA as well. But
|
||||
// there is a possibility of corner case bugs.
|
||||
assert(DIFF_FACTOR_LOG2 == 4);
|
||||
assert(AOM_BLEND_A64_MAX_ALPHA == 64);
|
||||
|
||||
if (mask_type == DIFFWTD_38) {
|
||||
build_compound_diffwtd_mask_d16_avx2(mask, src0, src0_stride, src1,
|
||||
src1_stride, h, w, shift);
|
||||
} else {
|
||||
build_compound_diffwtd_mask_d16_inv_avx2(mask, src0, src0_stride, src1,
|
||||
src1_stride, h, w, shift);
|
||||
}
|
||||
}
|
||||
|
||||
void av1_build_compound_diffwtd_mask_highbd_avx2(
|
||||
uint8_t *mask, DIFFWTD_MASK_TYPE mask_type, const uint8_t *src0,
|
||||
int src0_stride, const uint8_t *src1, int src1_stride, int h, int w,
|
||||
|
|
|
|||
23
third_party/aom/av1/common/x86/selfguided_avx2.c
vendored
23
third_party/aom/av1/common/x86/selfguided_avx2.c
vendored
|
|
@ -546,17 +546,18 @@ static void final_filter_fast(int32_t *dst, int dst_stride, const int32_t *A,
|
|||
}
|
||||
}
|
||||
|
||||
void av1_selfguided_restoration_avx2(const uint8_t *dgd8, int width, int height,
|
||||
int dgd_stride, int32_t *flt0,
|
||||
int32_t *flt1, int flt_stride,
|
||||
int sgr_params_idx, int bit_depth,
|
||||
int highbd) {
|
||||
int av1_selfguided_restoration_avx2(const uint8_t *dgd8, int width, int height,
|
||||
int dgd_stride, int32_t *flt0,
|
||||
int32_t *flt1, int flt_stride,
|
||||
int sgr_params_idx, int bit_depth,
|
||||
int highbd) {
|
||||
// The ALIGN_POWER_OF_TWO macro here ensures that column 1 of Atl, Btl,
|
||||
// Ctl and Dtl is 32-byte aligned.
|
||||
const int buf_elts = ALIGN_POWER_OF_TWO(RESTORATION_PROC_UNIT_PELS, 3);
|
||||
|
||||
DECLARE_ALIGNED(32, int32_t,
|
||||
buf[4 * ALIGN_POWER_OF_TWO(RESTORATION_PROC_UNIT_PELS, 3)]);
|
||||
int32_t *buf = aom_memalign(
|
||||
32, 4 * sizeof(*buf) * ALIGN_POWER_OF_TWO(RESTORATION_PROC_UNIT_PELS, 3));
|
||||
if (!buf) return -1;
|
||||
|
||||
const int width_ext = width + 2 * SGRPROJ_BORDER_HORZ;
|
||||
const int height_ext = height + 2 * SGRPROJ_BORDER_VERT;
|
||||
|
|
@ -625,6 +626,8 @@ void av1_selfguided_restoration_avx2(const uint8_t *dgd8, int width, int height,
|
|||
final_filter(flt1, flt_stride, A, B, buf_stride, dgd8, dgd_stride, width,
|
||||
height, highbd);
|
||||
}
|
||||
aom_free(buf);
|
||||
return 0;
|
||||
}
|
||||
|
||||
void apply_selfguided_restoration_avx2(const uint8_t *dat8, int width,
|
||||
|
|
@ -635,8 +638,10 @@ void apply_selfguided_restoration_avx2(const uint8_t *dat8, int width,
|
|||
int32_t *flt0 = tmpbuf;
|
||||
int32_t *flt1 = flt0 + RESTORATION_UNITPELS_MAX;
|
||||
assert(width * height <= RESTORATION_UNITPELS_MAX);
|
||||
av1_selfguided_restoration_avx2(dat8, width, height, stride, flt0, flt1,
|
||||
width, eps, bit_depth, highbd);
|
||||
const int ret = av1_selfguided_restoration_avx2(
|
||||
dat8, width, height, stride, flt0, flt1, width, eps, bit_depth, highbd);
|
||||
(void)ret;
|
||||
assert(!ret);
|
||||
const sgr_params_type *const params = &sgr_params[eps];
|
||||
int xq[2];
|
||||
decode_xq(xqd, xq, params);
|
||||
|
|
|
|||
24
third_party/aom/av1/common/x86/selfguided_sse4.c
vendored
24
third_party/aom/av1/common/x86/selfguided_sse4.c
vendored
|
|
@ -499,13 +499,15 @@ static void final_filter_fast(int32_t *dst, int dst_stride, const int32_t *A,
|
|||
}
|
||||
}
|
||||
|
||||
void av1_selfguided_restoration_sse4_1(const uint8_t *dgd8, int width,
|
||||
int height, int dgd_stride,
|
||||
int32_t *flt0, int32_t *flt1,
|
||||
int flt_stride, int sgr_params_idx,
|
||||
int bit_depth, int highbd) {
|
||||
DECLARE_ALIGNED(16, int32_t, buf[4 * RESTORATION_PROC_UNIT_PELS]);
|
||||
memset(buf, 0, sizeof(buf));
|
||||
int av1_selfguided_restoration_sse4_1(const uint8_t *dgd8, int width,
|
||||
int height, int dgd_stride, int32_t *flt0,
|
||||
int32_t *flt1, int flt_stride,
|
||||
int sgr_params_idx, int bit_depth,
|
||||
int highbd) {
|
||||
int32_t *buf = (int32_t *)aom_memalign(
|
||||
16, 4 * sizeof(*buf) * RESTORATION_PROC_UNIT_PELS);
|
||||
if (!buf) return -1;
|
||||
memset(buf, 0, 4 * sizeof(*buf) * RESTORATION_PROC_UNIT_PELS);
|
||||
|
||||
const int width_ext = width + 2 * SGRPROJ_BORDER_HORZ;
|
||||
const int height_ext = height + 2 * SGRPROJ_BORDER_VERT;
|
||||
|
|
@ -574,6 +576,8 @@ void av1_selfguided_restoration_sse4_1(const uint8_t *dgd8, int width,
|
|||
final_filter(flt1, flt_stride, A, B, buf_stride, dgd8, dgd_stride, width,
|
||||
height, highbd);
|
||||
}
|
||||
aom_free(buf);
|
||||
return 0;
|
||||
}
|
||||
|
||||
void apply_selfguided_restoration_sse4_1(const uint8_t *dat8, int width,
|
||||
|
|
@ -584,8 +588,10 @@ void apply_selfguided_restoration_sse4_1(const uint8_t *dat8, int width,
|
|||
int32_t *flt0 = tmpbuf;
|
||||
int32_t *flt1 = flt0 + RESTORATION_UNITPELS_MAX;
|
||||
assert(width * height <= RESTORATION_UNITPELS_MAX);
|
||||
av1_selfguided_restoration_sse4_1(dat8, width, height, stride, flt0, flt1,
|
||||
width, eps, bit_depth, highbd);
|
||||
const int ret = av1_selfguided_restoration_sse4_1(
|
||||
dat8, width, height, stride, flt0, flt1, width, eps, bit_depth, highbd);
|
||||
(void)ret;
|
||||
assert(!ret);
|
||||
const sgr_params_type *const params = &sgr_params[eps];
|
||||
int xq[2];
|
||||
decode_xq(xqd, xq, params);
|
||||
|
|
|
|||
Some files were not shown because too many files have changed in this diff Show more
Loading…
Add table
Add a link
Reference in a new issue