Update libaom to commit ID 1e227d41f0616de9548a673a83a21ef990b62591

This commit is contained in:
trav90 2018-10-19 23:05:00 -05:00 • committed by Roy Tam
commit 368651059a
526 changed files with 34535 additions and 15900 deletions

View file

@ -53,6 +53,8 @@ list(APPEND AOM_AV1_COMMON_SOURCES
"${AOM_ROOT}/av1/common/mv.h"
"${AOM_ROOT}/av1/common/mvref_common.c"
"${AOM_ROOT}/av1/common/mvref_common.h"
"${AOM_ROOT}/av1/common/obu_util.c"
"${AOM_ROOT}/av1/common/obu_util.h"
"${AOM_ROOT}/av1/common/odintrin.c"
"${AOM_ROOT}/av1/common/odintrin.h"
"${AOM_ROOT}/av1/common/onyxc_int.h"
@ -78,8 +80,8 @@ list(APPEND AOM_AV1_COMMON_SOURCES
"${AOM_ROOT}/av1/common/thread_common.h"
"${AOM_ROOT}/av1/common/tile_common.c"
"${AOM_ROOT}/av1/common/tile_common.h"
"${AOM_ROOT}/av1/common/timing.h"
"${AOM_ROOT}/av1/common/timing.c"
"${AOM_ROOT}/av1/common/timing.h"
"${AOM_ROOT}/av1/common/token_cdfs.h"
"${AOM_ROOT}/av1/common/txb_common.c"
"${AOM_ROOT}/av1/common/txb_common.h"
@ -176,6 +178,8 @@ list(APPEND AOM_AV1_ENCODER_SOURCES
"${AOM_ROOT}/av1/encoder/rd.h"
"${AOM_ROOT}/av1/encoder/rdopt.c"
"${AOM_ROOT}/av1/encoder/rdopt.h"
"${AOM_ROOT}/av1/encoder/reconinter_enc.c"
"${AOM_ROOT}/av1/encoder/reconinter_enc.h"
"${AOM_ROOT}/av1/encoder/segmentation.c"
"${AOM_ROOT}/av1/encoder/segmentation.h"
"${AOM_ROOT}/av1/encoder/speed_features.c"
@ -268,7 +272,8 @@ list(APPEND AOM_AV1_ENCODER_INTRIN_SSE4_1
"${AOM_ROOT}/av1/encoder/x86/av1_highbd_quantize_sse4.c"
"${AOM_ROOT}/av1/encoder/x86/corner_match_sse4.c"
"${AOM_ROOT}/av1/encoder/x86/encodetxb_sse4.c"
"${AOM_ROOT}/av1/encoder/x86/highbd_fwd_txfm_sse4.c")
"${AOM_ROOT}/av1/encoder/x86/highbd_fwd_txfm_sse4.c"
"${AOM_ROOT}/av1/encoder/x86/pickrst_sse4.c")
list(APPEND AOM_AV1_ENCODER_INTRIN_AVX2
"${AOM_ROOT}/av1/encoder/x86/av1_quantize_avx2.c"
@ -276,7 +281,9 @@ list(APPEND AOM_AV1_ENCODER_INTRIN_AVX2
"${AOM_ROOT}/av1/encoder/x86/error_intrin_avx2.c"
"${AOM_ROOT}/av1/encoder/x86/av1_fwd_txfm_avx2.h"
"${AOM_ROOT}/av1/encoder/x86/av1_fwd_txfm2d_avx2.c"
"${AOM_ROOT}/av1/encoder/x86/wedge_utils_avx2.c")
"${AOM_ROOT}/av1/encoder/x86/wedge_utils_avx2.c"
"${AOM_ROOT}/av1/encoder/x86/encodetxb_avx2.c"
"${AOM_ROOT}/av1/encoder/x86/pickrst_avx2.c")
list(APPEND AOM_AV1_ENCODER_INTRIN_NEON
"${AOM_ROOT}/av1/encoder/arm/neon/quantize_neon.c")
@ -301,6 +308,7 @@ list(APPEND AOM_AV1_COMMON_INTRIN_NEON
"${AOM_ROOT}/av1/common/arm/selfguided_neon.c"
"${AOM_ROOT}/av1/common/arm/av1_inv_txfm_neon.c"
"${AOM_ROOT}/av1/common/arm/av1_inv_txfm_neon.h"
"${AOM_ROOT}/av1/common/arm/warp_plane_neon.c"
"${AOM_ROOT}/av1/common/cdef_block_neon.c")
list(APPEND AOM_AV1_ENCODER_INTRIN_SSE4_2

View file

@ -14,28 +14,29 @@
#include "config/aom_config.h"
#include "config/aom_version.h"
#include "aom/aom_encoder.h"
#include "aom_ports/aom_once.h"
#include "aom_ports/mem_ops.h"
#include "aom_ports/system_state.h"
#include "aom/aom_encoder.h"
#include "aom/internal/aom_codec_internal.h"
#include "av1/encoder/encoder.h"
#include "aom/aomcx.h"
#include "av1/encoder/firstpass.h"
#include "av1/av1_iface_common.h"
#include "av1/encoder/bitstream.h"
#include "aom_ports/mem_ops.h"
#include "av1/encoder/encoder.h"
#include "av1/encoder/firstpass.h"
#define MAG_SIZE (4)
#define MAX_NUM_ENHANCEMENT_LAYERS 3
struct av1_extracfg {
int cpu_used; // available cpu percentage in 1/16
int dev_sf;
unsigned int enable_auto_alt_ref;
unsigned int enable_auto_bwd_ref;
unsigned int noise_sensitivity;
unsigned int sharpness;
unsigned int static_thresh;
unsigned int row_mt;
unsigned int tile_columns; // log2 number of tile columns
unsigned int tile_rows; // log2 number of tile rows
unsigned int arnr_max_frames;
@ -98,37 +99,40 @@ struct av1_extracfg {
float noise_level;
int noise_block_size;
#endif
unsigned int chroma_subsampling_x;
unsigned int chroma_subsampling_y;
};
static struct av1_extracfg default_extra_cfg = {
0, // cpu_used
0, // dev_sf
1, // enable_auto_alt_ref
0, // enable_auto_bwd_ref
0, // noise_sensitivity
0, // sharpness
0, // static_thresh
0, // tile_columns
0, // tile_rows
7, // arnr_max_frames
5, // arnr_strength
0, // min_gf_interval; 0 -> default decision
0, // max_gf_interval; 0 -> default decision
AOM_TUNE_PSNR, // tuning
10, // cq_level
0, // rc_max_intra_bitrate_pct
0, // rc_max_inter_bitrate_pct
0, // gf_cbr_boost_pct
0, // lossless
1, // enable_cdef
1, // enable_restoration
0, // disable_trellis_quant
0, // enable_qm
DEFAULT_QM_Y, // qm_y
DEFAULT_QM_U, // qm_u
DEFAULT_QM_V, // qm_v
DEFAULT_QM_FIRST, // qm_min
DEFAULT_QM_LAST, // qm_max
0, // cpu_used
1, // enable_auto_alt_ref
0, // enable_auto_bwd_ref
0, // noise_sensitivity
CONFIG_SHARP_SETTINGS, // sharpness
0, // static_thresh
0, // row_mt
0, // tile_columns
0, // tile_rows
7, // arnr_max_frames
5, // arnr_strength
0, // min_gf_interval; 0 -> default decision
0, // max_gf_interval; 0 -> default decision
AOM_TUNE_PSNR, // tuning
10, // cq_level
0, // rc_max_intra_bitrate_pct
0, // rc_max_inter_bitrate_pct
0, // gf_cbr_boost_pct
0, // lossless
!CONFIG_SHARP_SETTINGS, // enable_cdef
1, // enable_restoration
0, // disable_trellis_quant
0, // enable_qm
DEFAULT_QM_Y, // qm_y
DEFAULT_QM_U, // qm_u
DEFAULT_QM_V, // qm_v
DEFAULT_QM_FIRST, // qm_min
DEFAULT_QM_LAST, // qm_max
#if CONFIG_DIST_8X8
0,
#endif
@ -150,7 +154,7 @@ static struct av1_extracfg default_extra_cfg = {
0, // render width
0, // render height
AOM_SUPERBLOCK_SIZE_DYNAMIC, // superblock_size
0, // Single tile decoding is off by default.
1, // this depends on large_scale_tile.
0, // error_resilient_mode off by default.
0, // s_frame_mode off by default.
0, // film_grain_test_vector
@ -168,6 +172,8 @@ static struct av1_extracfg default_extra_cfg = {
0, // noise_level
32, // noise_block_size
#endif
0, // chroma_subsampling_x
0, // chroma_subsampling_y
};
struct aom_codec_alg_priv {
@ -251,10 +257,7 @@ static aom_codec_err_t validate_config(aom_codec_alg_priv_t *ctx,
RANGE_CHECK_HI(extra_cfg, min_gf_interval, MAX_LAG_BUFFERS - 1);
RANGE_CHECK_HI(extra_cfg, max_gf_interval, MAX_LAG_BUFFERS - 1);
if (extra_cfg->max_gf_interval > 0) {
RANGE_CHECK(extra_cfg, max_gf_interval, 2, (MAX_LAG_BUFFERS - 1));
}
if (extra_cfg->min_gf_interval > 0 && extra_cfg->max_gf_interval > 0) {
RANGE_CHECK(extra_cfg, max_gf_interval, extra_cfg->min_gf_interval,
RANGE_CHECK(extra_cfg, max_gf_interval, MAX(2, extra_cfg->min_gf_interval),
(MAX_LAG_BUFFERS - 1));
}
@ -284,13 +287,14 @@ static aom_codec_err_t validate_config(aom_codec_alg_priv_t *ctx,
RANGE_CHECK_HI(extra_cfg, enable_auto_alt_ref, 2);
RANGE_CHECK_HI(extra_cfg, enable_auto_bwd_ref, 2);
RANGE_CHECK(extra_cfg, cpu_used, 0, 8);
RANGE_CHECK(extra_cfg, dev_sf, 0, UINT8_MAX);
RANGE_CHECK_HI(extra_cfg, noise_sensitivity, 6);
RANGE_CHECK(extra_cfg, superblock_size, AOM_SUPERBLOCK_SIZE_64X64,
AOM_SUPERBLOCK_SIZE_DYNAMIC);
RANGE_CHECK_HI(cfg, large_scale_tile, 1);
RANGE_CHECK_HI(extra_cfg, single_tile_decoding, 1);
RANGE_CHECK_HI(extra_cfg, row_mt, 1);
RANGE_CHECK_HI(extra_cfg, tile_columns, 6);
RANGE_CHECK_HI(extra_cfg, tile_rows, 6);
@ -372,6 +376,9 @@ static aom_codec_err_t validate_config(aom_codec_alg_priv_t *ctx,
#endif
}
RANGE_CHECK_HI(extra_cfg, chroma_subsampling_x, 1);
RANGE_CHECK_HI(extra_cfg, chroma_subsampling_y, 1);
return AOM_CODEC_OK;
}
@ -581,7 +588,6 @@ static aom_codec_err_t set_encoder_config(
oxcf->sframe_mode = cfg->sframe_mode;
oxcf->sframe_enabled = cfg->sframe_dist != 0;
oxcf->speed = extra_cfg->cpu_used;
oxcf->dev_sf = extra_cfg->dev_sf;
oxcf->enable_auto_arf = extra_cfg->enable_auto_alt_ref;
oxcf->enable_auto_brf = extra_cfg->enable_auto_bwd_ref;
oxcf->noise_sensitivity = extra_cfg->noise_sensitivity;
@ -637,6 +643,8 @@ static aom_codec_err_t set_encoder_config(
oxcf->superblock_size = AOM_SUPERBLOCK_SIZE_64X64;
}
oxcf->row_mt = extra_cfg->row_mt;
oxcf->tile_columns = extra_cfg->tile_columns;
oxcf->tile_rows = extra_cfg->tile_rows;
@ -692,6 +700,24 @@ static aom_codec_err_t set_encoder_config(
oxcf->frame_periodic_boost = extra_cfg->frame_periodic_boost;
oxcf->motion_vector_unit_test = extra_cfg->motion_vector_unit_test;
#if CONFIG_REDUCED_ENCODER_BORDER
if (oxcf->superres_mode != SUPERRES_NONE ||
oxcf->resize_mode != RESIZE_NONE) {
warn(
"Superres / resize cannot be used with CONFIG_REDUCED_ENCODER_BORDER. "
"Disabling superres/resize.\n");
// return AOM_CODEC_INVALID_PARAM;
disable_superres(oxcf);
oxcf->resize_mode = RESIZE_NONE;
oxcf->resize_scale_denominator = SCALE_NUMERATOR;
oxcf->resize_kf_scale_denominator = SCALE_NUMERATOR;
}
#endif // CONFIG_REDUCED_ENCODER_BORDER
oxcf->chroma_subsampling_x = extra_cfg->chroma_subsampling_x;
oxcf->chroma_subsampling_y = extra_cfg->chroma_subsampling_y;
return AOM_CODEC_OK;
}
@ -731,6 +757,10 @@ static aom_codec_err_t encoder_set_config(aom_codec_alg_priv_t *ctx,
return res;
}
static aom_fixed_buf_t *encoder_get_global_headers(aom_codec_alg_priv_t *ctx) {
return av1_get_global_headers(ctx->cpi);
}
static aom_codec_err_t ctrl_get_quantizer(aom_codec_alg_priv_t *ctx,
va_list args) {
int *const arg = va_arg(args, int *);
@ -765,12 +795,6 @@ static aom_codec_err_t ctrl_set_cpuused(aom_codec_alg_priv_t *ctx,
return update_extra_cfg(ctx, &extra_cfg);
}
static aom_codec_err_t ctrl_set_devsf(aom_codec_alg_priv_t *ctx, va_list args) {
struct av1_extracfg extra_cfg = ctx->extra_cfg;
extra_cfg.dev_sf = CAST(AOME_SET_DEVSF, args);
return update_extra_cfg(ctx, &extra_cfg);
}
static aom_codec_err_t ctrl_set_enable_auto_alt_ref(aom_codec_alg_priv_t *ctx,
va_list args) {
struct av1_extracfg extra_cfg = ctx->extra_cfg;
@ -806,6 +830,13 @@ static aom_codec_err_t ctrl_set_static_thresh(aom_codec_alg_priv_t *ctx,
return update_extra_cfg(ctx, &extra_cfg);
}
static aom_codec_err_t ctrl_set_row_mt(aom_codec_alg_priv_t *ctx,
va_list args) {
struct av1_extracfg extra_cfg = ctx->extra_cfg;
extra_cfg.row_mt = CAST(AV1E_SET_ROW_MT, args);
return update_extra_cfg(ctx, &extra_cfg);
}
static aom_codec_err_t ctrl_set_tile_columns(aom_codec_alg_priv_t *ctx,
va_list args) {
struct av1_extracfg extra_cfg = ctx->extra_cfg;
@ -1669,6 +1700,20 @@ static aom_codec_err_t ctrl_set_superblock_size(aom_codec_alg_priv_t *ctx,
return update_extra_cfg(ctx, &extra_cfg);
}
static aom_codec_err_t ctrl_set_chroma_subsampling_x(aom_codec_alg_priv_t *ctx,
va_list args) {
struct av1_extracfg extra_cfg = ctx->extra_cfg;
extra_cfg.chroma_subsampling_x = CAST(AV1E_SET_CHROMA_SUBSAMPLING_X, args);
return update_extra_cfg(ctx, &extra_cfg);
}
static aom_codec_err_t ctrl_set_chroma_subsampling_y(aom_codec_alg_priv_t *ctx,
va_list args) {
struct av1_extracfg extra_cfg = ctx->extra_cfg;
extra_cfg.chroma_subsampling_y = CAST(AV1E_SET_CHROMA_SUBSAMPLING_Y, args);
return update_extra_cfg(ctx, &extra_cfg);
}
static aom_codec_ctrl_fn_map_t encoder_ctrl_maps[] = {
{ AV1_COPY_REFERENCE, ctrl_copy_reference },
{ AOME_USE_REFERENCE, ctrl_use_reference },
@ -1681,11 +1726,11 @@ static aom_codec_ctrl_fn_map_t encoder_ctrl_maps[] = {
{ AOME_SET_SCALEMODE, ctrl_set_scale_mode },
{ AOME_SET_SPATIAL_LAYER_ID, ctrl_set_spatial_layer_id },
{ AOME_SET_CPUUSED, ctrl_set_cpuused },
{ AOME_SET_DEVSF, ctrl_set_devsf },
{ AOME_SET_ENABLEAUTOALTREF, ctrl_set_enable_auto_alt_ref },
{ AOME_SET_ENABLEAUTOBWDREF, ctrl_set_enable_auto_bwd_ref },
{ AOME_SET_SHARPNESS, ctrl_set_sharpness },
{ AOME_SET_STATIC_THRESHOLD, ctrl_set_static_thresh },
{ AV1E_SET_ROW_MT, ctrl_set_row_mt },
{ AV1E_SET_TILE_COLUMNS, ctrl_set_tile_columns },
{ AV1E_SET_TILE_ROWS, ctrl_set_tile_rows },
{ AOME_SET_ARNR_MAXFRAMES, ctrl_set_arnr_max_frames },
@ -1754,7 +1799,8 @@ static aom_codec_ctrl_fn_map_t encoder_ctrl_maps[] = {
{ AV1E_GET_ACTIVEMAP, ctrl_get_active_map },
{ AV1_GET_NEW_FRAME_IMAGE, ctrl_get_new_frame_image },
{ AV1_COPY_NEW_FRAME_IMAGE, ctrl_copy_new_frame_image },
{ AV1E_SET_CHROMA_SUBSAMPLING_X, ctrl_set_chroma_subsampling_x },
{ AV1E_SET_CHROMA_SUBSAMPLING_Y, ctrl_set_chroma_subsampling_y },
{ -1, NULL },
};
@ -1850,13 +1896,13 @@ CODEC_INTERFACE(aom_codec_av1_cx) = {
},
{
// NOLINT
1, // 1 cfg map
encoder_usage_cfg_map, // aom_codec_enc_cfg_map_t
encoder_encode, // aom_codec_encode_fn_t
encoder_get_cxdata, // aom_codec_get_cx_data_fn_t
encoder_set_config, // aom_codec_enc_config_set_fn_t
NULL, // aom_codec_get_global_headers_fn_t
encoder_get_preview, // aom_codec_get_preview_frame_fn_t
NULL // aom_codec_enc_mr_get_mem_loc_fn_t
1, // 1 cfg map
encoder_usage_cfg_map, // aom_codec_enc_cfg_map_t
encoder_encode, // aom_codec_encode_fn_t
encoder_get_cxdata, // aom_codec_get_cx_data_fn_t
encoder_set_config, // aom_codec_enc_config_set_fn_t
encoder_get_global_headers, // aom_codec_get_global_headers_fn_t
encoder_get_preview, // aom_codec_get_preview_frame_fn_t
NULL // aom_codec_enc_mr_get_mem_loc_fn_t
}
};

View file

@ -26,6 +26,7 @@
#include "av1/common/alloccommon.h"
#include "av1/common/frame_buffers.h"
#include "av1/common/enums.h"
#include "av1/common/obu_util.h"
#include "av1/decoder/decoder.h"
#include "av1/decoder/decodeframe.h"
@ -46,6 +47,7 @@ struct aom_codec_alg_priv {
int last_show_frame; // Index of last output frame.
int byte_alignment;
int skip_loop_filter;
int skip_film_grain;
int decode_tile_row;
int decode_tile_col;
unsigned int tile_mode;
@ -103,6 +105,15 @@ static aom_codec_err_t decoder_init(aom_codec_ctx_t *ctx,
priv->cfg.cfg.ext_partition = 1;
}
av1_zero(priv->image_with_grain);
// Turn row_mt on by default.
priv->row_mt = 1;
// Turn on normal tile coding mode by default.
// 0 is for normal tile coding mode, and 1 is for large scale tile coding
// mode(refer to lightfield example).
priv->tile_mode = 0;
priv->decode_tile_row = -1;
priv->decode_tile_col = -1;
}
return AOM_CODEC_OK;
@ -216,7 +227,7 @@ static aom_codec_err_t decoder_peek_si_internal(const uint8_t *data,
while (1) {
data += bytes_read;
data_sz -= bytes_read;
const uint8_t *payload_start = data;
if (data_sz < payload_size) return AOM_CODEC_CORRUPT_FRAME;
// Check that the selected OBU is a sequence header
if (obu_header.type == OBU_SEQUENCE_HEADER) {
// Sanity check on sequence header size
@ -264,9 +275,9 @@ static aom_codec_err_t decoder_peek_si_internal(const uint8_t *data,
}
}
// skip past any unread OBU header data
data = payload_start + payload_size;
data += payload_size;
data_sz -= payload_size;
if (data_sz <= 0) break; // exit if we're out of OBUs
if (data_sz == 0) break; // exit if we're out of OBUs
status = aom_read_obu_header_and_size(
data, data_sz, si->is_annexb, &obu_header, &payload_size, &bytes_read);
if (status != AOM_CODEC_OK) return status;
@ -313,6 +324,7 @@ static void init_buffer_callbacks(aom_codec_alg_priv_t *ctx) {
cm->new_fb_idx = INVALID_IDX;
cm->byte_alignment = ctx->byte_alignment;
cm->skip_loop_filter = ctx->skip_loop_filter;
cm->skip_film_grain = ctx->skip_film_grain;
if (ctx->get_ext_fb_cb != NULL && ctx->release_ext_fb_cb != NULL) {
pool->get_fb_cb = ctx->get_ext_fb_cb;
@ -434,7 +446,7 @@ static aom_codec_err_t init_decoder(aom_codec_alg_priv_t *ctx) {
frame_worker_data->pbi->ext_tile_debug = ctx->ext_tile_debug;
frame_worker_data->pbi->row_mt = ctx->row_mt;
worker->hook = (AVxWorkerHook)frame_worker_hook;
worker->hook = frame_worker_hook;
if (!winterface->reset(worker)) {
set_error_detail(ctx, "Frame Worker thread creation failed");
return AOM_CODEC_MEM_ERROR;
@ -515,12 +527,11 @@ static aom_codec_err_t decode_one(aom_codec_alg_priv_t *ctx,
static aom_codec_err_t decoder_decode(aom_codec_alg_priv_t *ctx,
const uint8_t *data, size_t data_sz,
void *user_priv) {
const uint8_t *data_start = data;
const uint8_t *data_end = data + data_sz;
aom_codec_err_t res = AOM_CODEC_OK;
// Release any pending output frames from the previous decoder call.
// We need to do this even if the decoder is being flushed
// Release any pending output frames from the previous decoder_decode call.
// We need to do this even if the decoder is being flushed or the input
// arguments are invalid.
if (ctx->frame_workers) {
BufferPool *const pool = ctx->buffer_pool;
RefCntBuffer *const frame_bufs = pool->frame_bufs;
@ -538,10 +549,13 @@ static aom_codec_err_t decoder_decode(aom_codec_alg_priv_t *ctx,
unlock_buffer_pool(ctx->buffer_pool);
}
/* Sanity checks */
/* NULL data ptr allowed if data_sz is 0 too */
if (data == NULL && data_sz == 0) {
ctx->flushed = 1;
return AOM_CODEC_OK;
}
if (data == NULL || data_sz == 0) return AOM_CODEC_INVALID_PARAM;
// Reset flushed when receiving a valid frame.
ctx->flushed = 0;
@ -552,6 +566,9 @@ static aom_codec_err_t decoder_decode(aom_codec_alg_priv_t *ctx,
if (res != AOM_CODEC_OK) return res;
}
const uint8_t *data_start = data;
const uint8_t *data_end = data + data_sz;
if (ctx->is_annexb) {
// read the size of this temporal unit
size_t length_of_size;
@ -617,6 +634,7 @@ static aom_image_t *add_grain_if_needed(aom_image_t *img,
img->fmt != grain_img_buf->fmt) {
aom_img_free(grain_img_buf);
grain_img_buf = NULL;
*grain_img_ptr = NULL;
}
}
if (!grain_img_buf) {
@ -624,7 +642,14 @@ static aom_image_t *add_grain_if_needed(aom_image_t *img,
*grain_img_ptr = grain_img_buf;
}
av1_add_film_grain(grain_params, img, grain_img_buf);
if (grain_img_buf) {
grain_img_buf->user_priv = img->user_priv;
if (av1_add_film_grain(grain_params, img, grain_img_buf)) {
aom_img_free(grain_img_buf);
grain_img_buf = NULL;
*grain_img_ptr = NULL;
}
}
return grain_img_buf;
}
@ -720,8 +745,13 @@ static aom_image_t *decoder_get_frame(aom_codec_alg_priv_t *ctx,
img = &ctx->img;
img->temporal_id = cm->temporal_layer_id;
img->spatial_id = cm->spatial_layer_id;
if (cm->skip_film_grain) grain_params->apply_grain = 0;
aom_image_t *res = add_grain_if_needed(
img, &ctx->image_with_grain[*index], grain_params);
if (!res) {
aom_internal_error(&pbi->common.error, AOM_CODEC_CORRUPT_FRAME,
"Grain systhesis failed\n");
}
*index += 1; // Advance the iterator to point to the next image
return res;
}
@ -1128,6 +1158,19 @@ static aom_codec_err_t ctrl_set_skip_loop_filter(aom_codec_alg_priv_t *ctx,
return AOM_CODEC_OK;
}
static aom_codec_err_t ctrl_set_skip_film_grain(aom_codec_alg_priv_t *ctx,
va_list args) {
ctx->skip_film_grain = va_arg(args, int);
if (ctx->frame_workers) {
AVxWorker *const worker = ctx->frame_workers;
FrameWorkerData *const frame_worker_data = (FrameWorkerData *)worker->data1;
frame_worker_data->pbi->common.skip_film_grain = ctx->skip_film_grain;
}
return AOM_CODEC_OK;
}
static aom_codec_err_t ctrl_get_accounting(aom_codec_alg_priv_t *ctx,
va_list args) {
#if !CONFIG_ACCOUNTING
@ -1231,6 +1274,7 @@ static aom_codec_ctrl_fn_map_t decoder_ctrl_maps[] = {
{ AV1D_EXT_TILE_DEBUG, ctrl_ext_tile_debug },
{ AV1D_SET_ROW_MT, ctrl_set_row_mt },
{ AV1D_SET_EXT_REF_PTR, ctrl_set_ext_ref_ptr },
{ AV1D_SET_SKIP_FILM_GRAIN, ctrl_set_skip_film_grain },
// Getters
{ AOMD_GET_FRAME_CORRUPTED, ctrl_get_frame_corrupted },

View file

@ -8,10 +8,11 @@
* Media Patent License 1.0 was not distributed with this source code in the
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AV1_AV1_IFACE_COMMON_H_
#define AV1_AV1_IFACE_COMMON_H_
#ifndef AOM_AV1_AV1_IFACE_COMMON_H_
#define AOM_AV1_AV1_IFACE_COMMON_H_
#include "aom_ports/mem.h"
#include "aom_scale/yv12config.h"
static void yuvconfig2image(aom_image_t *img, const YV12_BUFFER_CONFIG *yv12,
void *user_priv) {
@ -132,4 +133,4 @@ static aom_codec_err_t image2yuvconfig(const aom_image_t *img,
return AOM_CODEC_OK;
}
#endif // AV1_AV1_IFACE_COMMON_H_
#endif // AOM_AV1_AV1_IFACE_COMMON_H_

View file

@ -9,8 +9,8 @@
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AV1_COMMON_ALLOCCOMMON_H_
#define AV1_COMMON_ALLOCCOMMON_H_
#ifndef AOM_AV1_COMMON_ALLOCCOMMON_H_
#define AOM_AV1_COMMON_ALLOCCOMMON_H_
#define INVALID_IDX -1 // Invalid buffer index.
@ -45,4 +45,4 @@ int av1_get_MBs(int width, int height);
} // extern "C"
#endif
#endif // AV1_COMMON_ALLOCCOMMON_H_
#endif // AOM_AV1_COMMON_ALLOCCOMMON_H_

File diff suppressed because it is too large Load diff

View file

@ -8,8 +8,8 @@
* Media Patent License 1.0 was not distributed with this source code in the
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AV1_COMMON_ARM_AV1_INV_TXFM_NEON_H_
#define AV1_COMMON_ARM_AV1_INV_TXFM_NEON_H_
#ifndef AOM_AV1_COMMON_ARM_AV1_INV_TXFM_NEON_H_
#define AOM_AV1_COMMON_ARM_AV1_INV_TXFM_NEON_H_
#include "config/aom_config.h"
#include "config/av1_rtcd.h"
@ -23,6 +23,8 @@
typedef void (*transform_1d_neon)(const int32_t *input, int32_t *output,
const int8_t cos_bit,
const int8_t *stage_ptr);
typedef void (*transform_neon)(int16x8_t *input, int16x8_t *output,
int8_t cos_bit, int bit);
DECLARE_ALIGNED(16, static const int16_t, av1_eob_to_eobxy_8x8_default[8]) = {
0x0707, 0x0707, 0x0707, 0x0707, 0x0707, 0x0707, 0x0707, 0x0707,
@ -149,4 +151,4 @@ static INLINE void get_eobx_eoby_scan_h_identity(int *eobx, int *eoby,
*eoby = eob_fill[temp_eoby];
}
#endif // AV1_COMMON_ARM_AV1_INV_TXFM_NEON_H_
#endif // AOM_AV1_COMMON_ARM_AV1_INV_TXFM_NEON_H_

View file

@ -34,8 +34,8 @@ void aom_blend_a64_hmask_neon(uint8_t *dst, uint32_t dst_stride,
uint8x8_t tmp0, tmp1;
uint8x16_t res_q;
uint16x8_t res, res_low, res_high;
uint32x2_t tmp0_32, tmp1_32;
uint16x4_t tmp0_16, tmp1_16;
uint32x2_t tmp0_32 = vdup_n_u32(0), tmp1_32 = vdup_n_u32(0);
uint16x4_t tmp0_16 = vdup_n_u16(0), tmp1_16 = vdup_n_u16(0);
const uint8x8_t vdup_64 = vdup_n_u8((uint8_t)64);
if (w >= 16) {

View file

@ -27,8 +27,8 @@ void aom_blend_a64_vmask_neon(uint8_t *dst, uint32_t dst_stride,
uint8x8_t tmp0, tmp1;
uint8x16_t tmp0_q, tmp1_q, res_q;
uint16x8_t res, res_low, res_high;
uint32x2_t tmp0_32, tmp1_32;
uint16x4_t tmp0_16, tmp1_16;
uint32x2_t tmp0_32 = vdup_n_u32(0), tmp1_32 = vdup_n_u32(0);
uint16x4_t tmp0_16 = vdup_n_u16(0), tmp1_16 = vdup_n_u16(0);
assert(IMPLIES(src0 == dst, src0_stride == dst_stride));
assert(IMPLIES(src1 == dst, src1_stride == dst_stride));

View file

@ -131,7 +131,7 @@ static void cfl_luma_subsampling_444_lbd_neon(const uint8_t *input,
} while ((pred_buf_q3 += CFL_BUF_LINE) < end);
}
#if __ARM_ARCH <= 7
#ifndef __aarch64__
uint16x8_t vpaddq_u16(uint16x8_t a, uint16x8_t b) {
return vcombine_u16(vpadd_u16(vget_low_u16(a), vget_high_u16(a)),
vpadd_u16(vget_low_u16(b), vget_high_u16(b)));
@ -311,7 +311,7 @@ static INLINE void subtract_average_neon(const uint16_t *src, int16_t *dst,
// Permute and add in such a way that each lane contains the block sum.
// [A+C+B+D, B+D+A+C, C+A+D+B, D+B+C+A]
#if __ARM_ARCH >= 8
#ifdef __aarch64__
sum_32x4 = vpaddq_u32(sum_32x4, sum_32x4);
sum_32x4 = vpaddq_u32(sum_32x4, sum_32x4);
#else

View file

@ -13,6 +13,8 @@
#include <assert.h>
#include <arm_neon.h>
#include "config/av1_rtcd.h"
#include "aom_dsp/aom_dsp_common.h"
#include "aom_ports/mem.h"
#include "av1/common/convolve.h"
@ -68,6 +70,33 @@ static INLINE uint8x8_t convolve8_horiz_8x8(
return vqmovun_s16(sum);
}
#if !defined(__aarch64__)
static INLINE uint8x8_t convolve8_horiz_4x1(
const int16x4_t s0, const int16x4_t s1, const int16x4_t s2,
const int16x4_t s3, const int16x4_t s4, const int16x4_t s5,
const int16x4_t s6, const int16x4_t s7, const int16_t *filter,
const int16x4_t shift_round_0, const int16x4_t shift_by_bits) {
int16x4_t sum;
sum = vmul_n_s16(s0, filter[0]);
sum = vmla_n_s16(sum, s1, filter[1]);
sum = vmla_n_s16(sum, s2, filter[2]);
sum = vmla_n_s16(sum, s5, filter[5]);
sum = vmla_n_s16(sum, s6, filter[6]);
sum = vmla_n_s16(sum, s7, filter[7]);
/* filter[3] can take a max value of 128. So the max value of the result :
* 128*255 + sum > 16 bits
*/
sum = vqadd_s16(sum, vmul_n_s16(s3, filter[3]));
sum = vqadd_s16(sum, vmul_n_s16(s4, filter[4]));
sum = vqrshl_s16(sum, shift_round_0);
sum = vqrshl_s16(sum, shift_by_bits);
return vqmovun_s16(vcombine_s16(sum, sum));
}
#endif // !defined(__arch64__)
static INLINE uint8x8_t convolve8_vert_8x4(
const int16x8_t s0, const int16x8_t s1, const int16x8_t s2,
const int16x8_t s3, const int16x8_t s4, const int16x8_t s5,
@ -175,7 +204,10 @@ void av1_convolve_x_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
(void)conv_params;
(void)filter_params_y;
uint8x8_t t0, t1, t2, t3;
uint8x8_t t0;
#if defined(__aarch64__)
uint8x8_t t1, t2, t3;
#endif
assert(bits >= 0);
assert((FILTER_BITS - conv_params->round_1) >= 0 ||
@ -188,7 +220,7 @@ void av1_convolve_x_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
const int16x8_t shift_by_bits = vdupq_n_s16(-bits);
src -= horiz_offset;
#if defined(__aarch64__)
if (h == 4) {
uint8x8_t d01, d23;
int16x4_t s0, s1, s2, s3, s4, s5, s6, s7, s8, s9, s10, d0, d1, d2, d3;
@ -275,12 +307,18 @@ void av1_convolve_x_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
w -= 4;
} while (w > 0);
} else {
#endif
int width;
const uint8_t *s;
int16x8_t s0, s1, s2, s3, s4, s5, s6, s7;
#if defined(__aarch64__)
int16x8_t s8, s9, s10;
uint8x8_t t4, t5, t6, t7;
int16x8_t s0, s1, s2, s3, s4, s5, s6, s7, s8, s9, s10;
#endif
if (w <= 4) {
#if defined(__aarch64__)
do {
load_u8_8x8(src, src_stride, &t0, &t1, &t2, &t3, &t4, &t5, &t6, &t7);
transpose_u8_8x8(&t0, &t1, &t2, &t3, &t4, &t5, &t6, &t7);
@ -387,10 +425,49 @@ void av1_convolve_x_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
}
h -= 8;
} while (h > 0);
#else
int16x8_t tt0;
int16x4_t x0, x1, x2, x3, x4, x5, x6, x7;
const int16x4_t shift_round_0_low = vget_low_s16(shift_round_0);
const int16x4_t shift_by_bits_low = vget_low_s16(shift_by_bits);
do {
t0 = vld1_u8(src); // a0 a1 a2 a3 a4 a5 a6 a7
tt0 = vreinterpretq_s16_u16(vmovl_u8(t0));
x0 = vget_low_s16(tt0); // a0 a1 a2 a3
x4 = vget_high_s16(tt0); // a4 a5 a6 a7
t0 = vld1_u8(src + 8); // a8 a9 a10 a11 a12 a13 a14 a15
tt0 = vreinterpretq_s16_u16(vmovl_u8(t0));
x7 = vget_low_s16(tt0); // a8 a9 a10 a11
x1 = vext_s16(x0, x4, 1); // a1 a2 a3 a4
x2 = vext_s16(x0, x4, 2); // a2 a3 a4 a5
x3 = vext_s16(x0, x4, 3); // a3 a4 a5 a6
x5 = vext_s16(x4, x7, 1); // a5 a6 a7 a8
x6 = vext_s16(x4, x7, 2); // a6 a7 a8 a9
x7 = vext_s16(x4, x7, 3); // a7 a8 a9 a10
src += src_stride;
t0 = convolve8_horiz_4x1(x0, x1, x2, x3, x4, x5, x6, x7, x_filter,
shift_round_0_low, shift_by_bits_low);
if (w == 4) {
vst1_lane_u32((uint32_t *)dst, vreinterpret_u32_u8(t0),
0); // 00 01 02 03
dst += dst_stride;
} else if (w == 2) {
vst1_lane_u16((uint16_t *)dst, vreinterpret_u16_u8(t0), 0); // 00 01
dst += dst_stride;
}
h -= 1;
} while (h > 0);
#endif
} else {
uint8_t *d;
int16x8_t s11, s12, s13, s14;
int16x8_t s11;
#if defined(__aarch64__)
int16x8_t s12, s13, s14;
do {
__builtin_prefetch(src + 0 * src_stride);
__builtin_prefetch(src + 1 * src_stride);
@ -479,8 +556,47 @@ void av1_convolve_x_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
dst += 8 * dst_stride;
h -= 8;
} while (h > 0);
#else
do {
t0 = vld1_u8(src); // a0 a1 a2 a3 a4 a5 a6 a7
s0 = vreinterpretq_s16_u16(vmovl_u8(t0));
width = w;
s = src + 8;
d = dst;
__builtin_prefetch(dst);
do {
t0 = vld1_u8(s); // a8 a9 a10 a11 a12 a13 a14 a15
s7 = vreinterpretq_s16_u16(vmovl_u8(t0));
s11 = s0;
s0 = s7;
s1 = vextq_s16(s11, s7, 1); // a1 a2 a3 a4 a5 a6 a7 a8
s2 = vextq_s16(s11, s7, 2); // a2 a3 a4 a5 a6 a7 a8 a9
s3 = vextq_s16(s11, s7, 3); // a3 a4 a5 a6 a7 a8 a9 a10
s4 = vextq_s16(s11, s7, 4); // a4 a5 a6 a7 a8 a9 a10 a11
s5 = vextq_s16(s11, s7, 5); // a5 a6 a7 a8 a9 a10 a11 a12
s6 = vextq_s16(s11, s7, 6); // a6 a7 a8 a9 a10 a11 a12 a13
s7 = vextq_s16(s11, s7, 7); // a7 a8 a9 a10 a11 a12 a13 a14
t0 = convolve8_horiz_8x8(s11, s1, s2, s3, s4, s5, s6, s7, x_filter,
shift_round_0, shift_by_bits);
vst1_u8(d, t0);
s += 8;
d += 8;
width -= 8;
} while (width > 0);
src += src_stride;
dst += dst_stride;
h -= 1;
} while (h > 0);
#endif
}
#if defined(__aarch64__)
}
#endif
}
void av1_convolve_y_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
@ -505,9 +621,12 @@ void av1_convolve_y_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
filter_params_y, subpel_y_q4 & SUBPEL_MASK);
if (w <= 4) {
uint8x8_t d01, d23;
int16x4_t s0, s1, s2, s3, s4, s5, s6, s7, s8, s9, s10, d0, d1, d2, d3;
uint8x8_t d01;
int16x4_t s0, s1, s2, s3, s4, s5, s6, s7, d0;
#if defined(__aarch64__)
uint8x8_t d23;
int16x4_t s8, s9, s10, d1, d2, d3;
#endif
s0 = vreinterpret_s16_u16(vget_low_u16(vmovl_u8(vld1_u8(src))));
src += src_stride;
s1 = vreinterpret_s16_u16(vget_low_u16(vmovl_u8(vld1_u8(src))));
@ -526,6 +645,7 @@ void av1_convolve_y_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
do {
s7 = vreinterpret_s16_u16(vget_low_u16(vmovl_u8(vld1_u8(src))));
src += src_stride;
#if defined(__aarch64__)
s8 = vreinterpret_s16_u16(vget_low_u16(vmovl_u8(vld1_u8(src))));
src += src_stride;
s9 = vreinterpret_s16_u16(vget_low_u16(vmovl_u8(vld1_u8(src))));
@ -591,14 +711,41 @@ void av1_convolve_y_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
s5 = s9;
s6 = s10;
h -= 4;
#else
__builtin_prefetch(dst + 0 * dst_stride);
__builtin_prefetch(src + 0 * src_stride);
d0 = convolve8_4x4(s0, s1, s2, s3, s4, s5, s6, s7, y_filter);
d01 = vqrshrun_n_s16(vcombine_s16(d0, d0), FILTER_BITS);
if (w == 4) {
vst1_lane_u32((uint32_t *)dst, vreinterpret_u32_u8(d01), 0);
dst += dst_stride;
} else if (w == 2) {
vst1_lane_u16((uint16_t *)dst, vreinterpret_u16_u8(d01), 0);
dst += dst_stride;
}
s0 = s1;
s1 = s2;
s2 = s3;
s3 = s4;
s4 = s5;
s5 = s6;
s6 = s7;
h -= 1;
#endif
} while (h > 0);
} else {
int height;
const uint8_t *s;
uint8_t *d;
uint8x8_t t0, t1, t2, t3;
int16x8_t s0, s1, s2, s3, s4, s5, s6, s7, s8, s9, s10;
uint8x8_t t0;
int16x8_t s0, s1, s2, s3, s4, s5, s6, s7;
#if defined(__aarch64__)
uint8x8_t t1, t2, t3;
int16x8_t s8, s9, s10;
#endif
do {
__builtin_prefetch(src + 0 * src_stride);
__builtin_prefetch(src + 1 * src_stride);
@ -628,6 +775,7 @@ void av1_convolve_y_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
do {
s7 = vreinterpretq_s16_u16(vmovl_u8(vld1_u8(s)));
s += src_stride;
#if defined(__aarch64__)
s8 = vreinterpretq_s16_u16(vmovl_u8(vld1_u8(s)));
s += src_stride;
s9 = vreinterpretq_s16_u16(vmovl_u8(vld1_u8(s)));
@ -670,6 +818,24 @@ void av1_convolve_y_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
s5 = s9;
s6 = s10;
height -= 4;
#else
__builtin_prefetch(d);
__builtin_prefetch(s);
t0 = convolve8_vert_8x4(s0, s1, s2, s3, s4, s5, s6, s7, y_filter);
vst1_u8(d, t0);
d += dst_stride;
s0 = s1;
s1 = s2;
s2 = s3;
s3 = s4;
s4 = s5;
s5 = s6;
s6 = s7;
height -= 1;
#endif
} while (height > 0);
src += 8;
dst += 8;
@ -686,7 +852,10 @@ void av1_convolve_2d_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
ConvolveParams *conv_params) {
int im_dst_stride;
int width, height;
uint8x8_t t0, t1, t2, t3, t4, t5, t6, t7;
uint8x8_t t0;
#if defined(__aarch64__)
uint8x8_t t1, t2, t3, t4, t5, t6, t7;
#endif
DECLARE_ALIGNED(16, int16_t,
im_block[(MAX_SB_SIZE + HORIZ_EXTRA_ROWS) * MAX_SB_SIZE]);
@ -724,13 +893,18 @@ void av1_convolve_2d_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
assert(conv_params->round_0 > 0);
if (w <= 4) {
int16x4_t s0, s1, s2, s3, s4, s5, s6, s7, s8, s9, s10, d0, d1, d2, d3;
int16x4_t s0, s1, s2, s3, s4, s5, s6, s7, d0;
#if defined(__aarch64__)
int16x4_t s8, s9, s10, d1, d2, d3;
#endif
const int16x4_t horiz_const = vdup_n_s16((1 << (bd + FILTER_BITS - 2)));
const int16x4_t shift_round_0 = vdup_n_s16(-(conv_params->round_0 - 1));
do {
s = src_ptr;
#if defined(__aarch64__)
__builtin_prefetch(s + 0 * src_stride);
__builtin_prefetch(s + 1 * src_stride);
__builtin_prefetch(s + 2 * src_stride);
@ -789,16 +963,56 @@ void av1_convolve_2d_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
src_ptr += 4 * src_stride;
dst_ptr += 4 * im_dst_stride;
height -= 4;
#else
int16x8_t tt0;
__builtin_prefetch(s);
t0 = vld1_u8(s); // a0 a1 a2 a3 a4 a5 a6 a7
tt0 = vreinterpretq_s16_u16(vmovl_u8(t0));
s0 = vget_low_s16(tt0);
s4 = vget_high_s16(tt0);
__builtin_prefetch(dst_ptr);
s += 8;
t0 = vld1_u8(s); // a8 a9 a10 a11 a12 a13 a14 a15
s7 = vget_low_s16(vreinterpretq_s16_u16(vmovl_u8(t0)));
s1 = vext_s16(s0, s4, 1); // a1 a2 a3 a4
s2 = vext_s16(s0, s4, 2); // a2 a3 a4 a5
s3 = vext_s16(s0, s4, 3); // a3 a4 a5 a6
s5 = vext_s16(s4, s7, 1); // a5 a6 a7 a8
s6 = vext_s16(s4, s7, 2); // a6 a7 a8 a9
s7 = vext_s16(s4, s7, 3); // a7 a8 a9 a10
d0 = convolve8_4x4_s16(s0, s1, s2, s3, s4, s5, s6, s7, x_filter_tmp,
horiz_const, shift_round_0);
if (w == 4) {
vst1_s16(dst_ptr, d0);
dst_ptr += im_dst_stride;
} else if (w == 2) {
vst1_lane_u32((uint32_t *)dst_ptr, vreinterpret_u32_s16(d0), 0);
dst_ptr += im_dst_stride;
}
src_ptr += src_stride;
height -= 1;
#endif
} while (height > 0);
} else {
int16_t *d_tmp;
int16x8_t s0, s1, s2, s3, s4, s5, s6, s7, res0;
#if defined(__aarch64__)
int16x8_t s8, s9, s10, res1, res2, res3, res4, res5, res6, res7;
int16x8_t s11, s12, s13, s14;
int16x8_t s0, s1, s2, s3, s4, s5, s6, s7, s8, s9, s10;
int16x8_t res0, res1, res2, res3, res4, res5, res6, res7;
#endif
const int16x8_t horiz_const = vdupq_n_s16((1 << (bd + FILTER_BITS - 2)));
const int16x8_t shift_round_0 = vdupq_n_s16(-(conv_params->round_0 - 1));
#if defined(__aarch64__)
do {
__builtin_prefetch(src_ptr + 0 * src_stride);
__builtin_prefetch(src_ptr + 1 * src_stride);
@ -886,6 +1100,45 @@ void av1_convolve_2d_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
dst_ptr += 8 * im_dst_stride;
height -= 8;
} while (height > 0);
#else
do {
t0 = vld1_u8(src_ptr);
s0 = vreinterpretq_s16_u16(vmovl_u8(t0)); // a0 a1 a2 a3 a4 a5 a6 a7
width = w;
s = src_ptr + 8;
d_tmp = dst_ptr;
__builtin_prefetch(dst_ptr);
do {
t0 = vld1_u8(s); // a8 a9 a10 a11 a12 a13 a14 a15
s7 = vreinterpretq_s16_u16(vmovl_u8(t0));
int16x8_t sum = s0;
s0 = s7;
s1 = vextq_s16(sum, s7, 1); // a1 a2 a3 a4 a5 a6 a7 a8
s2 = vextq_s16(sum, s7, 2); // a2 a3 a4 a5 a6 a7 a8 a9
s3 = vextq_s16(sum, s7, 3); // a3 a4 a5 a6 a7 a8 a9 a10
s4 = vextq_s16(sum, s7, 4); // a4 a5 a6 a7 a8 a9 a10 a11
s5 = vextq_s16(sum, s7, 5); // a5 a6 a7 a8 a9 a10 a11 a12
s6 = vextq_s16(sum, s7, 6); // a6 a7 a8 a9 a10 a11 a12 a13
s7 = vextq_s16(sum, s7, 7); // a7 a8 a9 a10 a11 a12 a13 a14
res0 = convolve8_8x8_s16(sum, s1, s2, s3, s4, s5, s6, s7, x_filter_tmp,
horiz_const, shift_round_0);
vst1q_s16(d_tmp, res0);
s += 8;
d_tmp += 8;
width -= 8;
} while (width > 0);
src_ptr += src_stride;
dst_ptr += im_dst_stride;
height -= 1;
} while (height > 0);
#endif
}
// vertical
@ -910,10 +1163,17 @@ void av1_convolve_2d_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
width = w;
if (width <= 4) {
int16x4_t s0, s1, s2, s3, s4, s5, s6, s7, s8, s9, s10;
uint16x4_t d0, d1, d2, d3;
uint16x8_t dd0, dd1;
uint8x8_t d01, d23;
int16x4_t s0, s1, s2, s3, s4, s5, s6, s7;
uint16x4_t d0;
uint16x8_t dd0;
uint8x8_t d01;
#if defined(__aarch64__)
int16x4_t s8, s9, s10;
uint16x4_t d1, d2, d3;
uint16x8_t dd1;
uint8x8_t d23;
#endif
d_u8 = dst_u8_ptr;
v_s = v_src_ptr;
@ -931,6 +1191,7 @@ void av1_convolve_2d_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
v_s += (7 * im_stride);
do {
#if defined(__aarch64__)
load_s16_4x4(v_s, im_stride, &s7, &s8, &s9, &s10);
v_s += (im_stride << 2);
@ -1008,11 +1269,48 @@ void av1_convolve_2d_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
s5 = s9;
s6 = s10;
height -= 4;
#else
s7 = vld1_s16(v_s);
v_s += im_stride;
__builtin_prefetch(d_u8 + 0 * dst_stride);
d0 = convolve8_vert_4x4_s32(s0, s1, s2, s3, s4, s5, s6, s7, y_filter,
round_shift_vec, offset_const,
sub_const_vec);
dd0 = vqrshlq_u16(vcombine_u16(d0, d0), vec_round_bits);
d01 = vqmovn_u16(dd0);
if (w == 4) {
vst1_lane_u32((uint32_t *)d_u8, vreinterpret_u32_u8(d01),
0); // 00 01 02 03
d_u8 += dst_stride;
} else if (w == 2) {
vst1_lane_u16((uint16_t *)d_u8, vreinterpret_u16_u8(d01),
0); // 00 01
d_u8 += dst_stride;
}
s0 = s1;
s1 = s2;
s2 = s3;
s3 = s4;
s4 = s5;
s5 = s6;
s6 = s7;
height -= 1;
#endif
} while (height > 0);
} else {
// if width is a multiple of 8 & height is a multiple of 4
int16x8_t s0, s1, s2, s3, s4, s5, s6, s7, s8, s9, s10;
uint8x8_t res0, res1, res2, res3;
int16x8_t s0, s1, s2, s3, s4, s5, s6, s7;
uint8x8_t res0;
#if defined(__aarch64__)
int16x8_t s8, s9, s10;
uint8x8_t res1, res2, res3;
#endif
do {
__builtin_prefetch(v_src_ptr + 0 * im_stride);
@ -1032,6 +1330,7 @@ void av1_convolve_2d_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
height = h;
do {
#if defined(__aarch64__)
load_s16_8x4(v_s, im_stride, &s7, &s8, &s9, &s10);
v_s += (im_stride << 2);
@ -1076,6 +1375,28 @@ void av1_convolve_2d_sr_neon(const uint8_t *src, int src_stride, uint8_t *dst,
s5 = s9;
s6 = s10;
height -= 4;
#else
s7 = vld1q_s16(v_s);
v_s += im_stride;
__builtin_prefetch(d_u8 + 0 * dst_stride);
res0 = convolve8_vert_8x4_s32(s0, s1, s2, s3, s4, s5, s6, s7,
y_filter, round_shift_vec, offset_const,
sub_const_vec, vec_round_bits);
vst1_u8(d_u8, res0);
d_u8 += dst_stride;
s0 = s1;
s1 = s2;
s2 = s3;
s3 = s4;
s4 = s5;
s5 = s6;
s6 = s7;
height -= 1;
#endif
} while (height > 0);
v_src_ptr += 8;
dst_u8_ptr += 8;

View file

@ -8,8 +8,8 @@
* be found in the AUTHORS file in the root of the source tree.
*/
#ifndef AV1_COMMON_ARM_CONVOLVE_NEON_H_
#define AV1_COMMON_ARM_CONVOLVE_NEON_H_
#ifndef AOM_AV1_COMMON_ARM_CONVOLVE_NEON_H_
#define AOM_AV1_COMMON_ARM_CONVOLVE_NEON_H_
#include <arm_neon.h>
@ -225,4 +225,4 @@ static INLINE uint16x4_t convolve8_4x4_s32(
return res;
}
#endif // AV1_COMMON_ARM_CONVOLVE_NEON_H_
#endif // AOM_AV1_COMMON_ARM_CONVOLVE_NEON_H_

View file

@ -22,12 +22,108 @@
#include "av1/common/arm/mem_neon.h"
#include "av1/common/arm/transpose_neon.h"
#if !defined(__aarch64__)
static INLINE void compute_avg_4x1(uint16x4_t res0, uint16x4_t d0,
const uint16_t fwd_offset,
const uint16_t bck_offset,
const int16x4_t sub_const_vec,
const int16_t round_bits,
const int use_jnt_comp_avg, uint8x8_t *t0) {
int16x4_t tmp0;
uint16x4_t tmp_u0;
uint32x4_t sum0;
int32x4_t dst0;
int16x8_t tmp4;
if (use_jnt_comp_avg) {
const int32x4_t round_bits_vec = vdupq_n_s32((int32_t)(-round_bits));
sum0 = vmull_n_u16(res0, fwd_offset);
sum0 = vmlal_n_u16(sum0, d0, bck_offset);
sum0 = vshrq_n_u32(sum0, DIST_PRECISION_BITS);
dst0 = vsubq_s32(vreinterpretq_s32_u32(sum0), vmovl_s16(sub_const_vec));
dst0 = vqrshlq_s32(dst0, round_bits_vec);
tmp0 = vqmovn_s32(dst0);
tmp4 = vcombine_s16(tmp0, tmp0);
*t0 = vqmovun_s16(tmp4);
} else {
const int16x4_t round_bits_vec = vdup_n_s16(-round_bits);
tmp_u0 = vhadd_u16(res0, d0);
tmp0 = vsub_s16(vreinterpret_s16_u16(tmp_u0), sub_const_vec);
tmp0 = vqrshl_s16(tmp0, round_bits_vec);
tmp4 = vcombine_s16(tmp0, tmp0);
*t0 = vqmovun_s16(tmp4);
}
}
static INLINE void compute_avg_8x1(uint16x8_t res0, uint16x8_t d0,
const uint16_t fwd_offset,
const uint16_t bck_offset,
const int16x4_t sub_const,
const int16_t round_bits,
const int use_jnt_comp_avg, uint8x8_t *t0) {
int16x4_t tmp0, tmp2;
int16x8_t f0;
uint32x4_t sum0, sum2;
int32x4_t dst0, dst2;
uint16x8_t tmp_u0;
if (use_jnt_comp_avg) {
const int32x4_t sub_const_vec = vmovl_s16(sub_const);
const int32x4_t round_bits_vec = vdupq_n_s32(-(int32_t)round_bits);
sum0 = vmull_n_u16(vget_low_u16(res0), fwd_offset);
sum0 = vmlal_n_u16(sum0, vget_low_u16(d0), bck_offset);
sum0 = vshrq_n_u32(sum0, DIST_PRECISION_BITS);
sum2 = vmull_n_u16(vget_high_u16(res0), fwd_offset);
sum2 = vmlal_n_u16(sum2, vget_high_u16(d0), bck_offset);
sum2 = vshrq_n_u32(sum2, DIST_PRECISION_BITS);
dst0 = vsubq_s32(vreinterpretq_s32_u32(sum0), sub_const_vec);
dst2 = vsubq_s32(vreinterpretq_s32_u32(sum2), sub_const_vec);
dst0 = vqrshlq_s32(dst0, round_bits_vec);
dst2 = vqrshlq_s32(dst2, round_bits_vec);
tmp0 = vqmovn_s32(dst0);
tmp2 = vqmovn_s32(dst2);
f0 = vcombine_s16(tmp0, tmp2);
*t0 = vqmovun_s16(f0);
} else {
const int16x8_t sub_const_vec = vcombine_s16(sub_const, sub_const);
const int16x8_t round_bits_vec = vdupq_n_s16(-round_bits);
tmp_u0 = vhaddq_u16(res0, d0);
f0 = vsubq_s16(vreinterpretq_s16_u16(tmp_u0), sub_const_vec);
f0 = vqrshlq_s16(f0, round_bits_vec);
*t0 = vqmovun_s16(f0);
}
}
#endif // !defined(__arch64__)
static INLINE void compute_avg_4x4(
uint16x4_t res0, uint16x4_t res1, uint16x4_t res2, uint16x4_t res3,
uint16x4_t d0, uint16x4_t d1, uint16x4_t d2, uint16x4_t d3,
const uint16_t fwd_offset, const uint16_t bck_offset,
const int16x4_t sub_const_vec, const int16_t round_bits,
const int32_t use_jnt_comp_avg, uint8x8_t *t0, uint8x8_t *t1) {
const int use_jnt_comp_avg, uint8x8_t *t0, uint8x8_t *t1) {
int16x4_t tmp0, tmp1, tmp2, tmp3;
uint16x4_t tmp_u0, tmp_u1, tmp_u2, tmp_u3;
uint32x4_t sum0, sum1, sum2, sum3;
@ -107,7 +203,7 @@ static INLINE void compute_avg_8x4(
uint16x8_t d0, uint16x8_t d1, uint16x8_t d2, uint16x8_t d3,
const uint16_t fwd_offset, const uint16_t bck_offset,
const int16x4_t sub_const, const int16_t round_bits,
const int32_t use_jnt_comp_avg, uint8x8_t *t0, uint8x8_t *t1, uint8x8_t *t2,
const int use_jnt_comp_avg, uint8x8_t *t0, uint8x8_t *t1, uint8x8_t *t2,
uint8x8_t *t3) {
int16x4_t tmp0, tmp1, tmp2, tmp3, tmp4, tmp5, tmp6, tmp7;
int16x8_t f0, f1, f2, f3;
@ -231,7 +327,6 @@ static INLINE void jnt_convolve_2d_horiz_neon(
int16_t *dst_ptr;
int dst_stride;
int width, height;
uint8x8_t t0, t1, t2, t3, t4, t5, t6, t7;
dst_ptr = im_block;
dst_stride = im_stride;
@ -239,15 +334,22 @@ static INLINE void jnt_convolve_2d_horiz_neon(
width = w;
if (w == 4) {
int16x4_t s0, s1, s2, s3, s4, s5, s6, s7, s8, s9, s10, d0, d1, d2, d3;
int16x8_t tt0, tt1, tt2, tt3;
int16x4_t s0, s1, s2, s3, s4, s5, s6, s7, d0;
int16x8_t tt0;
uint8x8_t t0;
const int16x4_t horiz_const = vdup_n_s16((1 << (bd + FILTER_BITS - 2)));
const int16x4_t shift_round_0 = vdup_n_s16(-(round_0));
#if defined(__aarch64__)
int16x4_t s8, s9, s10, d1, d2, d3;
int16x8_t tt1, tt2, tt3;
uint8x8_t t1, t2, t3;
#endif
do {
s = src;
__builtin_prefetch(s + 0 * src_stride);
#if defined(__aarch64__)
__builtin_prefetch(s + 1 * src_stride);
__builtin_prefetch(s + 2 * src_stride);
__builtin_prefetch(s + 3 * src_stride);
@ -301,17 +403,48 @@ static INLINE void jnt_convolve_2d_horiz_neon(
src += 4 * src_stride;
dst_ptr += 4 * dst_stride;
height -= 4;
#else
t0 = vld1_u8(s); // a0 a1 a2 a3 a4 a5 a6 a7
tt0 = vreinterpretq_s16_u16(vmovl_u8(t0)); // a0 a1 a2 a3 a4 a5 a6 a7
s0 = vget_low_s16(tt0); // a0 a1 a2 a3
s4 = vget_high_s16(tt0); // a4 a5 a6 a7
__builtin_prefetch(dst_ptr);
s += 8;
t0 = vld1_u8(s); // a8 a9 a10 a11
// a8 a9 a10 a11
s7 = vget_low_s16(vreinterpretq_s16_u16(vmovl_u8(t0)));
s1 = vext_s16(s0, s4, 1); // a1 a2 a3 a4
s2 = vext_s16(s0, s4, 2); // a2 a3 a4 a5
s3 = vext_s16(s0, s4, 3); // a3 a4 a5 a6
s5 = vext_s16(s4, s7, 1); // a5 a6 a7 a8
s6 = vext_s16(s4, s7, 2); // a6 a7 a8 a9
s7 = vext_s16(s4, s7, 3); // a7 a8 a9 a10
d0 = convolve8_4x4_s16(s0, s1, s2, s3, s4, s5, s6, s7, x_filter_tmp,
horiz_const, shift_round_0);
vst1_s16(dst_ptr, d0);
src += src_stride;
dst_ptr += dst_stride;
height -= 1;
#endif
} while (height > 0);
} else {
int16_t *d_tmp;
int16x8_t s11, s12, s13, s14;
int16x8_t s0, s1, s2, s3, s4, s5, s6, s7, s8, s9, s10;
int16x8_t res0, res1, res2, res3, res4, res5, res6, res7;
int16x8_t s0, s1, s2, s3, s4, s5, s6, s7;
int16x8_t res0;
uint8x8_t t0;
const int16x8_t horiz_const = vdupq_n_s16((1 << (bd + FILTER_BITS - 2)));
const int16x8_t shift_round_0 = vdupq_n_s16(-(round_0));
do {
#if defined(__aarch64__)
uint8x8_t t1, t2, t3, t4, t5, t6, t7;
int16x8_t s8, s9, s10, s11, s12, s13, s14;
int16x8_t res1, res2, res3, res4, res5, res6, res7;
__builtin_prefetch(src + 0 * src_stride);
__builtin_prefetch(src + 1 * src_stride);
__builtin_prefetch(src + 2 * src_stride);
@ -390,6 +523,42 @@ static INLINE void jnt_convolve_2d_horiz_neon(
src += 8 * src_stride;
dst_ptr += 8 * dst_stride;
height -= 8;
#else
int16x8_t temp_0;
t0 = vld1_u8(src);
s0 = vreinterpretq_s16_u16(vmovl_u8(t0)); // a0 a1 a2 a3 a4 a5 a6 a7
width = w;
s = src + 8;
d_tmp = dst_ptr;
__builtin_prefetch(dst_ptr);
do {
t0 = vld1_u8(s); // a8 a9 a10 a11 a12 a13 a14 a15
s7 = vreinterpretq_s16_u16(vmovl_u8(t0));
temp_0 = s0;
s0 = s7;
s1 = vextq_s16(temp_0, s7, 1); // a1 a2 a3 a4 a5 a6 a7 a8
s2 = vextq_s16(temp_0, s7, 2); // a2 a3 a4 a5 a6 a7 a8 a9
s3 = vextq_s16(temp_0, s7, 3); // a3 a4 a5 a6 a7 a8 a9 a10
s4 = vextq_s16(temp_0, s7, 4); // a4 a5 a6 a7 a8 a9 a10 a11
s5 = vextq_s16(temp_0, s7, 5); // a5 a6 a7 a8 a9 a10 a11 a12
s6 = vextq_s16(temp_0, s7, 6); // a6 a7 a8 a9 a10 a11 a12 a13
s7 = vextq_s16(temp_0, s7, 7); // a7 a8 a9 a10 a11 a12 a13 a14
res0 = convolve8_8x8_s16(temp_0, s1, s2, s3, s4, s5, s6, s7,
x_filter_tmp, horiz_const, shift_round_0);
vst1q_s16(d_tmp, res0);
s += 8;
d_tmp += 8;
width -= 8;
} while (width > 0);
src += src_stride;
dst_ptr += dst_stride;
height -= 1;
#endif
} while (height > 0);
}
}
@ -420,10 +589,15 @@ static INLINE void jnt_convolve_2d_vert_neon(
const int do_average = conv_params->do_average;
const int use_jnt_comp_avg = conv_params->use_jnt_comp_avg;
int16x4_t s0, s1, s2, s3, s4, s5, s6, s7, s8, s9, s10;
uint16x4_t res4, res5, res6, res7;
uint16x4_t d0, d1, d2, d3;
uint8x8_t t0, t1;
int16x4_t s0, s1, s2, s3, s4, s5, s6, s7;
uint16x4_t res4, d0;
uint8x8_t t0;
#if defined(__aarch64__)
int16x4_t s8, s9, s10;
uint16x4_t res5, res6, res7, d1, d2, d3;
uint8x8_t t1;
#endif
dst = conv_params->dst;
src_ptr = im_block;
@ -450,6 +624,7 @@ static INLINE void jnt_convolve_2d_vert_neon(
s += (7 * im_stride);
do {
#if defined(__aarch64__)
load_s16_4x4(s, im_stride, &s7, &s8, &s9, &s10);
s += (im_stride << 2);
@ -480,17 +655,13 @@ static INLINE void jnt_convolve_2d_vert_neon(
bck_offset, sub_const_vec, round_bits, use_jnt_comp_avg,
&t0, &t1);
vst1_lane_u32((uint32_t *)d_u8, vreinterpret_u32_u8(t0),
0); // 00 01 02 03
vst1_lane_u32((uint32_t *)d_u8, vreinterpret_u32_u8(t0), 0);
d_u8 += dst8_stride;
vst1_lane_u32((uint32_t *)d_u8, vreinterpret_u32_u8(t0),
1); // 10 11 12 13
vst1_lane_u32((uint32_t *)d_u8, vreinterpret_u32_u8(t0), 1);
d_u8 += dst8_stride;
vst1_lane_u32((uint32_t *)d_u8, vreinterpret_u32_u8(t1),
0); // 20 21 22 23
vst1_lane_u32((uint32_t *)d_u8, vreinterpret_u32_u8(t1), 0);
d_u8 += dst8_stride;
vst1_lane_u32((uint32_t *)d_u8, vreinterpret_u32_u8(t1),
1); // 30 31 32 33
vst1_lane_u32((uint32_t *)d_u8, vreinterpret_u32_u8(t1), 1);
d_u8 += dst8_stride;
} else {
@ -505,6 +676,39 @@ static INLINE void jnt_convolve_2d_vert_neon(
s5 = s9;
s6 = s10;
height -= 4;
#else
s7 = vld1_s16(s);
s += (im_stride);
__builtin_prefetch(d + 0 * dst_stride);
__builtin_prefetch(d_u8 + 0 * dst8_stride);
d0 = convolve8_4x4_s32(s0, s1, s2, s3, s4, s5, s6, s7, y_filter,
round_shift_vec, offset_const);
if (do_average) {
res4 = vld1_u16(d);
d += (dst_stride);
compute_avg_4x1(res4, d0, fwd_offset, bck_offset, sub_const_vec,
round_bits, use_jnt_comp_avg, &t0);
vst1_lane_u32((uint32_t *)d_u8, vreinterpret_u32_u8(t0), 0);
d_u8 += dst8_stride;
} else {
vst1_u16(d, d0);
d += (dst_stride);
}
s0 = s1;
s1 = s2;
s2 = s3;
s3 = s4;
s4 = s5;
s5 = s6;
s6 = s7;
height--;
#endif
} while (height > 0);
src_ptr += 4;
dst_ptr += 4;
@ -722,8 +926,10 @@ void av1_jnt_convolve_x_neon(const uint8_t *src, int src_stride, uint8_t *dst8,
uint8_t *dst_u8_ptr;
CONV_BUF_TYPE *d, *dst_ptr;
int width, height;
uint8x8_t t0, t1, t2, t3, t4, t5, t6, t7;
uint8x8_t t0;
#if defined(__aarch64__)
uint8x8_t t1, t2, t3, t4, t5, t6, t7;
#endif
s = src_ptr;
dst_ptr = dst;
dst_u8_ptr = dst8;
@ -731,11 +937,18 @@ void av1_jnt_convolve_x_neon(const uint8_t *src, int src_stride, uint8_t *dst8,
height = h;
if ((w == 4) || (h == 4)) {
int16x4_t s0, s1, s2, s3, s4, s5, s6, s7, s8, s9, s10, d0, d1, d2, d3;
int16x8_t tt0, tt1, tt2, tt3;
uint16x4_t res4, res5, res6, res7;
uint32x2_t tu0, tu1;
int16x4_t s0, s1, s2, s3, s4, s5, s6, s7, d0;
int16x8_t tt0;
uint16x4_t res4;
#if defined(__aarch64__)
int16x4_t s8, s9, s10, d1, d2, d3;
int16x8_t tt1, tt2, tt3;
uint16x4_t res5, res6, res7;
uint32x2_t tu0 = vdup_n_u32(0), tu1 = vdup_n_u32(0);
int16x8_t u0, u1;
#else
int16x4_t temp_0;
#endif
const int16x4_t zero = vdup_n_s16(0);
const int16x4_t round_offset_vec = vdup_n_s16(round_offset);
const int16x4_t shift_round_0 = vdup_n_s16(-conv_params->round_0 + 1);
@ -746,6 +959,7 @@ void av1_jnt_convolve_x_neon(const uint8_t *src, int src_stride, uint8_t *dst8,
d_u8 = dst_u8_ptr;
width = w;
__builtin_prefetch(s + 0 * src_stride);
#if defined(__aarch64__)
__builtin_prefetch(s + 1 * src_stride);
__builtin_prefetch(s + 2 * src_stride);
__builtin_prefetch(s + 3 * src_stride);
@ -854,15 +1068,66 @@ void av1_jnt_convolve_x_neon(const uint8_t *src, int src_stride, uint8_t *dst8,
dst_ptr += (dst_stride << 2);
dst_u8_ptr += (dst8_stride << 2);
height -= 4;
#else
t0 = vld1_u8(s); // a0 a1 a2 a3 a4 a5 a6 a7
tt0 = vreinterpretq_s16_u16(vmovl_u8(t0)); // a0 a1 a2 a3 a4 a5 a6 a7
s0 = vget_low_s16(tt0); // a0 a1 a2 a3
s4 = vget_high_s16(tt0); // a4 a5 a6 a7
__builtin_prefetch(d);
s += 8;
do {
t0 = vld1_u8(s); // a8 a9 a10 a11
// a8 a9 a10 a11
s7 = vget_low_s16(vreinterpretq_s16_u16(vmovl_u8(t0)));
temp_0 = s7;
s1 = vext_s16(s0, s4, 1); // a1 a2 a3 a4
s2 = vext_s16(s0, s4, 2); // a2 a3 a4 a5
s3 = vext_s16(s0, s4, 3); // a3 a4 a5 a6
s5 = vext_s16(s4, s7, 1); // a5 a6 a7 a8
s6 = vext_s16(s4, s7, 2); // a6 a7 a8 a9
s7 = vext_s16(s4, s7, 3); // a7 a8 a9 a10
d0 = convolve8_4x4_s16(s0, s1, s2, s3, s4, s5, s6, s7, x_filter_tmp,
zero, shift_round_0);
d0 = vrshl_s16(d0, horiz_const);
d0 = vadd_s16(d0, round_offset_vec);
s0 = s4;
s4 = temp_0;
if (conv_params->do_average) {
__builtin_prefetch(d);
__builtin_prefetch(d_u8);
res4 = vld1_u16(d);
compute_avg_4x1(res4, vreinterpret_u16_s16(d0), fwd_offset,
bck_offset, round_offset_vec, round_bits,
use_jnt_comp_avg, &t0);
vst1_lane_u32((uint32_t *)d_u8, vreinterpret_u32_u8(t0),
0); // 00 01 02 03
} else {
vst1_u16(d, vreinterpret_u16_s16(d0));
}
s += 4;
width -= 4;
d += 4;
d_u8 += 4;
} while (width > 0);
src_ptr += (src_stride);
dst_ptr += (dst_stride);
dst_u8_ptr += (dst8_stride);
height--;
#endif
} while (height > 0);
} else {
CONV_BUF_TYPE *d_tmp;
uint8_t *d_u8_tmp;
int16x8_t s11, s12, s13, s14;
int16x8_t s0, s1, s2, s3, s4, s5, s6, s7, s8, s9, s10;
int16x8_t res0, res1, res2, res3, res4, res5, res6, res7;
uint16x8_t res8, res9, res10, res11;
int16x8_t s0, s1, s2, s3, s4, s5, s6, s7;
int16x8_t res0;
uint16x8_t res8;
const int16x8_t round_offset128 = vdupq_n_s16(round_offset);
const int16x4_t round_offset64 = vdup_n_s16(round_offset);
const int16x8_t shift_round_0 = vdupq_n_s16(-conv_params->round_0 + 1);
@ -872,6 +1137,11 @@ void av1_jnt_convolve_x_neon(const uint8_t *src, int src_stride, uint8_t *dst8,
d = dst_ptr = dst;
d_u8 = dst_u8_ptr = dst8;
do {
#if defined(__aarch64__)
int16x8_t s11, s12, s13, s14;
int16x8_t s8, s9, s10;
int16x8_t res1, res2, res3, res4, res5, res6, res7;
uint16x8_t res9, res10, res11;
__builtin_prefetch(src_ptr + 0 * src_stride);
__builtin_prefetch(src_ptr + 1 * src_stride);
__builtin_prefetch(src_ptr + 2 * src_stride);
@ -1007,6 +1277,67 @@ void av1_jnt_convolve_x_neon(const uint8_t *src, int src_stride, uint8_t *dst8,
dst_ptr += 8 * dst_stride;
dst_u8_ptr += 8 * dst8_stride;
height -= 8;
#else
int16x8_t temp_0;
__builtin_prefetch(src_ptr);
t0 = vld1_u8(src_ptr);
s0 = vreinterpretq_s16_u16(vmovl_u8(t0)); // a0 a1 a2 a3 a4 a5 a6 a7
width = w;
s = src_ptr + 8;
d = dst_ptr;
d_u8_tmp = dst_u8_ptr;
__builtin_prefetch(dst_ptr);
do {
d_u8 = d_u8_tmp;
d_tmp = d;
t0 = vld1_u8(s); // a8 a9 a10 a11 a12 a13 a14 a15
s7 = vreinterpretq_s16_u16(vmovl_u8(t0));
temp_0 = s0;
s0 = s7;
s1 = vextq_s16(temp_0, s7, 1); // a1 a2 a3 a4 a5 a6 a7 a8
s2 = vextq_s16(temp_0, s7, 2); // a2 a3 a4 a5 a6 a7 a8 a9
s3 = vextq_s16(temp_0, s7, 3); // a3 a4 a5 a6 a7 a8 a9 a10
s4 = vextq_s16(temp_0, s7, 4); // a4 a5 a6 a7 a8 a9 a10 a11
s5 = vextq_s16(temp_0, s7, 5); // a5 a6 a7 a8 a9 a10 a11 a12
s6 = vextq_s16(temp_0, s7, 6); // a6 a7 a8 a9 a10 a11 a12 a13
s7 = vextq_s16(temp_0, s7, 7); // a7 a8 a9 a10 a11 a12 a13 a14
res0 = convolve8_8x8_s16(temp_0, s1, s2, s3, s4, s5, s6, s7,
x_filter_tmp, zero, shift_round_0);
res0 = vrshlq_s16(res0, horiz_const);
res0 = vaddq_s16(res0, round_offset128);
if (conv_params->do_average) {
res8 = vld1q_u16(d_tmp);
d_tmp += (dst_stride);
compute_avg_8x1(res8, vreinterpretq_u16_s16(res0), fwd_offset,
bck_offset, round_offset64, round_bits,
use_jnt_comp_avg, &t0);
vst1_u8(d_u8, t0);
d_u8 += (dst8_stride);
} else {
vst1q_u16(d_tmp, vreinterpretq_u16_s16(res0));
d_tmp += (dst_stride);
}
s += 8;
d += 8;
width -= 8;
d_u8_tmp += 8;
} while (width > 0);
src_ptr += src_stride;
dst_ptr += dst_stride;
dst_u8_ptr += dst8_stride;
height--;
#endif
} while (height > 0);
}
}
@ -1057,7 +1388,6 @@ void av1_jnt_convolve_y_neon(const uint8_t *src, int src_stride, uint8_t *dst8,
uint8_t *dst_u8_ptr;
CONV_BUF_TYPE *d, *dst_ptr;
int width, height;
uint8x8_t t0, t1, t2, t3, t4, t5, t6, t7;
s = src_ptr;
dst_ptr = dst;
@ -1070,11 +1400,18 @@ void av1_jnt_convolve_y_neon(const uint8_t *src, int src_stride, uint8_t *dst8,
assert((conv_params->round_1 - 2) >= bits);
if ((w == 4) || (h == 4)) {
int16x4_t s0, s1, s2, s3, s4, s5, s6, s7, s8, s9, s10, d0, d1, d2, d3;
uint16x4_t res4, res5, res6, res7;
uint32x2_t tu0, tu1, tu2, tu3;
int16x4_t s0, s1, s2, s3, s4, s5, s6, s7, d0;
uint16x4_t res4;
uint32x2_t tu0 = vdup_n_u32(0), tu1 = vdup_n_u32(0), tu2 = vdup_n_u32(0),
tu3 = vdup_n_u32(0);
int16x8_t u0, u1, u2, u3;
uint8x8_t t0;
#if defined(__aarch64__)
int16x4_t s8, s9, s10, d1, d2, d3;
uint16x4_t res5, res6, res7;
uint8x8_t t1;
#endif
const int16x4_t round_offset64 = vdup_n_s16(round_offset);
const int16x4_t shift_vec = vdup_n_s16(-shift_value);
const int16x4_t zero = vdup_n_s16(0);
@ -1111,6 +1448,7 @@ void av1_jnt_convolve_y_neon(const uint8_t *src, int src_stride, uint8_t *dst8,
s += (7 * src_stride);
do {
#if defined(__aarch64__)
load_unaligned_u8_4x4(s, src_stride, &tu0, &tu1);
u0 = vreinterpretq_s16_u16(vmovl_u8(vreinterpret_u8_u32(tu0)));
@ -1154,17 +1492,13 @@ void av1_jnt_convolve_y_neon(const uint8_t *src, int src_stride, uint8_t *dst8,
round_offset64, round_bits, use_jnt_comp_avg, &t0,
&t1);
vst1_lane_u32((uint32_t *)d_u8, vreinterpret_u32_u8(t0),
0); // 00 01 02 03
vst1_lane_u32((uint32_t *)d_u8, vreinterpret_u32_u8(t0), 0);
d_u8 += dst8_stride;
vst1_lane_u32((uint32_t *)d_u8, vreinterpret_u32_u8(t0),
1); // 10 11 12 13
vst1_lane_u32((uint32_t *)d_u8, vreinterpret_u32_u8(t0), 1);
d_u8 += dst8_stride;
vst1_lane_u32((uint32_t *)d_u8, vreinterpret_u32_u8(t1),
0); // 20 21 22 23
vst1_lane_u32((uint32_t *)d_u8, vreinterpret_u32_u8(t1), 0);
d_u8 += dst8_stride;
vst1_lane_u32((uint32_t *)d_u8, vreinterpret_u32_u8(t1),
1); // 30 31 32 33
vst1_lane_u32((uint32_t *)d_u8, vreinterpret_u32_u8(t1), 1);
d_u8 += dst8_stride;
} else {
store_u16_4x4(d, dst_stride, vreinterpret_u16_s16(d0),
@ -1183,6 +1517,44 @@ void av1_jnt_convolve_y_neon(const uint8_t *src, int src_stride, uint8_t *dst8,
s += (src_stride << 2);
height -= 4;
#else
load_unaligned_u8_4x1(s, src_stride, &tu0);
u0 = vreinterpretq_s16_u16(vmovl_u8(vreinterpret_u8_u32(tu0)));
s7 = vget_low_s16(u0);
d0 = convolve8_4x4_s16(s0, s1, s2, s3, s4, s5, s6, s7, y_filter_tmp,
zero, shift_vec);
d0 = vadd_s16(d0, round_offset64);
if (conv_params->do_average) {
__builtin_prefetch(d);
res4 = vld1_u16(d);
d += (dst_stride);
compute_avg_4x1(res4, vreinterpret_u16_s16(d0), fwd_offset,
bck_offset, round_offset64, round_bits,
use_jnt_comp_avg, &t0);
vst1_lane_u32((uint32_t *)d_u8, vreinterpret_u32_u8(t0), 0);
d_u8 += dst8_stride;
} else {
vst1_u16(d, vreinterpret_u16_s16(d0));
d += (dst_stride);
}
s0 = s1;
s1 = s2;
s2 = s3;
s3 = s4;
s4 = s5;
s5 = s6;
s6 = s7;
s += (src_stride);
height--;
#endif
} while (height > 0);
src_ptr += 4;
dst_ptr += 4;
@ -1191,15 +1563,19 @@ void av1_jnt_convolve_y_neon(const uint8_t *src, int src_stride, uint8_t *dst8,
} while (width > 0);
} else {
CONV_BUF_TYPE *d_tmp;
int16x8_t s11, s12, s13, s14;
int16x8_t s0, s1, s2, s3, s4, s5, s6, s7, s8, s9, s10;
int16x8_t res0, res1, res2, res3, res4, res5, res6, res7;
uint16x8_t res8, res9, res10, res11;
int16x8_t s0, s1, s2, s3, s4, s5, s6, s7;
int16x8_t res0;
uint16x8_t res8;
uint8x8_t t0, t1, t2, t3, t4, t5, t6, t7;
const int16x8_t round_offset128 = vdupq_n_s16(round_offset);
const int16x8_t shift_vec = vdupq_n_s16(-shift_value);
const int16x4_t round_offset64 = vdup_n_s16(round_offset);
const int16x8_t zero = vdupq_n_s16(0);
#if defined(__aarch64__)
int16x8_t s8, s9, s10, s11, s12, s13, s14;
int16x8_t res1, res2, res3, res4, res5, res6, res7;
uint16x8_t res10, res11, res9;
#endif
dst_ptr = dst;
dst_u8_ptr = dst8;
do {
@ -1227,6 +1603,7 @@ void av1_jnt_convolve_y_neon(const uint8_t *src, int src_stride, uint8_t *dst8,
d_u8 = dst_u8_ptr;
do {
#if defined(__aarch64__)
load_u8_8x8(s, src_stride, &t0, &t1, &t2, &t3, &t4, &t5, &t6, &t7);
s7 = vreinterpretq_s16_u16(vmovl_u8(t0));
@ -1316,6 +1693,43 @@ void av1_jnt_convolve_y_neon(const uint8_t *src, int src_stride, uint8_t *dst8,
s6 = s14;
s += (8 * src_stride);
height -= 8;
#else
s7 = vreinterpretq_s16_u16(vmovl_u8(vld1_u8(s)));
__builtin_prefetch(dst_ptr);
res0 = convolve8_8x8_s16(s0, s1, s2, s3, s4, s5, s6, s7, y_filter_tmp,
zero, shift_vec);
res0 = vaddq_s16(res0, round_offset128);
s0 = s1;
s1 = s2;
s2 = s3;
s3 = s4;
s4 = s5;
s5 = s6;
s6 = s7;
if (conv_params->do_average) {
__builtin_prefetch(d_tmp);
res8 = vld1q_u16(d_tmp);
d_tmp += (dst_stride);
compute_avg_8x1(res8, vreinterpretq_u16_s16(res0), fwd_offset,
bck_offset, round_offset64, round_bits,
use_jnt_comp_avg, &t0);
vst1_u8(d_u8, t0);
d_u8 += (dst8_stride);
} else {
vst1q_u16(d_tmp, vreinterpretq_u16_s16(res0));
d_tmp += dst_stride;
}
s += (src_stride);
height--;
#endif
} while (height > 0);
src_ptr += 8;
dst_ptr += 8;

View file

@ -8,8 +8,8 @@
* be found in the AUTHORS file in the root of the source tree.
*/
#ifndef AV1_COMMON_ARM_MEM_NEON_H_
#define AV1_COMMON_ARM_MEM_NEON_H_
#ifndef AOM_AV1_COMMON_ARM_MEM_NEON_H_
#define AOM_AV1_COMMON_ARM_MEM_NEON_H_
#include <arm_neon.h>
#include <string.h>
@ -362,6 +362,15 @@ static INLINE void load_unaligned_u8_4x4(const uint8_t *buf, int stride,
*tu1 = vset_lane_u32(a, *tu1, 1);
}
static INLINE void load_unaligned_u8_4x1(const uint8_t *buf, int stride,
uint32x2_t *tu0) {
uint32_t a;
memcpy(&a, buf, 4);
buf += stride;
*tu0 = vset_lane_u32(a, *tu0, 0);
}
static INLINE void load_unaligned_u8_4x2(const uint8_t *buf, int stride,
uint32x2_t *tu0) {
uint32_t a;
@ -482,4 +491,4 @@ static INLINE void store_u32_4x4(uint32_t *s, int32_t p, uint32x4_t s1,
vst1q_u32(s, s4);
}
#endif // AV1_COMMON_ARM_MEM_NEON_H_
#endif // AOM_AV1_COMMON_ARM_MEM_NEON_H_

View file

@ -1007,10 +1007,11 @@ static INLINE void cross_sum_fast_odd_row_inp16(uint16_t *buf, int32x4_t *a0,
vaddq_u32(vmovl_u16(vget_high_u16(xl)), vmovl_u16(vget_high_u16(x))));
}
void final_filter_fast_internal(uint16_t *A, int32_t *B, const int buf_stride,
int16_t *src, const int src_stride,
int32_t *dst, const int dst_stride,
const int width, const int height) {
static void final_filter_fast_internal(uint16_t *A, int32_t *B,
const int buf_stride, int16_t *src,
const int src_stride, int32_t *dst,
const int dst_stride, const int width,
const int height) {
int16x8_t s0;
int32_t *B_tmp, *dst_ptr;
uint16_t *A_tmp;
@ -1340,10 +1341,10 @@ static INLINE void src_convert_hbd_copy(const uint16_t *src, int src_stride,
}
}
void av1_selfguided_restoration_neon(const uint8_t *dat8, int width, int height,
int stride, int32_t *flt0, int32_t *flt1,
int flt_stride, int sgr_params_idx,
int bit_depth, int highbd) {
int av1_selfguided_restoration_neon(const uint8_t *dat8, int width, int height,
int stride, int32_t *flt0, int32_t *flt1,
int flt_stride, int sgr_params_idx,
int bit_depth, int highbd) {
const sgr_params_type *const params = &sgr_params[sgr_params_idx];
assert(!(params->r[0] == 0 && params->r[1] == 0));
@ -1376,6 +1377,7 @@ void av1_selfguided_restoration_neon(const uint8_t *dat8, int width, int height,
if (params->r[1] > 0)
restoration_internal(dgd16, width, height, dgd16_stride, flt1, flt_stride,
bit_depth, sgr_params_idx, 1);
return 0;
}
void apply_selfguided_restoration_neon(const uint8_t *dat8, int width,

View file

@ -8,8 +8,8 @@
* be found in the AUTHORS file in the root of the source tree.
*/
#ifndef AV1_COMMON_ARM_TRANSPOSE_NEON_H_
#define AV1_COMMON_ARM_TRANSPOSE_NEON_H_
#ifndef AOM_AV1_COMMON_ARM_TRANSPOSE_NEON_H_
#define AOM_AV1_COMMON_ARM_TRANSPOSE_NEON_H_
#include <arm_neon.h>
@ -386,6 +386,83 @@ static INLINE void transpose_s16_8x8(int16x8_t *a0, int16x8_t *a1,
vget_high_s16(vreinterpretq_s16_s32(c3.val[1])));
}
static INLINE int16x8x2_t vpx_vtrnq_s64_to_s16(int32x4_t a0, int32x4_t a1) {
int16x8x2_t b0;
b0.val[0] = vcombine_s16(vreinterpret_s16_s32(vget_low_s32(a0)),
vreinterpret_s16_s32(vget_low_s32(a1)));
b0.val[1] = vcombine_s16(vreinterpret_s16_s32(vget_high_s32(a0)),
vreinterpret_s16_s32(vget_high_s32(a1)));
return b0;
}
static INLINE void transpose_s16_8x8q(int16x8_t *a0, int16x8_t *out) {
// Swap 16 bit elements. Goes from:
// a0: 00 01 02 03 04 05 06 07
// a1: 10 11 12 13 14 15 16 17
// a2: 20 21 22 23 24 25 26 27
// a3: 30 31 32 33 34 35 36 37
// a4: 40 41 42 43 44 45 46 47
// a5: 50 51 52 53 54 55 56 57
// a6: 60 61 62 63 64 65 66 67
// a7: 70 71 72 73 74 75 76 77
// to:
// b0.val[0]: 00 10 02 12 04 14 06 16
// b0.val[1]: 01 11 03 13 05 15 07 17
// b1.val[0]: 20 30 22 32 24 34 26 36
// b1.val[1]: 21 31 23 33 25 35 27 37
// b2.val[0]: 40 50 42 52 44 54 46 56
// b2.val[1]: 41 51 43 53 45 55 47 57
// b3.val[0]: 60 70 62 72 64 74 66 76
// b3.val[1]: 61 71 63 73 65 75 67 77
const int16x8x2_t b0 = vtrnq_s16(*a0, *(a0 + 1));
const int16x8x2_t b1 = vtrnq_s16(*(a0 + 2), *(a0 + 3));
const int16x8x2_t b2 = vtrnq_s16(*(a0 + 4), *(a0 + 5));
const int16x8x2_t b3 = vtrnq_s16(*(a0 + 6), *(a0 + 7));
// Swap 32 bit elements resulting in:
// c0.val[0]: 00 10 20 30 04 14 24 34
// c0.val[1]: 02 12 22 32 06 16 26 36
// c1.val[0]: 01 11 21 31 05 15 25 35
// c1.val[1]: 03 13 23 33 07 17 27 37
// c2.val[0]: 40 50 60 70 44 54 64 74
// c2.val[1]: 42 52 62 72 46 56 66 76
// c3.val[0]: 41 51 61 71 45 55 65 75
// c3.val[1]: 43 53 63 73 47 57 67 77
const int32x4x2_t c0 = vtrnq_s32(vreinterpretq_s32_s16(b0.val[0]),
vreinterpretq_s32_s16(b1.val[0]));
const int32x4x2_t c1 = vtrnq_s32(vreinterpretq_s32_s16(b0.val[1]),
vreinterpretq_s32_s16(b1.val[1]));
const int32x4x2_t c2 = vtrnq_s32(vreinterpretq_s32_s16(b2.val[0]),
vreinterpretq_s32_s16(b3.val[0]));
const int32x4x2_t c3 = vtrnq_s32(vreinterpretq_s32_s16(b2.val[1]),
vreinterpretq_s32_s16(b3.val[1]));
// Swap 64 bit elements resulting in:
// d0.val[0]: 00 10 20 30 40 50 60 70
// d0.val[1]: 04 14 24 34 44 54 64 74
// d1.val[0]: 01 11 21 31 41 51 61 71
// d1.val[1]: 05 15 25 35 45 55 65 75
// d2.val[0]: 02 12 22 32 42 52 62 72
// d2.val[1]: 06 16 26 36 46 56 66 76
// d3.val[0]: 03 13 23 33 43 53 63 73
// d3.val[1]: 07 17 27 37 47 57 67 77
const int16x8x2_t d0 = vpx_vtrnq_s64_to_s16(c0.val[0], c2.val[0]);
const int16x8x2_t d1 = vpx_vtrnq_s64_to_s16(c1.val[0], c3.val[0]);
const int16x8x2_t d2 = vpx_vtrnq_s64_to_s16(c0.val[1], c2.val[1]);
const int16x8x2_t d3 = vpx_vtrnq_s64_to_s16(c1.val[1], c3.val[1]);
*out = d0.val[0];
*(out + 1) = d1.val[0];
*(out + 2) = d2.val[0];
*(out + 3) = d3.val[0];
*(out + 4) = d0.val[1];
*(out + 5) = d1.val[1];
*(out + 6) = d2.val[1];
*(out + 7) = d3.val[1];
}
static INLINE void transpose_s16_4x4d(int16x4_t *a0, int16x4_t *a1,
int16x4_t *a2, int16x4_t *a3) {
// Swap 16 bit elements. Goes from:
@ -457,4 +534,4 @@ static INLINE void transpose_s32_4x4(int32x4_t *a0, int32x4_t *a1,
*a3 = c1.val[1];
}
#endif // AV1_COMMON_ARM_TRANSPOSE_NEON_H_
#endif // AOM_AV1_COMMON_ARM_TRANSPOSE_NEON_H_

View file

@ -0,0 +1,714 @@
/*
* Copyright (c) 2018, Alliance for Open Media. All rights reserved
*
* This source code is subject to the terms of the BSD 2 Clause License and
* the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
* was not distributed with this source code in the LICENSE file, you can
* obtain it at www.aomedia.org/license/software. If the Alliance for Open
* Media Patent License 1.0 was not distributed with this source code in the
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#include <assert.h>
#include <arm_neon.h>
#include <memory.h>
#include <math.h>
#include "aom_dsp/aom_dsp_common.h"
#include "aom_ports/mem.h"
#include "config/av1_rtcd.h"
#include "av1/common/warped_motion.h"
#include "av1/common/scale.h"
/* This is a modified version of 'warped_filter' from warped_motion.c:
* Each coefficient is stored in 8 bits instead of 16 bits
* The coefficients are rearranged in the column order 0, 2, 4, 6, 1, 3, 5, 7
This is done in order to avoid overflow: Since the tap with the largest
coefficient could be any of taps 2, 3, 4 or 5, we can't use the summation
order ((0 + 1) + (4 + 5)) + ((2 + 3) + (6 + 7)) used in the regular
convolve functions.
Instead, we use the summation order
((0 + 2) + (4 + 6)) + ((1 + 3) + (5 + 7)).
The rearrangement of coefficients in this table is so that we can get the
coefficients into the correct order more quickly.
*/
/* clang-format off */
DECLARE_ALIGNED(8, static const int8_t,
filter_8bit_neon[WARPEDPIXEL_PREC_SHIFTS * 3 + 1][8]) = {
#if WARPEDPIXEL_PREC_BITS == 6
// [-1, 0)
{ 0, 127, 0, 0, 0, 1, 0, 0}, { 0, 127, 0, 0, -1, 2, 0, 0},
{ 1, 127, -1, 0, -3, 4, 0, 0}, { 1, 126, -2, 0, -4, 6, 1, 0},
{ 1, 126, -3, 0, -5, 8, 1, 0}, { 1, 125, -4, 0, -6, 11, 1, 0},
{ 1, 124, -4, 0, -7, 13, 1, 0}, { 2, 123, -5, 0, -8, 15, 1, 0},
{ 2, 122, -6, 0, -9, 18, 1, 0}, { 2, 121, -6, 0, -10, 20, 1, 0},
{ 2, 120, -7, 0, -11, 22, 2, 0}, { 2, 119, -8, 0, -12, 25, 2, 0},
{ 3, 117, -8, 0, -13, 27, 2, 0}, { 3, 116, -9, 0, -13, 29, 2, 0},
{ 3, 114, -10, 0, -14, 32, 3, 0}, { 3, 113, -10, 0, -15, 35, 2, 0},
{ 3, 111, -11, 0, -15, 37, 3, 0}, { 3, 109, -11, 0, -16, 40, 3, 0},
{ 3, 108, -12, 0, -16, 42, 3, 0}, { 4, 106, -13, 0, -17, 45, 3, 0},
{ 4, 104, -13, 0, -17, 47, 3, 0}, { 4, 102, -14, 0, -17, 50, 3, 0},
{ 4, 100, -14, 0, -17, 52, 3, 0}, { 4, 98, -15, 0, -18, 55, 4, 0},
{ 4, 96, -15, 0, -18, 58, 3, 0}, { 4, 94, -16, 0, -18, 60, 4, 0},
{ 4, 91, -16, 0, -18, 63, 4, 0}, { 4, 89, -16, 0, -18, 65, 4, 0},
{ 4, 87, -17, 0, -18, 68, 4, 0}, { 4, 85, -17, 0, -18, 70, 4, 0},
{ 4, 82, -17, 0, -18, 73, 4, 0}, { 4, 80, -17, 0, -18, 75, 4, 0},
{ 4, 78, -18, 0, -18, 78, 4, 0}, { 4, 75, -18, 0, -17, 80, 4, 0},
{ 4, 73, -18, 0, -17, 82, 4, 0}, { 4, 70, -18, 0, -17, 85, 4, 0},
{ 4, 68, -18, 0, -17, 87, 4, 0}, { 4, 65, -18, 0, -16, 89, 4, 0},
{ 4, 63, -18, 0, -16, 91, 4, 0}, { 4, 60, -18, 0, -16, 94, 4, 0},
{ 3, 58, -18, 0, -15, 96, 4, 0}, { 4, 55, -18, 0, -15, 98, 4, 0},
{ 3, 52, -17, 0, -14, 100, 4, 0}, { 3, 50, -17, 0, -14, 102, 4, 0},
{ 3, 47, -17, 0, -13, 104, 4, 0}, { 3, 45, -17, 0, -13, 106, 4, 0},
{ 3, 42, -16, 0, -12, 108, 3, 0}, { 3, 40, -16, 0, -11, 109, 3, 0},
{ 3, 37, -15, 0, -11, 111, 3, 0}, { 2, 35, -15, 0, -10, 113, 3, 0},
{ 3, 32, -14, 0, -10, 114, 3, 0}, { 2, 29, -13, 0, -9, 116, 3, 0},
{ 2, 27, -13, 0, -8, 117, 3, 0}, { 2, 25, -12, 0, -8, 119, 2, 0},
{ 2, 22, -11, 0, -7, 120, 2, 0}, { 1, 20, -10, 0, -6, 121, 2, 0},
{ 1, 18, -9, 0, -6, 122, 2, 0}, { 1, 15, -8, 0, -5, 123, 2, 0},
{ 1, 13, -7, 0, -4, 124, 1, 0}, { 1, 11, -6, 0, -4, 125, 1, 0},
{ 1, 8, -5, 0, -3, 126, 1, 0}, { 1, 6, -4, 0, -2, 126, 1, 0},
{ 0, 4, -3, 0, -1, 127, 1, 0}, { 0, 2, -1, 0, 0, 127, 0, 0},
// [0, 1)
{ 0, 0, 1, 0, 0, 127, 0, 0}, { 0, -1, 2, 0, 0, 127, 0, 0},
{ 0, -3, 4, 1, 1, 127, -2, 0}, { 0, -5, 6, 1, 1, 127, -2, 0},
{ 0, -6, 8, 1, 2, 126, -3, 0}, {-1, -7, 11, 2, 2, 126, -4, -1},
{-1, -8, 13, 2, 3, 125, -5, -1}, {-1, -10, 16, 3, 3, 124, -6, -1},
{-1, -11, 18, 3, 4, 123, -7, -1}, {-1, -12, 20, 3, 4, 122, -7, -1},
{-1, -13, 23, 3, 4, 121, -8, -1}, {-2, -14, 25, 4, 5, 120, -9, -1},
{-1, -15, 27, 4, 5, 119, -10, -1}, {-1, -16, 30, 4, 5, 118, -11, -1},
{-2, -17, 33, 5, 6, 116, -12, -1}, {-2, -17, 35, 5, 6, 114, -12, -1},
{-2, -18, 38, 5, 6, 113, -13, -1}, {-2, -19, 41, 6, 7, 111, -14, -2},
{-2, -19, 43, 6, 7, 110, -15, -2}, {-2, -20, 46, 6, 7, 108, -15, -2},
{-2, -20, 49, 6, 7, 106, -16, -2}, {-2, -21, 51, 7, 7, 104, -16, -2},
{-2, -21, 54, 7, 7, 102, -17, -2}, {-2, -21, 56, 7, 8, 100, -18, -2},
{-2, -22, 59, 7, 8, 98, -18, -2}, {-2, -22, 62, 7, 8, 96, -19, -2},
{-2, -22, 64, 7, 8, 94, -19, -2}, {-2, -22, 67, 8, 8, 91, -20, -2},
{-2, -22, 69, 8, 8, 89, -20, -2}, {-2, -22, 72, 8, 8, 87, -21, -2},
{-2, -21, 74, 8, 8, 84, -21, -2}, {-2, -22, 77, 8, 8, 82, -21, -2},
{-2, -21, 79, 8, 8, 79, -21, -2}, {-2, -21, 82, 8, 8, 77, -22, -2},
{-2, -21, 84, 8, 8, 74, -21, -2}, {-2, -21, 87, 8, 8, 72, -22, -2},
{-2, -20, 89, 8, 8, 69, -22, -2}, {-2, -20, 91, 8, 8, 67, -22, -2},
{-2, -19, 94, 8, 7, 64, -22, -2}, {-2, -19, 96, 8, 7, 62, -22, -2},
{-2, -18, 98, 8, 7, 59, -22, -2}, {-2, -18, 100, 8, 7, 56, -21, -2},
{-2, -17, 102, 7, 7, 54, -21, -2}, {-2, -16, 104, 7, 7, 51, -21, -2},
{-2, -16, 106, 7, 6, 49, -20, -2}, {-2, -15, 108, 7, 6, 46, -20, -2},
{-2, -15, 110, 7, 6, 43, -19, -2}, {-2, -14, 111, 7, 6, 41, -19, -2},
{-1, -13, 113, 6, 5, 38, -18, -2}, {-1, -12, 114, 6, 5, 35, -17, -2},
{-1, -12, 116, 6, 5, 33, -17, -2}, {-1, -11, 118, 5, 4, 30, -16, -1},
{-1, -10, 119, 5, 4, 27, -15, -1}, {-1, -9, 120, 5, 4, 25, -14, -2},
{-1, -8, 121, 4, 3, 23, -13, -1}, {-1, -7, 122, 4, 3, 20, -12, -1},
{-1, -7, 123, 4, 3, 18, -11, -1}, {-1, -6, 124, 3, 3, 16, -10, -1},
{-1, -5, 125, 3, 2, 13, -8, -1}, {-1, -4, 126, 2, 2, 11, -7, -1},
{ 0, -3, 126, 2, 1, 8, -6, 0}, { 0, -2, 127, 1, 1, 6, -5, 0},
{ 0, -2, 127, 1, 1, 4, -3, 0}, { 0, 0, 127, 0, 0, 2, -1, 0},
// [1, 2)
{ 0, 0, 127, 0, 0, 1, 0, 0}, { 0, 0, 127, 0, 0, -1, 2, 0},
{ 0, 1, 127, -1, 0, -3, 4, 0}, { 0, 1, 126, -2, 0, -4, 6, 1},
{ 0, 1, 126, -3, 0, -5, 8, 1}, { 0, 1, 125, -4, 0, -6, 11, 1},
{ 0, 1, 124, -4, 0, -7, 13, 1}, { 0, 2, 123, -5, 0, -8, 15, 1},
{ 0, 2, 122, -6, 0, -9, 18, 1}, { 0, 2, 121, -6, 0, -10, 20, 1},
{ 0, 2, 120, -7, 0, -11, 22, 2}, { 0, 2, 119, -8, 0, -12, 25, 2},
{ 0, 3, 117, -8, 0, -13, 27, 2}, { 0, 3, 116, -9, 0, -13, 29, 2},
{ 0, 3, 114, -10, 0, -14, 32, 3}, { 0, 3, 113, -10, 0, -15, 35, 2},
{ 0, 3, 111, -11, 0, -15, 37, 3}, { 0, 3, 109, -11, 0, -16, 40, 3},
{ 0, 3, 108, -12, 0, -16, 42, 3}, { 0, 4, 106, -13, 0, -17, 45, 3},
{ 0, 4, 104, -13, 0, -17, 47, 3}, { 0, 4, 102, -14, 0, -17, 50, 3},
{ 0, 4, 100, -14, 0, -17, 52, 3}, { 0, 4, 98, -15, 0, -18, 55, 4},
{ 0, 4, 96, -15, 0, -18, 58, 3}, { 0, 4, 94, -16, 0, -18, 60, 4},
{ 0, 4, 91, -16, 0, -18, 63, 4}, { 0, 4, 89, -16, 0, -18, 65, 4},
{ 0, 4, 87, -17, 0, -18, 68, 4}, { 0, 4, 85, -17, 0, -18, 70, 4},
{ 0, 4, 82, -17, 0, -18, 73, 4}, { 0, 4, 80, -17, 0, -18, 75, 4},
{ 0, 4, 78, -18, 0, -18, 78, 4}, { 0, 4, 75, -18, 0, -17, 80, 4},
{ 0, 4, 73, -18, 0, -17, 82, 4}, { 0, 4, 70, -18, 0, -17, 85, 4},
{ 0, 4, 68, -18, 0, -17, 87, 4}, { 0, 4, 65, -18, 0, -16, 89, 4},
{ 0, 4, 63, -18, 0, -16, 91, 4}, { 0, 4, 60, -18, 0, -16, 94, 4},
{ 0, 3, 58, -18, 0, -15, 96, 4}, { 0, 4, 55, -18, 0, -15, 98, 4},
{ 0, 3, 52, -17, 0, -14, 100, 4}, { 0, 3, 50, -17, 0, -14, 102, 4},
{ 0, 3, 47, -17, 0, -13, 104, 4}, { 0, 3, 45, -17, 0, -13, 106, 4},
{ 0, 3, 42, -16, 0, -12, 108, 3}, { 0, 3, 40, -16, 0, -11, 109, 3},
{ 0, 3, 37, -15, 0, -11, 111, 3}, { 0, 2, 35, -15, 0, -10, 113, 3},
{ 0, 3, 32, -14, 0, -10, 114, 3}, { 0, 2, 29, -13, 0, -9, 116, 3},
{ 0, 2, 27, -13, 0, -8, 117, 3}, { 0, 2, 25, -12, 0, -8, 119, 2},
{ 0, 2, 22, -11, 0, -7, 120, 2}, { 0, 1, 20, -10, 0, -6, 121, 2},
{ 0, 1, 18, -9, 0, -6, 122, 2}, { 0, 1, 15, -8, 0, -5, 123, 2},
{ 0, 1, 13, -7, 0, -4, 124, 1}, { 0, 1, 11, -6, 0, -4, 125, 1},
{ 0, 1, 8, -5, 0, -3, 126, 1}, { 0, 1, 6, -4, 0, -2, 126, 1},
{ 0, 0, 4, -3, 0, -1, 127, 1}, { 0, 0, 2, -1, 0, 0, 127, 0},
// dummy (replicate row index 191)
{ 0, 0, 2, -1, 0, 0, 127, 0},
#else
// [-1, 0)
{ 0, 127, 0, 0, 0, 1, 0, 0}, { 1, 127, -1, 0, -3, 4, 0, 0},
{ 1, 126, -3, 0, -5, 8, 1, 0}, { 1, 124, -4, 0, -7, 13, 1, 0},
{ 2, 122, -6, 0, -9, 18, 1, 0}, { 2, 120, -7, 0, -11, 22, 2, 0},
{ 3, 117, -8, 0, -13, 27, 2, 0}, { 3, 114, -10, 0, -14, 32, 3, 0},
{ 3, 111, -11, 0, -15, 37, 3, 0}, { 3, 108, -12, 0, -16, 42, 3, 0},
{ 4, 104, -13, 0, -17, 47, 3, 0}, { 4, 100, -14, 0, -17, 52, 3, 0},
{ 4, 96, -15, 0, -18, 58, 3, 0}, { 4, 91, -16, 0, -18, 63, 4, 0},
{ 4, 87, -17, 0, -18, 68, 4, 0}, { 4, 82, -17, 0, -18, 73, 4, 0},
{ 4, 78, -18, 0, -18, 78, 4, 0}, { 4, 73, -18, 0, -17, 82, 4, 0},
{ 4, 68, -18, 0, -17, 87, 4, 0}, { 4, 63, -18, 0, -16, 91, 4, 0},
{ 3, 58, -18, 0, -15, 96, 4, 0}, { 3, 52, -17, 0, -14, 100, 4, 0},
{ 3, 47, -17, 0, -13, 104, 4, 0}, { 3, 42, -16, 0, -12, 108, 3, 0},
{ 3, 37, -15, 0, -11, 111, 3, 0}, { 3, 32, -14, 0, -10, 114, 3, 0},
{ 2, 27, -13, 0, -8, 117, 3, 0}, { 2, 22, -11, 0, -7, 120, 2, 0},
{ 1, 18, -9, 0, -6, 122, 2, 0}, { 1, 13, -7, 0, -4, 124, 1, 0},
{ 1, 8, -5, 0, -3, 126, 1, 0}, { 0, 4, -3, 0, -1, 127, 1, 0},
// [0, 1)
{ 0, 0, 1, 0, 0, 127, 0, 0}, { 0, -3, 4, 1, 1, 127, -2, 0},
{ 0, -6, 8, 1, 2, 126, -3, 0}, {-1, -8, 13, 2, 3, 125, -5, -1},
{-1, -11, 18, 3, 4, 123, -7, -1}, {-1, -13, 23, 3, 4, 121, -8, -1},
{-1, -15, 27, 4, 5, 119, -10, -1}, {-2, -17, 33, 5, 6, 116, -12, -1},
{-2, -18, 38, 5, 6, 113, -13, -1}, {-2, -19, 43, 6, 7, 110, -15, -2},
{-2, -20, 49, 6, 7, 106, -16, -2}, {-2, -21, 54, 7, 7, 102, -17, -2},
{-2, -22, 59, 7, 8, 98, -18, -2}, {-2, -22, 64, 7, 8, 94, -19, -2},
{-2, -22, 69, 8, 8, 89, -20, -2}, {-2, -21, 74, 8, 8, 84, -21, -2},
{-2, -21, 79, 8, 8, 79, -21, -2}, {-2, -21, 84, 8, 8, 74, -21, -2},
{-2, -20, 89, 8, 8, 69, -22, -2}, {-2, -19, 94, 8, 7, 64, -22, -2},
{-2, -18, 98, 8, 7, 59, -22, -2}, {-2, -17, 102, 7, 7, 54, -21, -2},
{-2, -16, 106, 7, 6, 49, -20, -2}, {-2, -15, 110, 7, 6, 43, -19, -2},
{-1, -13, 113, 6, 5, 38, -18, -2}, {-1, -12, 116, 6, 5, 33, -17, -2},
{-1, -10, 119, 5, 4, 27, -15, -1}, {-1, -8, 121, 4, 3, 23, -13, -1},
{-1, -7, 123, 4, 3, 18, -11, -1}, {-1, -5, 125, 3, 2, 13, -8, -1},
{ 0, -3, 126, 2, 1, 8, -6, 0}, { 0, -2, 127, 1, 1, 4, -3, 0},
// [1, 2)
{ 0, 0, 127, 0, 0, 1, 0, 0}, { 0, 1, 127, -1, 0, -3, 4, 0},
{ 0, 1, 126, -3, 0, -5, 8, 1}, { 0, 1, 124, -4, 0, -7, 13, 1},
{ 0, 2, 122, -6, 0, -9, 18, 1}, { 0, 2, 120, -7, 0, -11, 22, 2},
{ 0, 3, 117, -8, 0, -13, 27, 2}, { 0, 3, 114, -10, 0, -14, 32, 3},
{ 0, 3, 111, -11, 0, -15, 37, 3}, { 0, 3, 108, -12, 0, -16, 42, 3},
{ 0, 4, 104, -13, 0, -17, 47, 3}, { 0, 4, 100, -14, 0, -17, 52, 3},
{ 0, 4, 96, -15, 0, -18, 58, 3}, { 0, 4, 91, -16, 0, -18, 63, 4},
{ 0, 4, 87, -17, 0, -18, 68, 4}, { 0, 4, 82, -17, 0, -18, 73, 4},
{ 0, 4, 78, -18, 0, -18, 78, 4}, { 0, 4, 73, -18, 0, -17, 82, 4},
{ 0, 4, 68, -18, 0, -17, 87, 4}, { 0, 4, 63, -18, 0, -16, 91, 4},
{ 0, 3, 58, -18, 0, -15, 96, 4}, { 0, 3, 52, -17, 0, -14, 100, 4},
{ 0, 3, 47, -17, 0, -13, 104, 4}, { 0, 3, 42, -16, 0, -12, 108, 3},
{ 0, 3, 37, -15, 0, -11, 111, 3}, { 0, 3, 32, -14, 0, -10, 114, 3},
{ 0, 2, 27, -13, 0, -8, 117, 3}, { 0, 2, 22, -11, 0, -7, 120, 2},
{ 0, 1, 18, -9, 0, -6, 122, 2}, { 0, 1, 13, -7, 0, -4, 124, 1},
{ 0, 1, 8, -5, 0, -3, 126, 1}, { 0, 0, 4, -3, 0, -1, 127, 1},
// dummy (replicate row index 95)
{ 0, 0, 4, -3, 0, -1, 127, 1},
#endif // WARPEDPIXEL_PREC_BITS == 6
};
/* clang-format on */
static INLINE void convolve(int32x2x2_t x0, int32x2x2_t x1, uint8x8_t src_0,
uint8x8_t src_1, int16x4_t *res) {
int16x8_t coeff_0, coeff_1;
int16x8_t pix_0, pix_1;
coeff_0 = vcombine_s16(vreinterpret_s16_s32(x0.val[0]),
vreinterpret_s16_s32(x1.val[0]));
coeff_1 = vcombine_s16(vreinterpret_s16_s32(x0.val[1]),
vreinterpret_s16_s32(x1.val[1]));
pix_0 = vreinterpretq_s16_u16(vmovl_u8(src_0));
pix_0 = vmulq_s16(coeff_0, pix_0);
pix_1 = vreinterpretq_s16_u16(vmovl_u8(src_1));
pix_0 = vmlaq_s16(pix_0, coeff_1, pix_1);
*res = vpadd_s16(vget_low_s16(pix_0), vget_high_s16(pix_0));
}
static INLINE void horizontal_filter_neon(uint8x16_t src_1, uint8x16_t src_2,
uint8x16_t src_3, uint8x16_t src_4,
int16x8_t *tmp_dst, int sx, int alpha,
int k, const int offset_bits_horiz,
const int reduce_bits_horiz) {
const uint8x16_t mask = { 255, 0, 255, 0, 255, 0, 255, 0,
255, 0, 255, 0, 255, 0, 255, 0 };
const int32x4_t add_const = vdupq_n_s32((int32_t)(1 << offset_bits_horiz));
const int16x8_t shift = vdupq_n_s16(-(int16_t)reduce_bits_horiz);
int16x8_t f0, f1, f2, f3, f4, f5, f6, f7;
int32x2x2_t b0, b1;
uint8x8_t src_1_low, src_2_low, src_3_low, src_4_low, src_5_low, src_6_low;
int32x4_t tmp_res_low, tmp_res_high;
uint16x8_t res;
int16x4_t res_0246_even, res_0246_odd, res_1357_even, res_1357_odd;
uint8x16_t tmp_0 = vandq_u8(src_1, mask);
uint8x16_t tmp_1 = vandq_u8(src_2, mask);
uint8x16_t tmp_2 = vandq_u8(src_3, mask);
uint8x16_t tmp_3 = vandq_u8(src_4, mask);
tmp_2 = vextq_u8(tmp_0, tmp_0, 1);
tmp_3 = vextq_u8(tmp_1, tmp_1, 1);
src_1 = vaddq_u8(tmp_0, tmp_2);
src_2 = vaddq_u8(tmp_1, tmp_3);
src_1_low = vget_low_u8(src_1);
src_2_low = vget_low_u8(src_2);
src_3_low = vget_low_u8(vextq_u8(src_1, src_1, 4));
src_4_low = vget_low_u8(vextq_u8(src_2, src_2, 4));
src_5_low = vget_low_u8(vextq_u8(src_1, src_1, 2));
src_6_low = vget_low_u8(vextq_u8(src_1, src_1, 6));
// Loading the 8 filter taps
f0 = vmovl_s8(
vld1_s8(filter_8bit_neon[(sx + 0 * alpha) >> WARPEDDIFF_PREC_BITS]));
f1 = vmovl_s8(
vld1_s8(filter_8bit_neon[(sx + 1 * alpha) >> WARPEDDIFF_PREC_BITS]));
f2 = vmovl_s8(
vld1_s8(filter_8bit_neon[(sx + 2 * alpha) >> WARPEDDIFF_PREC_BITS]));
f3 = vmovl_s8(
vld1_s8(filter_8bit_neon[(sx + 3 * alpha) >> WARPEDDIFF_PREC_BITS]));
f4 = vmovl_s8(
vld1_s8(filter_8bit_neon[(sx + 4 * alpha) >> WARPEDDIFF_PREC_BITS]));
f5 = vmovl_s8(
vld1_s8(filter_8bit_neon[(sx + 5 * alpha) >> WARPEDDIFF_PREC_BITS]));
f6 = vmovl_s8(
vld1_s8(filter_8bit_neon[(sx + 6 * alpha) >> WARPEDDIFF_PREC_BITS]));
f7 = vmovl_s8(
vld1_s8(filter_8bit_neon[(sx + 7 * alpha) >> WARPEDDIFF_PREC_BITS]));
b0 = vtrn_s32(vreinterpret_s32_s16(vget_low_s16(f0)),
vreinterpret_s32_s16(vget_low_s16(f2)));
b1 = vtrn_s32(vreinterpret_s32_s16(vget_low_s16(f4)),
vreinterpret_s32_s16(vget_low_s16(f6)));
convolve(b0, b1, src_1_low, src_3_low, &res_0246_even);
b0 = vtrn_s32(vreinterpret_s32_s16(vget_low_s16(f1)),
vreinterpret_s32_s16(vget_low_s16(f3)));
b1 = vtrn_s32(vreinterpret_s32_s16(vget_low_s16(f5)),
vreinterpret_s32_s16(vget_low_s16(f7)));
convolve(b0, b1, src_2_low, src_4_low, &res_0246_odd);
b0 = vtrn_s32(vreinterpret_s32_s16(vget_high_s16(f0)),
vreinterpret_s32_s16(vget_high_s16(f2)));
b1 = vtrn_s32(vreinterpret_s32_s16(vget_high_s16(f4)),
vreinterpret_s32_s16(vget_high_s16(f6)));
convolve(b0, b1, src_2_low, src_4_low, &res_1357_even);
b0 = vtrn_s32(vreinterpret_s32_s16(vget_high_s16(f1)),
vreinterpret_s32_s16(vget_high_s16(f3)));
b1 = vtrn_s32(vreinterpret_s32_s16(vget_high_s16(f5)),
vreinterpret_s32_s16(vget_high_s16(f7)));
convolve(b0, b1, src_5_low, src_6_low, &res_1357_odd);
tmp_res_low = vaddl_s16(res_0246_even, res_1357_even);
tmp_res_high = vaddl_s16(res_0246_odd, res_1357_odd);
tmp_res_low = vaddq_s32(tmp_res_low, add_const);
tmp_res_high = vaddq_s32(tmp_res_high, add_const);
res = vcombine_u16(vqmovun_s32(tmp_res_low), vqmovun_s32(tmp_res_high));
res = vqrshlq_u16(res, shift);
tmp_dst[k + 7] = vreinterpretq_s16_u16(res);
}
static INLINE void vertical_filter_neon(const int16x8_t *src,
int32x4_t *res_low, int32x4_t *res_high,
int sy, int gamma) {
int16x4_t src_0, src_1, fltr_0, fltr_1;
int32x4_t res_0, res_1;
int32x2_t res_0_im, res_1_im;
int32x4_t res_even, res_odd, im_res_0, im_res_1;
int16x8_t f0, f1, f2, f3, f4, f5, f6, f7;
int16x8x2_t b0, b1, b2, b3;
int32x4x2_t c0, c1, c2, c3;
int32x4x2_t d0, d1, d2, d3;
b0 = vtrnq_s16(src[0], src[1]);
b1 = vtrnq_s16(src[2], src[3]);
b2 = vtrnq_s16(src[4], src[5]);
b3 = vtrnq_s16(src[6], src[7]);
c0 = vtrnq_s32(vreinterpretq_s32_s16(b0.val[0]),
vreinterpretq_s32_s16(b0.val[1]));
c1 = vtrnq_s32(vreinterpretq_s32_s16(b1.val[0]),
vreinterpretq_s32_s16(b1.val[1]));
c2 = vtrnq_s32(vreinterpretq_s32_s16(b2.val[0]),
vreinterpretq_s32_s16(b2.val[1]));
c3 = vtrnq_s32(vreinterpretq_s32_s16(b3.val[0]),
vreinterpretq_s32_s16(b3.val[1]));
f0 = vld1q_s16(
(int16_t *)(warped_filter + ((sy + 0 * gamma) >> WARPEDDIFF_PREC_BITS)));
f1 = vld1q_s16(
(int16_t *)(warped_filter + ((sy + 1 * gamma) >> WARPEDDIFF_PREC_BITS)));
f2 = vld1q_s16(
(int16_t *)(warped_filter + ((sy + 2 * gamma) >> WARPEDDIFF_PREC_BITS)));
f3 = vld1q_s16(
(int16_t *)(warped_filter + ((sy + 3 * gamma) >> WARPEDDIFF_PREC_BITS)));
f4 = vld1q_s16(
(int16_t *)(warped_filter + ((sy + 4 * gamma) >> WARPEDDIFF_PREC_BITS)));
f5 = vld1q_s16(
(int16_t *)(warped_filter + ((sy + 5 * gamma) >> WARPEDDIFF_PREC_BITS)));
f6 = vld1q_s16(
(int16_t *)(warped_filter + ((sy + 6 * gamma) >> WARPEDDIFF_PREC_BITS)));
f7 = vld1q_s16(
(int16_t *)(warped_filter + ((sy + 7 * gamma) >> WARPEDDIFF_PREC_BITS)));
d0 = vtrnq_s32(vreinterpretq_s32_s16(f0), vreinterpretq_s32_s16(f2));
d1 = vtrnq_s32(vreinterpretq_s32_s16(f4), vreinterpretq_s32_s16(f6));
d2 = vtrnq_s32(vreinterpretq_s32_s16(f1), vreinterpretq_s32_s16(f3));
d3 = vtrnq_s32(vreinterpretq_s32_s16(f5), vreinterpretq_s32_s16(f7));
// row:0,1 even_col:0,2
src_0 = vget_low_s16(vreinterpretq_s16_s32(c0.val[0]));
fltr_0 = vget_low_s16(vreinterpretq_s16_s32(d0.val[0]));
res_0 = vmull_s16(src_0, fltr_0);
// row:0,1,2,3 even_col:0,2
src_0 = vget_low_s16(vreinterpretq_s16_s32(c1.val[0]));
fltr_0 = vget_low_s16(vreinterpretq_s16_s32(d0.val[1]));
res_0 = vmlal_s16(res_0, src_0, fltr_0);
res_0_im = vpadd_s32(vget_low_s32(res_0), vget_high_s32(res_0));
// row:0,1 even_col:4,6
src_1 = vget_low_s16(vreinterpretq_s16_s32(c0.val[1]));
fltr_1 = vget_low_s16(vreinterpretq_s16_s32(d1.val[0]));
res_1 = vmull_s16(src_1, fltr_1);
// row:0,1,2,3 even_col:4,6
src_1 = vget_low_s16(vreinterpretq_s16_s32(c1.val[1]));
fltr_1 = vget_low_s16(vreinterpretq_s16_s32(d1.val[1]));
res_1 = vmlal_s16(res_1, src_1, fltr_1);
res_1_im = vpadd_s32(vget_low_s32(res_1), vget_high_s32(res_1));
// row:0,1,2,3 even_col:0,2,4,6
im_res_0 = vcombine_s32(res_0_im, res_1_im);
// row:4,5 even_col:0,2
src_0 = vget_low_s16(vreinterpretq_s16_s32(c2.val[0]));
fltr_0 = vget_high_s16(vreinterpretq_s16_s32(d0.val[0]));
res_0 = vmull_s16(src_0, fltr_0);
// row:4,5,6,7 even_col:0,2
src_0 = vget_low_s16(vreinterpretq_s16_s32(c3.val[0]));
fltr_0 = vget_high_s16(vreinterpretq_s16_s32(d0.val[1]));
res_0 = vmlal_s16(res_0, src_0, fltr_0);
res_0_im = vpadd_s32(vget_low_s32(res_0), vget_high_s32(res_0));
// row:4,5 even_col:4,6
src_1 = vget_low_s16(vreinterpretq_s16_s32(c2.val[1]));
fltr_1 = vget_high_s16(vreinterpretq_s16_s32(d1.val[0]));
res_1 = vmull_s16(src_1, fltr_1);
// row:4,5,6,7 even_col:4,6
src_1 = vget_low_s16(vreinterpretq_s16_s32(c3.val[1]));
fltr_1 = vget_high_s16(vreinterpretq_s16_s32(d1.val[1]));
res_1 = vmlal_s16(res_1, src_1, fltr_1);
res_1_im = vpadd_s32(vget_low_s32(res_1), vget_high_s32(res_1));
// row:4,5,6,7 even_col:0,2,4,6
im_res_1 = vcombine_s32(res_0_im, res_1_im);
// row:0-7 even_col:0,2,4,6
res_even = vaddq_s32(im_res_0, im_res_1);
// row:0,1 odd_col:1,3
src_0 = vget_high_s16(vreinterpretq_s16_s32(c0.val[0]));
fltr_0 = vget_low_s16(vreinterpretq_s16_s32(d2.val[0]));
res_0 = vmull_s16(src_0, fltr_0);
// row:0,1,2,3 odd_col:1,3
src_0 = vget_high_s16(vreinterpretq_s16_s32(c1.val[0]));
fltr_0 = vget_low_s16(vreinterpretq_s16_s32(d2.val[1]));
res_0 = vmlal_s16(res_0, src_0, fltr_0);
res_0_im = vpadd_s32(vget_low_s32(res_0), vget_high_s32(res_0));
// row:0,1 odd_col:5,7
src_1 = vget_high_s16(vreinterpretq_s16_s32(c0.val[1]));
fltr_1 = vget_low_s16(vreinterpretq_s16_s32(d3.val[0]));
res_1 = vmull_s16(src_1, fltr_1);
// row:0,1,2,3 odd_col:5,7
src_1 = vget_high_s16(vreinterpretq_s16_s32(c1.val[1]));
fltr_1 = vget_low_s16(vreinterpretq_s16_s32(d3.val[1]));
res_1 = vmlal_s16(res_1, src_1, fltr_1);
res_1_im = vpadd_s32(vget_low_s32(res_1), vget_high_s32(res_1));
// row:0,1,2,3 odd_col:1,3,5,7
im_res_0 = vcombine_s32(res_0_im, res_1_im);
// row:4,5 odd_col:1,3
src_0 = vget_high_s16(vreinterpretq_s16_s32(c2.val[0]));
fltr_0 = vget_high_s16(vreinterpretq_s16_s32(d2.val[0]));
res_0 = vmull_s16(src_0, fltr_0);
// row:4,5,6,7 odd_col:1,3
src_0 = vget_high_s16(vreinterpretq_s16_s32(c3.val[0]));
fltr_0 = vget_high_s16(vreinterpretq_s16_s32(d2.val[1]));
res_0 = vmlal_s16(res_0, src_0, fltr_0);
res_0_im = vpadd_s32(vget_low_s32(res_0), vget_high_s32(res_0));
// row:4,5 odd_col:5,7
src_1 = vget_high_s16(vreinterpretq_s16_s32(c2.val[1]));
fltr_1 = vget_high_s16(vreinterpretq_s16_s32(d3.val[0]));
res_1 = vmull_s16(src_1, fltr_1);
// row:4,5,6,7 odd_col:5,7
src_1 = vget_high_s16(vreinterpretq_s16_s32(c3.val[1]));
fltr_1 = vget_high_s16(vreinterpretq_s16_s32(d3.val[1]));
res_1 = vmlal_s16(res_1, src_1, fltr_1);
res_1_im = vpadd_s32(vget_low_s32(res_1), vget_high_s32(res_1));
// row:4,5,6,7 odd_col:1,3,5,7
im_res_1 = vcombine_s32(res_0_im, res_1_im);
// row:0-7 odd_col:1,3,5,7
res_odd = vaddq_s32(im_res_0, im_res_1);
// reordering as 0 1 2 3 | 4 5 6 7
c0 = vtrnq_s32(res_even, res_odd);
// Final store
*res_low = vcombine_s32(vget_low_s32(c0.val[0]), vget_low_s32(c0.val[1]));
*res_high = vcombine_s32(vget_high_s32(c0.val[0]), vget_high_s32(c0.val[1]));
}
void av1_warp_affine_neon(const int32_t *mat, const uint8_t *ref, int width,
int height, int stride, uint8_t *pred, int p_col,
int p_row, int p_width, int p_height, int p_stride,
int subsampling_x, int subsampling_y,
ConvolveParams *conv_params, int16_t alpha,
int16_t beta, int16_t gamma, int16_t delta) {
int16x8_t tmp[15];
const int bd = 8;
const int w0 = conv_params->fwd_offset;
const int w1 = conv_params->bck_offset;
const int32x4_t fwd = vdupq_n_s32((int32_t)w0);
const int32x4_t bwd = vdupq_n_s32((int32_t)w1);
const int16x8_t sub_constant = vdupq_n_s16((1 << (bd - 1)) + (1 << bd));
int limit = 0;
uint8x16_t vec_dup, mask_val;
int32x4_t res_lo, res_hi;
int16x8_t result_final;
uint8x16_t src_1, src_2, src_3, src_4;
uint8x16_t indx_vec = {
0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15
};
uint8x16_t cmp_vec;
const int reduce_bits_horiz = conv_params->round_0;
const int reduce_bits_vert = conv_params->is_compound
? conv_params->round_1
: 2 * FILTER_BITS - reduce_bits_horiz;
const int32x4_t shift_vert = vdupq_n_s32(-(int32_t)reduce_bits_vert);
const int offset_bits_horiz = bd + FILTER_BITS - 1;
assert(IMPLIES(conv_params->is_compound, conv_params->dst != NULL));
const int offset_bits_vert = bd + 2 * FILTER_BITS - reduce_bits_horiz;
int32x4_t add_const_vert = vdupq_n_s32((int32_t)(1 << offset_bits_vert));
const int round_bits =
2 * FILTER_BITS - conv_params->round_0 - conv_params->round_1;
const int16x4_t round_bits_vec = vdup_n_s16(-(int16_t)round_bits);
const int offset_bits = bd + 2 * FILTER_BITS - conv_params->round_0;
const int16x4_t res_sub_const =
vdup_n_s16(-((1 << (offset_bits - conv_params->round_1)) +
(1 << (offset_bits - conv_params->round_1 - 1))));
int k;
assert(IMPLIES(conv_params->do_average, conv_params->is_compound));
for (int i = 0; i < p_height; i += 8) {
for (int j = 0; j < p_width; j += 8) {
const int32_t src_x = (p_col + j + 4) << subsampling_x;
const int32_t src_y = (p_row + i + 4) << subsampling_y;
const int32_t dst_x = mat[2] * src_x + mat[3] * src_y + mat[0];
const int32_t dst_y = mat[4] * src_x + mat[5] * src_y + mat[1];
const int32_t x4 = dst_x >> subsampling_x;
const int32_t y4 = dst_y >> subsampling_y;
int32_t ix4 = x4 >> WARPEDMODEL_PREC_BITS;
int32_t sx4 = x4 & ((1 << WARPEDMODEL_PREC_BITS) - 1);
int32_t iy4 = y4 >> WARPEDMODEL_PREC_BITS;
int32_t sy4 = y4 & ((1 << WARPEDMODEL_PREC_BITS) - 1);
sx4 += alpha * (-4) + beta * (-4) + (1 << (WARPEDDIFF_PREC_BITS - 1)) +
(WARPEDPIXEL_PREC_SHIFTS << WARPEDDIFF_PREC_BITS);
sy4 += gamma * (-4) + delta * (-4) + (1 << (WARPEDDIFF_PREC_BITS - 1)) +
(WARPEDPIXEL_PREC_SHIFTS << WARPEDDIFF_PREC_BITS);
sx4 &= ~((1 << WARP_PARAM_REDUCE_BITS) - 1);
sy4 &= ~((1 << WARP_PARAM_REDUCE_BITS) - 1);
// horizontal
if (ix4 <= -7) {
for (k = -7; k < AOMMIN(8, p_height - i); ++k) {
int iy = iy4 + k;
if (iy < 0)
iy = 0;
else if (iy > height - 1)
iy = height - 1;
int16_t dup_val =
(1 << (bd + FILTER_BITS - reduce_bits_horiz - 1)) +
ref[iy * stride] * (1 << (FILTER_BITS - reduce_bits_horiz));
tmp[k + 7] = vdupq_n_s16(dup_val);
}
} else if (ix4 >= width + 6) {
for (k = -7; k < AOMMIN(8, p_height - i); ++k) {
int iy = iy4 + k;
if (iy < 0)
iy = 0;
else if (iy > height - 1)
iy = height - 1;
int16_t dup_val = (1 << (bd + FILTER_BITS - reduce_bits_horiz - 1)) +
ref[iy * stride + (width - 1)] *
(1 << (FILTER_BITS - reduce_bits_horiz));
tmp[k + 7] = vdupq_n_s16(dup_val);
}
} else if (((ix4 - 7) < 0) || ((ix4 + 9) > width)) {
const int out_of_boundary_left = -(ix4 - 6);
const int out_of_boundary_right = (ix4 + 8) - width;
for (k = -7; k < AOMMIN(8, p_height - i); ++k) {
int iy = iy4 + k;
if (iy < 0)
iy = 0;
else if (iy > height - 1)
iy = height - 1;
int sx = sx4 + beta * (k + 4);
const uint8_t *src = ref + iy * stride + ix4 - 7;
src_1 = vld1q_u8(src);
if (out_of_boundary_left >= 0) {
limit = out_of_boundary_left + 1;
cmp_vec = vdupq_n_u8(out_of_boundary_left);
vec_dup = vdupq_n_u8(*(src + limit));
mask_val = vcleq_u8(indx_vec, cmp_vec);
src_1 = vbslq_u8(mask_val, vec_dup, src_1);
}
if (out_of_boundary_right >= 0) {
limit = 15 - (out_of_boundary_right + 1);
cmp_vec = vdupq_n_u8(15 - out_of_boundary_right);
vec_dup = vdupq_n_u8(*(src + limit));
mask_val = vcgeq_u8(indx_vec, cmp_vec);
src_1 = vbslq_u8(mask_val, vec_dup, src_1);
}
src_2 = vextq_u8(src_1, src_1, 1);
src_3 = vextq_u8(src_2, src_2, 1);
src_4 = vextq_u8(src_3, src_3, 1);
horizontal_filter_neon(src_1, src_2, src_3, src_4, tmp, sx, alpha, k,
offset_bits_horiz, reduce_bits_horiz);
}
} else {
for (k = -7; k < AOMMIN(8, p_height - i); ++k) {
int iy = iy4 + k;
if (iy < 0)
iy = 0;
else if (iy > height - 1)
iy = height - 1;
int sx = sx4 + beta * (k + 4);
const uint8_t *src = ref + iy * stride + ix4 - 7;
src_1 = vld1q_u8(src);
src_2 = vextq_u8(src_1, src_1, 1);
src_3 = vextq_u8(src_2, src_2, 1);
src_4 = vextq_u8(src_3, src_3, 1);
horizontal_filter_neon(src_1, src_2, src_3, src_4, tmp, sx, alpha, k,
offset_bits_horiz, reduce_bits_horiz);
}
}
// vertical
for (k = -4; k < AOMMIN(4, p_height - i - 4); ++k) {
int sy = sy4 + delta * (k + 4);
const int16x8_t *v_src = tmp + (k + 4);
vertical_filter_neon(v_src, &res_lo, &res_hi, sy, gamma);
res_lo = vaddq_s32(res_lo, add_const_vert);
res_hi = vaddq_s32(res_hi, add_const_vert);
if (conv_params->is_compound) {
uint16_t *const p =
(uint16_t *)&conv_params
->dst[(i + k + 4) * conv_params->dst_stride + j];
res_lo = vrshlq_s32(res_lo, shift_vert);
if (conv_params->do_average) {
uint8_t *const dst8 = &pred[(i + k + 4) * p_stride + j];
uint16x4_t tmp16_lo = vld1_u16(p);
int32x4_t tmp32_lo = vreinterpretq_s32_u32(vmovl_u16(tmp16_lo));
int16x4_t tmp16_low;
if (conv_params->use_jnt_comp_avg) {
res_lo = vmulq_s32(res_lo, bwd);
tmp32_lo = vmulq_s32(tmp32_lo, fwd);
tmp32_lo = vaddq_s32(tmp32_lo, res_lo);
tmp16_low = vshrn_n_s32(tmp32_lo, DIST_PRECISION_BITS);
} else {
tmp32_lo = vaddq_s32(tmp32_lo, res_lo);
tmp16_low = vshrn_n_s32(tmp32_lo, 1);
}
int16x4_t res_low = vadd_s16(tmp16_low, res_sub_const);
res_low = vqrshl_s16(res_low, round_bits_vec);
int16x8_t final_res_low = vcombine_s16(res_low, res_low);
uint8x8_t res_8_low = vqmovun_s16(final_res_low);
vst1_lane_u32((uint32_t *)dst8, vreinterpret_u32_u8(res_8_low), 0);
} else {
uint16x4_t res_u16_low = vqmovun_s32(res_lo);
vst1_u16(p, res_u16_low);
}
if (p_width > 4) {
uint16_t *const p4 =
(uint16_t *)&conv_params
->dst[(i + k + 4) * conv_params->dst_stride + j + 4];
res_hi = vrshlq_s32(res_hi, shift_vert);
if (conv_params->do_average) {
uint8_t *const dst8_4 = &pred[(i + k + 4) * p_stride + j + 4];
uint16x4_t tmp16_hi = vld1_u16(p4);
int32x4_t tmp32_hi = vreinterpretq_s32_u32(vmovl_u16(tmp16_hi));
int16x4_t tmp16_high;
if (conv_params->use_jnt_comp_avg) {
res_hi = vmulq_s32(res_hi, bwd);
tmp32_hi = vmulq_s32(tmp32_hi, fwd);
tmp32_hi = vaddq_s32(tmp32_hi, res_hi);
tmp16_high = vshrn_n_s32(tmp32_hi, DIST_PRECISION_BITS);
} else {
tmp32_hi = vaddq_s32(tmp32_hi, res_hi);
tmp16_high = vshrn_n_s32(tmp32_hi, 1);
}
int16x4_t res_high = vadd_s16(tmp16_high, res_sub_const);
res_high = vqrshl_s16(res_high, round_bits_vec);
int16x8_t final_res_high = vcombine_s16(res_high, res_high);
uint8x8_t res_8_high = vqmovun_s16(final_res_high);
vst1_lane_u32((uint32_t *)dst8_4, vreinterpret_u32_u8(res_8_high),
0);
} else {
uint16x4_t res_u16_high = vqmovun_s32(res_hi);
vst1_u16(p4, res_u16_high);
}
}
} else {
res_lo = vrshlq_s32(res_lo, shift_vert);
res_hi = vrshlq_s32(res_hi, shift_vert);
result_final = vcombine_s16(vmovn_s32(res_lo), vmovn_s32(res_hi));
result_final = vsubq_s16(result_final, sub_constant);
uint8_t *const p = (uint8_t *)&pred[(i + k + 4) * p_stride + j];
uint8x8_t val = vqmovun_s16(result_final);
if (p_width == 4) {
vst1_lane_u32((uint32_t *)p, vreinterpret_u32_u8(val), 0);
} else {
vst1_u8(p, val);
}
}
}
}
}
}

View file

@ -26,7 +26,6 @@
Apply horizontal filter and store in a temporary buffer. When applying
vertical filter, overwrite the original pixel values.
*/
void av1_wiener_convolve_add_src_neon(const uint8_t *src, ptrdiff_t src_stride,
uint8_t *dst, ptrdiff_t dst_stride,
const int16_t *filter_x, int x_step_q4,
@ -78,8 +77,10 @@ void av1_wiener_convolve_add_src_neon(const uint8_t *src, ptrdiff_t src_stride,
/* if height is a multiple of 8 */
if (!(h & 7)) {
int16x8_t res0, res1, res2, res3;
uint16x8_t res4, res5, res6, res7, res8, res9, res10, res11;
uint16x8_t res4;
uint8x8_t t0, t1, t2, t3, t4, t5, t6, t7;
#if defined(__aarch64__)
uint16x8_t res5, res6, res7, res8, res9, res10, res11;
uint8x8_t t8, t9, t10, t11, t12, t13, t14;
do {
@ -190,16 +191,64 @@ void av1_wiener_convolve_add_src_neon(const uint8_t *src, ptrdiff_t src_stride,
dst_ptr += 8 * MAX_SB_SIZE;
height -= 8;
} while (height > 0);
#else
uint8x8_t temp_0;
do {
const uint8_t *s;
__builtin_prefetch(src_ptr);
t0 = vld1_u8(src_ptr); // a0 a1 a2 a3 a4 a5 a6 a7
s = src_ptr + 8;
d_tmp = dst_ptr;
width = w;
__builtin_prefetch(dst_ptr);
do {
t7 = vld1_u8(s); // a8 a9 a10 a11 a12 a13 a14 a15
temp_0 = t0;
t0 = t7;
t1 = vext_u8(temp_0, t7, 1); // a1 a2 a3 a4 a5 a6 a7 a8
t2 = vext_u8(temp_0, t7, 2); // a2 a3 a4 a5 a6 a7 a8 a9
t3 = vext_u8(temp_0, t7, 3); // a3 a4 a5 a6 a7 a8 a9 a10
t4 = vext_u8(temp_0, t7, 4); // a4 a5 a6 a7 a8 a9 a10 a11
t5 = vext_u8(temp_0, t7, 5); // a5 a6 a7 a8 a9 a10 a11 a12
t6 = vext_u8(temp_0, t7, 6); // a6 a7 a8 a9 a10 a11 a12 a13
t7 = vext_u8(temp_0, t7, 7); // a7 a8 a9 a10 a11 a12 a13 a14
res0 = vreinterpretq_s16_u16(vaddl_u8(temp_0, t6));
res1 = vreinterpretq_s16_u16(vaddl_u8(t1, t5));
res2 = vreinterpretq_s16_u16(vaddl_u8(t2, t4));
res3 = vreinterpretq_s16_u16(vmovl_u8(t3));
res4 = wiener_convolve8_horiz_8x8(res0, res1, res2, res3, filter_x_tmp,
bd, conv_params->round_0);
vst1q_u16(d_tmp, res4);
s += 8;
d_tmp += 8;
width -= 8;
} while (width > 0);
src_ptr += src_stride;
dst_ptr += MAX_SB_SIZE;
height--;
} while (height > 0);
#endif
} else {
/*if height is a multiple of 4*/
int16x8_t tt0, tt1, tt2, tt3;
const uint8_t *s;
uint16x4_t res0, res1, res2, res3, res4, res5, res6, res7;
uint16x8_t d0, d1, d2, d3;
int16x4_t s0, s1, s2, s3, s4, s5, s6, s7, s8, s9, s10;
int16x4_t s11, s12, s13, s14;
int16x8_t tt0, tt1, tt2, tt3;
uint16x8_t d0;
uint8x8_t t0, t1, t2, t3;
#if defined(__aarch64__)
uint16x4_t res0, res1, res2, res3, res4, res5, res6, res7;
uint16x8_t d1, d2, d3;
int16x4_t s0, s1, s2, s3, s4, s5, s6, s7, s8, s9, s10;
int16x4_t s11, s12, s13, s14;
do {
__builtin_prefetch(src_ptr + 0 * src_stride);
__builtin_prefetch(src_ptr + 1 * src_stride);
@ -292,11 +341,61 @@ void av1_wiener_convolve_add_src_neon(const uint8_t *src, ptrdiff_t src_stride,
dst_ptr += 4 * MAX_SB_SIZE;
height -= 4;
} while (height > 0);
#else
uint8x8_t temp_0, t4, t5, t6, t7;
do {
__builtin_prefetch(src_ptr);
t0 = vld1_u8(src_ptr); // a0 a1 a2 a3 a4 a5 a6 a7
__builtin_prefetch(dst_ptr);
s = src_ptr + 8;
d_tmp = dst_ptr;
width = w;
do {
t7 = vld1_u8(s); // a8 a9 a10 a11 a12 a13 a14 a15
temp_0 = t0;
t0 = t7;
t1 = vext_u8(temp_0, t7, 1); // a1 a2 a3 a4 a5 a6 a7 a8
t2 = vext_u8(temp_0, t7, 2); // a2 a3 a4 a5 a6 a7 a8 a9
t3 = vext_u8(temp_0, t7, 3); // a3 a4 a5 a6 a7 a8 a9 a10
t4 = vext_u8(temp_0, t7, 4); // a4 a5 a6 a7 a8 a9 a10 a11
t5 = vext_u8(temp_0, t7, 5); // a5 a6 a7 a8 a9 a10 a11 a12
t6 = vext_u8(temp_0, t7, 6); // a6 a7 a8 a9 a10 a11 a12 a13
t7 = vext_u8(temp_0, t7, 7); // a7 a8 a9 a10 a11 a12 a13 a14
tt0 = vreinterpretq_s16_u16(vaddl_u8(temp_0, t6));
tt1 = vreinterpretq_s16_u16(vaddl_u8(t1, t5));
tt2 = vreinterpretq_s16_u16(vaddl_u8(t2, t4));
tt3 = vreinterpretq_s16_u16(vmovl_u8(t3));
d0 = wiener_convolve8_horiz_8x8(tt0, tt1, tt2, tt3, filter_x_tmp, bd,
conv_params->round_0);
vst1q_u16(d_tmp, d0);
s += 8;
d_tmp += 8;
width -= 8;
} while (width > 0);
src_ptr += src_stride;
dst_ptr += MAX_SB_SIZE;
height -= 1;
} while (height > 0);
#endif
}
{
int16x8_t s0, s1, s2, s3, s4, s5, s6, s7, s8, s9, s10;
uint8x8_t t0, t1, t2, t3;
int16x8_t s0, s1, s2, s3, s4, s5, s6, s7;
uint8x8_t t0;
#if defined(__aarch64__)
int16x8_t s8, s9, s10;
uint8x8_t t1, t2, t3;
#endif
int16_t *src_tmp_ptr, *s;
uint8_t *dst_tmp_ptr;
height = h;
@ -324,6 +423,7 @@ void av1_wiener_convolve_add_src_neon(const uint8_t *src, ptrdiff_t src_stride,
d = dst_tmp_ptr;
height = h;
#if defined(__aarch64__)
do {
__builtin_prefetch(dst_tmp_ptr + 0 * dst_stride);
__builtin_prefetch(dst_tmp_ptr + 1 * dst_stride);
@ -397,5 +497,34 @@ void av1_wiener_convolve_add_src_neon(const uint8_t *src, ptrdiff_t src_stride,
w -= 8;
} while (w > 0);
#else
do {
__builtin_prefetch(dst_tmp_ptr + 0 * dst_stride);
s7 = vld1q_s16(s);
s += src_stride;
t0 = wiener_convolve8_vert_4x8(s0, s1, s2, s3, s4, s5, s6, filter_y_tmp,
bd, conv_params->round_1);
vst1_u8(d, t0);
d += dst_stride;
s0 = s1;
s1 = s2;
s2 = s3;
s3 = s4;
s4 = s5;
s5 = s6;
s6 = s7;
height -= 1;
} while (height > 0);
src_tmp_ptr += 8;
dst_tmp_ptr += 8;
w -= 8;
} while (w > 0);
#endif
}
}

View file

@ -11,56 +11,7 @@
#include <stdlib.h>
#include "av1/common/av1_inv_txfm1d.h"
static void range_check_buf(int32_t stage, const int32_t *input,
const int32_t *buf, int32_t size, int8_t bit) {
#if CONFIG_COEFFICIENT_RANGE_CHECKING
const int64_t max_value = (1LL << (bit - 1)) - 1;
const int64_t min_value = -(1LL << (bit - 1));
int in_range = 1;
for (int i = 0; i < size; ++i) {
if (buf[i] < min_value || buf[i] > max_value) {
in_range = 0;
}
}
if (!in_range) {
fprintf(stderr, "Error: coeffs contain out-of-range values\n");
fprintf(stderr, "size: %d\n", size);
fprintf(stderr, "stage: %d\n", stage);
fprintf(stderr, "allowed range: [%" PRId64 ";%" PRId64 "]\n", min_value,
max_value);
fprintf(stderr, "coeffs: ");
fprintf(stderr, "[");
for (int j = 0; j < size; j++) {
if (j > 0) fprintf(stderr, ", ");
fprintf(stderr, "%d", input[j]);
}
fprintf(stderr, "]\n");
fprintf(stderr, " buf: ");
fprintf(stderr, "[");
for (int j = 0; j < size; j++) {
if (j > 0) fprintf(stderr, ", ");
fprintf(stderr, "%d", buf[j]);
}
fprintf(stderr, "]\n\n");
}
assert(in_range);
#else
(void)stage;
(void)input;
(void)buf;
(void)size;
(void)bit;
#endif
}
#include "av1/common/av1_txfm.h"
// TODO(angiebird): Make 1-d txfm functions static
//
@ -84,7 +35,7 @@ void av1_idct4_new(const int32_t *input, int32_t *output, int8_t cos_bit,
bf1[1] = input[2];
bf1[2] = input[1];
bf1[3] = input[3];
range_check_buf(stage, input, bf1, size, stage_range[stage]);
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
// stage 2
stage++;
@ -94,7 +45,7 @@ void av1_idct4_new(const int32_t *input, int32_t *output, int8_t cos_bit,
bf1[1] = half_btf(cospi[32], bf0[0], -cospi[32], bf0[1], cos_bit);
bf1[2] = half_btf(cospi[48], bf0[2], -cospi[16], bf0[3], cos_bit);
bf1[3] = half_btf(cospi[16], bf0[2], cospi[48], bf0[3], cos_bit);
range_check_buf(stage, input, bf1, size, stage_range[stage]);
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
// stage 3
stage++;
@ -129,7 +80,7 @@ void av1_idct8_new(const int32_t *input, int32_t *output, int8_t cos_bit,
bf1[5] = input[5];
bf1[6] = input[3];
bf1[7] = input[7];
range_check_buf(stage, input, bf1, size, stage_range[stage]);
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
// stage 2
stage++;
@ -143,7 +94,7 @@ void av1_idct8_new(const int32_t *input, int32_t *output, int8_t cos_bit,
bf1[5] = half_btf(cospi[24], bf0[5], -cospi[40], bf0[6], cos_bit);
bf1[6] = half_btf(cospi[40], bf0[5], cospi[24], bf0[6], cos_bit);
bf1[7] = half_btf(cospi[8], bf0[4], cospi[56], bf0[7], cos_bit);
range_check_buf(stage, input, bf1, size, stage_range[stage]);
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
// stage 3
stage++;
@ -157,7 +108,7 @@ void av1_idct8_new(const int32_t *input, int32_t *output, int8_t cos_bit,
bf1[5] = clamp_value(bf0[4] - bf0[5], stage_range[stage]);
bf1[6] = clamp_value(-bf0[6] + bf0[7], stage_range[stage]);
bf1[7] = clamp_value(bf0[6] + bf0[7], stage_range[stage]);
range_check_buf(stage, input, bf1, size, stage_range[stage]);
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
// stage 4
stage++;
@ -171,7 +122,7 @@ void av1_idct8_new(const int32_t *input, int32_t *output, int8_t cos_bit,
bf1[5] = half_btf(-cospi[32], bf0[5], cospi[32], bf0[6], cos_bit);
bf1[6] = half_btf(cospi[32], bf0[5], cospi[32], bf0[6], cos_bit);
bf1[7] = bf0[7];
range_check_buf(stage, input, bf1, size, stage_range[stage]);
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
// stage 5
stage++;
@ -218,7 +169,7 @@ void av1_idct16_new(const int32_t *input, int32_t *output, int8_t cos_bit,
bf1[13] = input[11];
bf1[14] = input[7];
bf1[15] = input[15];
range_check_buf(stage, input, bf1, size, stage_range[stage]);
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
// stage 2
stage++;
@ -240,7 +191,7 @@ void av1_idct16_new(const int32_t *input, int32_t *output, int8_t cos_bit,
bf1[13] = half_btf(cospi[20], bf0[10], cospi[44], bf0[13], cos_bit);
bf1[14] = half_btf(cospi[36], bf0[9], cospi[28], bf0[14], cos_bit);
bf1[15] = half_btf(cospi[4], bf0[8], cospi[60], bf0[15], cos_bit);
range_check_buf(stage, input, bf1, size, stage_range[stage]);
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
// stage 3
stage++;
@ -262,7 +213,7 @@ void av1_idct16_new(const int32_t *input, int32_t *output, int8_t cos_bit,
bf1[13] = clamp_value(bf0[12] - bf0[13], stage_range[stage]);
bf1[14] = clamp_value(-bf0[14] + bf0[15], stage_range[stage]);
bf1[15] = clamp_value(bf0[14] + bf0[15], stage_range[stage]);
range_check_buf(stage, input, bf1, size, stage_range[stage]);
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
// stage 4
stage++;
@ -284,7 +235,7 @@ void av1_idct16_new(const int32_t *input, int32_t *output, int8_t cos_bit,
bf1[13] = half_btf(-cospi[16], bf0[10], cospi[48], bf0[13], cos_bit);
bf1[14] = half_btf(cospi[48], bf0[9], cospi[16], bf0[14], cos_bit);
bf1[15] = bf0[15];
range_check_buf(stage, input, bf1, size, stage_range[stage]);
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
// stage 5
stage++;
@ -306,7 +257,7 @@ void av1_idct16_new(const int32_t *input, int32_t *output, int8_t cos_bit,
bf1[13] = clamp_value(-bf0[13] + bf0[14], stage_range[stage]);
bf1[14] = clamp_value(bf0[13] + bf0[14], stage_range[stage]);
bf1[15] = clamp_value(bf0[12] + bf0[15], stage_range[stage]);
range_check_buf(stage, input, bf1, size, stage_range[stage]);
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
// stage 6
stage++;
@ -328,7 +279,7 @@ void av1_idct16_new(const int32_t *input, int32_t *output, int8_t cos_bit,
bf1[13] = half_btf(cospi[32], bf0[10], cospi[32], bf0[13], cos_bit);
bf1[14] = bf0[14];
bf1[15] = bf0[15];
range_check_buf(stage, input, bf1, size, stage_range[stage]);
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
// stage 7
stage++;
@ -399,7 +350,7 @@ void av1_idct32_new(const int32_t *input, int32_t *output, int8_t cos_bit,
bf1[29] = input[23];
bf1[30] = input[15];
bf1[31] = input[31];
range_check_buf(stage, input, bf1, size, stage_range[stage]);
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
// stage 2
stage++;
@ -437,7 +388,7 @@ void av1_idct32_new(const int32_t *input, int32_t *output, int8_t cos_bit,
bf1[29] = half_btf(cospi[18], bf0[18], cospi[46], bf0[29], cos_bit);
bf1[30] = half_btf(cospi[34], bf0[17], cospi[30], bf0[30], cos_bit);
bf1[31] = half_btf(cospi[2], bf0[16], cospi[62], bf0[31], cos_bit);
range_check_buf(stage, input, bf1, size, stage_range[stage]);
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
// stage 3
stage++;
@ -475,7 +426,7 @@ void av1_idct32_new(const int32_t *input, int32_t *output, int8_t cos_bit,
bf1[29] = clamp_value(bf0[28] - bf0[29], stage_range[stage]);
bf1[30] = clamp_value(-bf0[30] + bf0[31], stage_range[stage]);
bf1[31] = clamp_value(bf0[30] + bf0[31], stage_range[stage]);
range_check_buf(stage, input, bf1, size, stage_range[stage]);
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
// stage 4
stage++;
@ -513,7 +464,7 @@ void av1_idct32_new(const int32_t *input, int32_t *output, int8_t cos_bit,
bf1[29] = half_btf(-cospi[8], bf0[18], cospi[56], bf0[29], cos_bit);
bf1[30] = half_btf(cospi[56], bf0[17], cospi[8], bf0[30], cos_bit);
bf1[31] = bf0[31];
range_check_buf(stage, input, bf1, size, stage_range[stage]);
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
// stage 5
stage++;
@ -551,7 +502,7 @@ void av1_idct32_new(const int32_t *input, int32_t *output, int8_t cos_bit,
bf1[29] = clamp_value(-bf0[29] + bf0[30], stage_range[stage]);
bf1[30] = clamp_value(bf0[29] + bf0[30], stage_range[stage]);
bf1[31] = clamp_value(bf0[28] + bf0[31], stage_range[stage]);
range_check_buf(stage, input, bf1, size, stage_range[stage]);
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
// stage 6
stage++;
@ -589,7 +540,7 @@ void av1_idct32_new(const int32_t *input, int32_t *output, int8_t cos_bit,
bf1[29] = half_btf(cospi[48], bf0[18], cospi[16], bf0[29], cos_bit);
bf1[30] = bf0[30];
bf1[31] = bf0[31];
range_check_buf(stage, input, bf1, size, stage_range[stage]);
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
// stage 7
stage++;
@ -627,7 +578,7 @@ void av1_idct32_new(const int32_t *input, int32_t *output, int8_t cos_bit,
bf1[29] = clamp_value(bf0[26] + bf0[29], stage_range[stage]);
bf1[30] = clamp_value(bf0[25] + bf0[30], stage_range[stage]);
bf1[31] = clamp_value(bf0[24] + bf0[31], stage_range[stage]);
range_check_buf(stage, input, bf1, size, stage_range[stage]);
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
// stage 8
stage++;
@ -665,7 +616,7 @@ void av1_idct32_new(const int32_t *input, int32_t *output, int8_t cos_bit,
bf1[29] = bf0[29];
bf1[30] = bf0[30];
bf1[31] = bf0[31];
range_check_buf(stage, input, bf1, size, stage_range[stage]);
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
// stage 9
stage++;
@ -760,7 +711,6 @@ void av1_iadst4_new(const int32_t *input, int32_t *output, int8_t cos_bit,
output[1] = round_shift(x1, bit);
output[2] = round_shift(x2, bit);
output[3] = round_shift(x3, bit);
range_check_buf(6, input, output, 4, stage_range[6]);
}
void av1_iadst8_new(const int32_t *input, int32_t *output, int8_t cos_bit,
@ -786,7 +736,7 @@ void av1_iadst8_new(const int32_t *input, int32_t *output, int8_t cos_bit,
bf1[5] = input[4];
bf1[6] = input[1];
bf1[7] = input[6];
range_check_buf(stage, input, bf1, size, stage_range[stage]);
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
// stage 2
stage++;
@ -800,7 +750,7 @@ void av1_iadst8_new(const int32_t *input, int32_t *output, int8_t cos_bit,
bf1[5] = half_btf(cospi[28], bf0[4], -cospi[36], bf0[5], cos_bit);
bf1[6] = half_btf(cospi[52], bf0[6], cospi[12], bf0[7], cos_bit);
bf1[7] = half_btf(cospi[12], bf0[6], -cospi[52], bf0[7], cos_bit);
range_check_buf(stage, input, bf1, size, stage_range[stage]);
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
// stage 3
stage++;
@ -814,7 +764,7 @@ void av1_iadst8_new(const int32_t *input, int32_t *output, int8_t cos_bit,
bf1[5] = clamp_value(bf0[1] - bf0[5], stage_range[stage]);
bf1[6] = clamp_value(bf0[2] - bf0[6], stage_range[stage]);
bf1[7] = clamp_value(bf0[3] - bf0[7], stage_range[stage]);
range_check_buf(stage, input, bf1, size, stage_range[stage]);
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
// stage 4
stage++;
@ -828,7 +778,7 @@ void av1_iadst8_new(const int32_t *input, int32_t *output, int8_t cos_bit,
bf1[5] = half_btf(cospi[48], bf0[4], -cospi[16], bf0[5], cos_bit);
bf1[6] = half_btf(-cospi[48], bf0[6], cospi[16], bf0[7], cos_bit);
bf1[7] = half_btf(cospi[16], bf0[6], cospi[48], bf0[7], cos_bit);
range_check_buf(stage, input, bf1, size, stage_range[stage]);
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
// stage 5
stage++;
@ -842,7 +792,7 @@ void av1_iadst8_new(const int32_t *input, int32_t *output, int8_t cos_bit,
bf1[5] = clamp_value(bf0[5] + bf0[7], stage_range[stage]);
bf1[6] = clamp_value(bf0[4] - bf0[6], stage_range[stage]);
bf1[7] = clamp_value(bf0[5] - bf0[7], stage_range[stage]);
range_check_buf(stage, input, bf1, size, stage_range[stage]);
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
// stage 6
stage++;
@ -856,7 +806,7 @@ void av1_iadst8_new(const int32_t *input, int32_t *output, int8_t cos_bit,
bf1[5] = bf0[5];
bf1[6] = half_btf(cospi[32], bf0[6], cospi[32], bf0[7], cos_bit);
bf1[7] = half_btf(cospi[32], bf0[6], -cospi[32], bf0[7], cos_bit);
range_check_buf(stage, input, bf1, size, stage_range[stage]);
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
// stage 7
stage++;
@ -903,7 +853,7 @@ void av1_iadst16_new(const int32_t *input, int32_t *output, int8_t cos_bit,
bf1[13] = input[12];
bf1[14] = input[1];
bf1[15] = input[14];
range_check_buf(stage, input, bf1, size, stage_range[stage]);
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
// stage 2
stage++;
@ -925,7 +875,7 @@ void av1_iadst16_new(const int32_t *input, int32_t *output, int8_t cos_bit,
bf1[13] = half_btf(cospi[14], bf0[12], -cospi[50], bf0[13], cos_bit);
bf1[14] = half_btf(cospi[58], bf0[14], cospi[6], bf0[15], cos_bit);
bf1[15] = half_btf(cospi[6], bf0[14], -cospi[58], bf0[15], cos_bit);
range_check_buf(stage, input, bf1, size, stage_range[stage]);
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
// stage 3
stage++;
@ -947,7 +897,7 @@ void av1_iadst16_new(const int32_t *input, int32_t *output, int8_t cos_bit,
bf1[13] = clamp_value(bf0[5] - bf0[13], stage_range[stage]);
bf1[14] = clamp_value(bf0[6] - bf0[14], stage_range[stage]);
bf1[15] = clamp_value(bf0[7] - bf0[15], stage_range[stage]);
range_check_buf(stage, input, bf1, size, stage_range[stage]);
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
// stage 4
stage++;
@ -969,7 +919,7 @@ void av1_iadst16_new(const int32_t *input, int32_t *output, int8_t cos_bit,
bf1[13] = half_btf(cospi[8], bf0[12], cospi[56], bf0[13], cos_bit);
bf1[14] = half_btf(-cospi[24], bf0[14], cospi[40], bf0[15], cos_bit);
bf1[15] = half_btf(cospi[40], bf0[14], cospi[24], bf0[15], cos_bit);
range_check_buf(stage, input, bf1, size, stage_range[stage]);
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
// stage 5
stage++;
@ -991,7 +941,7 @@ void av1_iadst16_new(const int32_t *input, int32_t *output, int8_t cos_bit,
bf1[13] = clamp_value(bf0[9] - bf0[13], stage_range[stage]);
bf1[14] = clamp_value(bf0[10] - bf0[14], stage_range[stage]);
bf1[15] = clamp_value(bf0[11] - bf0[15], stage_range[stage]);
range_check_buf(stage, input, bf1, size, stage_range[stage]);
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
// stage 6
stage++;
@ -1013,7 +963,7 @@ void av1_iadst16_new(const int32_t *input, int32_t *output, int8_t cos_bit,
bf1[13] = half_btf(cospi[48], bf0[12], -cospi[16], bf0[13], cos_bit);
bf1[14] = half_btf(-cospi[48], bf0[14], cospi[16], bf0[15], cos_bit);
bf1[15] = half_btf(cospi[16], bf0[14], cospi[48], bf0[15], cos_bit);
range_check_buf(stage, input, bf1, size, stage_range[stage]);
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
// stage 7
stage++;
@ -1035,7 +985,7 @@ void av1_iadst16_new(const int32_t *input, int32_t *output, int8_t cos_bit,
bf1[13] = clamp_value(bf0[13] + bf0[15], stage_range[stage]);
bf1[14] = clamp_value(bf0[12] - bf0[14], stage_range[stage]);
bf1[15] = clamp_value(bf0[13] - bf0[15], stage_range[stage]);
range_check_buf(stage, input, bf1, size, stage_range[stage]);
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
// stage 8
stage++;
@ -1057,7 +1007,7 @@ void av1_iadst16_new(const int32_t *input, int32_t *output, int8_t cos_bit,
bf1[13] = bf0[13];
bf1[14] = half_btf(cospi[32], bf0[14], cospi[32], bf0[15], cos_bit);
bf1[15] = half_btf(cospi[32], bf0[14], -cospi[32], bf0[15], cos_bit);
range_check_buf(stage, input, bf1, size, stage_range[stage]);
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
// stage 9
stage++;
@ -1193,7 +1143,7 @@ void av1_idct64_new(const int32_t *input, int32_t *output, int8_t cos_bit,
bf1[61] = input[47];
bf1[62] = input[31];
bf1[63] = input[63];
range_check_buf(stage, input, bf1, size, stage_range[stage]);
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
// stage 2
stage++;
@ -1263,7 +1213,7 @@ void av1_idct64_new(const int32_t *input, int32_t *output, int8_t cos_bit,
bf1[61] = half_btf(cospi[17], bf0[34], cospi[47], bf0[61], cos_bit);
bf1[62] = half_btf(cospi[33], bf0[33], cospi[31], bf0[62], cos_bit);
bf1[63] = half_btf(cospi[1], bf0[32], cospi[63], bf0[63], cos_bit);
range_check_buf(stage, input, bf1, size, stage_range[stage]);
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
// stage 3
stage++;
@ -1333,7 +1283,7 @@ void av1_idct64_new(const int32_t *input, int32_t *output, int8_t cos_bit,
bf1[61] = clamp_value(bf0[60] - bf0[61], stage_range[stage]);
bf1[62] = clamp_value(-bf0[62] + bf0[63], stage_range[stage]);
bf1[63] = clamp_value(bf0[62] + bf0[63], stage_range[stage]);
range_check_buf(stage, input, bf1, size, stage_range[stage]);
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
// stage 4
stage++;
@ -1403,7 +1353,7 @@ void av1_idct64_new(const int32_t *input, int32_t *output, int8_t cos_bit,
bf1[61] = half_btf(-cospi[4], bf0[34], cospi[60], bf0[61], cos_bit);
bf1[62] = half_btf(cospi[60], bf0[33], cospi[4], bf0[62], cos_bit);
bf1[63] = bf0[63];
range_check_buf(stage, input, bf1, size, stage_range[stage]);
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
// stage 5
stage++;
@ -1473,7 +1423,7 @@ void av1_idct64_new(const int32_t *input, int32_t *output, int8_t cos_bit,
bf1[61] = clamp_value(-bf0[61] + bf0[62], stage_range[stage]);
bf1[62] = clamp_value(bf0[61] + bf0[62], stage_range[stage]);
bf1[63] = clamp_value(bf0[60] + bf0[63], stage_range[stage]);
range_check_buf(stage, input, bf1, size, stage_range[stage]);
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
// stage 6
stage++;
@ -1543,7 +1493,7 @@ void av1_idct64_new(const int32_t *input, int32_t *output, int8_t cos_bit,
bf1[61] = half_btf(cospi[56], bf0[34], cospi[8], bf0[61], cos_bit);
bf1[62] = bf0[62];
bf1[63] = bf0[63];
range_check_buf(stage, input, bf1, size, stage_range[stage]);
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
// stage 7
stage++;
@ -1613,7 +1563,7 @@ void av1_idct64_new(const int32_t *input, int32_t *output, int8_t cos_bit,
bf1[61] = clamp_value(bf0[58] + bf0[61], stage_range[stage]);
bf1[62] = clamp_value(bf0[57] + bf0[62], stage_range[stage]);
bf1[63] = clamp_value(bf0[56] + bf0[63], stage_range[stage]);
range_check_buf(stage, input, bf1, size, stage_range[stage]);
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
// stage 8
stage++;
@ -1683,7 +1633,7 @@ void av1_idct64_new(const int32_t *input, int32_t *output, int8_t cos_bit,
bf1[61] = bf0[61];
bf1[62] = bf0[62];
bf1[63] = bf0[63];
range_check_buf(stage, input, bf1, size, stage_range[stage]);
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
// stage 9
stage++;
@ -1753,7 +1703,7 @@ void av1_idct64_new(const int32_t *input, int32_t *output, int8_t cos_bit,
bf1[61] = clamp_value(bf0[50] + bf0[61], stage_range[stage]);
bf1[62] = clamp_value(bf0[49] + bf0[62], stage_range[stage]);
bf1[63] = clamp_value(bf0[48] + bf0[63], stage_range[stage]);
range_check_buf(stage, input, bf1, size, stage_range[stage]);
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
// stage 10
stage++;
@ -1823,7 +1773,7 @@ void av1_idct64_new(const int32_t *input, int32_t *output, int8_t cos_bit,
bf1[61] = bf0[61];
bf1[62] = bf0[62];
bf1[63] = bf0[63];
range_check_buf(stage, input, bf1, size, stage_range[stage]);
av1_range_check_buf(stage, input, bf1, size, stage_range[stage]);
// stage 11
stage++;

View file

@ -9,8 +9,8 @@
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AV1_INV_TXFM1D_H_
#define AV1_INV_TXFM1D_H_
#ifndef AOM_AV1_COMMON_AV1_INV_TXFM1D_H_
#define AOM_AV1_COMMON_AV1_INV_TXFM1D_H_
#include "av1/common/av1_txfm.h"
@ -58,4 +58,4 @@ void av1_iidentity32_c(const int32_t *input, int32_t *output, int8_t cos_bit,
}
#endif
#endif // AV1_INV_TXFM1D_H_
#endif // AOM_AV1_COMMON_AV1_INV_TXFM1D_H_

View file

@ -9,8 +9,8 @@
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AV1_INV_TXFM2D_CFG_H_
#define AV1_INV_TXFM2D_CFG_H_
#ifndef AOM_AV1_COMMON_AV1_INV_TXFM1D_CFG_H_
#define AOM_AV1_COMMON_AV1_INV_TXFM1D_CFG_H_
#include "av1/common/av1_inv_txfm1d.h"
// sum of fwd_shift_##
@ -44,4 +44,4 @@ extern const int8_t *inv_txfm_shift_ls[TX_SIZES_ALL];
extern const int8_t inv_cos_bit_col[5 /*row*/][5 /*col*/];
extern const int8_t inv_cos_bit_row[5 /*row*/][5 /*col*/];
#endif // AV1_INV_TXFM2D_CFG_H_
#endif // AOM_AV1_COMMON_AV1_INV_TXFM1D_CFG_H_

File diff suppressed because it is too large Load diff

View file

@ -9,8 +9,8 @@
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AV1_COMMON_LOOPFILTER_H_
#define AV1_COMMON_LOOPFILTER_H_
#ifndef AOM_AV1_COMMON_AV1_LOOPFILTER_H_
#define AOM_AV1_COMMON_AV1_LOOPFILTER_H_
#include "config/aom_config.h"
@ -60,51 +60,20 @@ typedef struct {
uint8_t lfl_y_hor[MI_SIZE_64X64][MI_SIZE_64X64];
uint8_t lfl_y_ver[MI_SIZE_64X64][MI_SIZE_64X64];
// U plane vertical edge and horizontal edge filter level
uint8_t lfl_u_hor[MI_SIZE_64X64][MI_SIZE_64X64];
uint8_t lfl_u_ver[MI_SIZE_64X64][MI_SIZE_64X64];
// U plane filter level
uint8_t lfl_u[MI_SIZE_64X64][MI_SIZE_64X64];
// V plane vertical edge and horizontal edge filter level
uint8_t lfl_v_hor[MI_SIZE_64X64][MI_SIZE_64X64];
uint8_t lfl_v_ver[MI_SIZE_64X64][MI_SIZE_64X64];
// V plane filter level
uint8_t lfl_v[MI_SIZE_64X64][MI_SIZE_64X64];
// other info
FilterMask skip;
FilterMask is_vert_border;
FilterMask is_horz_border;
// Y or UV planes, 5 tx sizes: 4x4, 8x8, 16x16, 32x32, 64x64
FilterMask tx_size_ver[2][5];
FilterMask tx_size_hor[2][5];
} LoopFilterMask;
// To determine whether to apply loop filtering at one transform block edge,
// we need information of the neighboring transform block. Specifically,
// in determining a vertical edge, we need the information of the tx block
// to its left. For a horizontal edge, we need info of the tx block above it.
// Thus, we need to record info of right column and bottom row of tx blocks.
// We record the information of the neighboring superblock, when bitmask
// building for a superblock is finished. And it will be used for next
// superblock bitmask building.
// Information includes:
// ------------------------------------------------------------
// MI_SIZE_64X64
// Y tx_size above |--------------|
// Y tx_size left |--------------|
// UV tx_size above |--------------|
// UV tx_size left |--------------|
// Y level above |--------------|
// Y level left |--------------|
// U level above |--------------|
// U level left |--------------|
// V level above |--------------|
// V level left |--------------|
// skip |--------------|
// ------------------------------------------------------------
typedef struct {
TX_SIZE tx_size_y_above[MI_SIZE_64X64];
TX_SIZE tx_size_y_left[MI_SIZE_64X64];
TX_SIZE tx_size_uv_above[MI_SIZE_64X64];
TX_SIZE tx_size_uv_left[MI_SIZE_64X64];
uint8_t y_level_above[MI_SIZE_64X64];
uint8_t y_level_left[MI_SIZE_64X64];
uint8_t u_level_above[MI_SIZE_64X64];
uint8_t u_level_left[MI_SIZE_64X64];
uint8_t v_level_above[MI_SIZE_64X64];
uint8_t v_level_left[MI_SIZE_64X64];
uint8_t skip[MI_SIZE_64X64];
} LpfSuperblockInfo;
#endif // LOOP_FILTER_BITMASK
struct loopfilter {
@ -130,7 +99,6 @@ struct loopfilter {
LoopFilterMask *lfm;
size_t lfm_num;
int lfm_stride;
LpfSuperblockInfo neighbor_sb_lpf_info;
#endif // LOOP_FILTER_BITMASK
};
@ -157,9 +125,15 @@ void av1_loop_filter_init(struct AV1Common *cm);
void av1_loop_filter_frame_init(struct AV1Common *cm, int plane_start,
int plane_end);
#if LOOP_FILTER_BITMASK
void av1_loop_filter_frame(YV12_BUFFER_CONFIG *frame, struct AV1Common *cm,
struct macroblockd *mbd, int is_decoding,
int plane_start, int plane_end, int partial_frame);
#else
void av1_loop_filter_frame(YV12_BUFFER_CONFIG *frame, struct AV1Common *cm,
struct macroblockd *mbd, int plane_start,
int plane_end, int partial_frame);
#endif
void av1_filter_block_plane_vert(const struct AV1Common *const cm,
const MACROBLOCKD *const xd, const int plane,
@ -180,6 +154,9 @@ typedef struct LoopFilterWorkerData {
MACROBLOCKD *xd;
} LFWorkerData;
uint8_t get_filter_level(const struct AV1Common *cm,
const loop_filter_info_n *lfi_n, const int dir_idx,
int plane, const MB_MODE_INFO *mbmi);
#if LOOP_FILTER_BITMASK
void av1_setup_bitmask(struct AV1Common *const cm, int mi_row, int mi_col,
int plane, int subsampling_x, int subsampling_y,
@ -192,10 +169,59 @@ void av1_filter_block_plane_ver(struct AV1Common *const cm,
void av1_filter_block_plane_hor(struct AV1Common *const cm,
struct macroblockd_plane *const plane, int pl,
int mi_row, int mi_col);
LoopFilterMask *get_loop_filter_mask(const struct AV1Common *const cm,
int mi_row, int mi_col);
int get_index_shift(int mi_col, int mi_row, int *index);
static const FilterMask left_txform_mask[TX_SIZES] = {
{ { 0x0000000000000001ULL, // TX_4X4,
0x0000000000000000ULL, 0x0000000000000000ULL, 0x0000000000000000ULL } },
{ { 0x0000000000010001ULL, // TX_8X8,
0x0000000000000000ULL, 0x0000000000000000ULL, 0x0000000000000000ULL } },
{ { 0x0001000100010001ULL, // TX_16X16,
0x0000000000000000ULL, 0x0000000000000000ULL, 0x0000000000000000ULL } },
{ { 0x0001000100010001ULL, // TX_32X32,
0x0001000100010001ULL, 0x0000000000000000ULL, 0x0000000000000000ULL } },
{ { 0x0001000100010001ULL, // TX_64X64,
0x0001000100010001ULL, 0x0001000100010001ULL, 0x0001000100010001ULL } },
};
static const uint64_t above_txform_mask[2][TX_SIZES] = {
{
0x0000000000000001ULL, // TX_4X4
0x0000000000000003ULL, // TX_8X8
0x000000000000000fULL, // TX_16X16
0x00000000000000ffULL, // TX_32X32
0x000000000000ffffULL, // TX_64X64
},
{
0x0000000000000001ULL, // TX_4X4
0x0000000000000005ULL, // TX_8X8
0x0000000000000055ULL, // TX_16X16
0x0000000000005555ULL, // TX_32X32
0x0000000055555555ULL, // TX_64X64
},
};
extern const int mask_id_table_tx_4x4[BLOCK_SIZES_ALL];
extern const int mask_id_table_tx_8x8[BLOCK_SIZES_ALL];
extern const int mask_id_table_tx_16x16[BLOCK_SIZES_ALL];
extern const int mask_id_table_tx_32x32[BLOCK_SIZES_ALL];
extern const FilterMask left_mask_univariant_reordered[67];
extern const FilterMask above_mask_univariant_reordered[67];
#endif
#ifdef __cplusplus
} // extern "C"
#endif
#endif // AV1_COMMON_LOOPFILTER_H_
#endif // AOM_AV1_COMMON_AV1_LOOPFILTER_H_

View file

@ -76,12 +76,12 @@ specialize qw/av1_wiener_convolve_add_src sse2 avx2 neon/;
specialize qw/av1_highbd_wiener_convolve_add_src ssse3/;
specialize qw/av1_highbd_wiener_convolve_add_src avx2/;
# directional intra predictor functions
add_proto qw/void av1_dr_prediction_z1/, "uint8_t *dst, ptrdiff_t stride, int bw, int bh, const uint8_t *above, const uint8_t *left, int upsample_above, int dx, int dy";
add_proto qw/void av1_dr_prediction_z2/, "uint8_t *dst, ptrdiff_t stride, int bw, int bh, const uint8_t *above, const uint8_t *left, int upsample_above, int upsample_left, int dx, int dy";
add_proto qw/void av1_dr_prediction_z3/, "uint8_t *dst, ptrdiff_t stride, int bw, int bh, const uint8_t *above, const uint8_t *left, int upsample_left, int dx, int dy";
# FILTER_INTRA predictor functions
add_proto qw/void av1_filter_intra_predictor/, "uint8_t *dst, ptrdiff_t stride, TX_SIZE tx_size, const uint8_t *above, const uint8_t *left, int mode";
specialize qw/av1_filter_intra_predictor sse4_1/;
@ -108,6 +108,22 @@ specialize qw/av1_highbd_convolve8_vert/, "$sse2_x86_64";
add_proto qw/void av1_inv_txfm_add/, "const tran_low_t *dqcoeff, uint8_t *dst, int stride, const TxfmParam *txfm_param";
specialize qw/av1_inv_txfm_add ssse3 avx2 neon/;
add_proto qw/void av1_highbd_inv_txfm_add/, "const tran_low_t *dqcoeff, uint8_t *dst, int stride, const TxfmParam *txfm_param";
specialize qw/av1_highbd_inv_txfm_add sse4_1 avx2/;
add_proto qw/void av1_highbd_inv_txfm_add_4x4/, "const tran_low_t *dqcoeff, uint8_t *dst, int stride, const TxfmParam *txfm_param";
specialize qw/av1_highbd_inv_txfm_add_4x4 sse4_1/;
add_proto qw/void av1_highbd_inv_txfm_add_8x8/, "const tran_low_t *dqcoeff, uint8_t *dst, int stride, const TxfmParam *txfm_param";
specialize qw/av1_highbd_inv_txfm_add_8x8 sse4_1/;
add_proto qw/void av1_highbd_inv_txfm_add_16x8/, "const tran_low_t *dqcoeff, uint8_t *dst, int stride, const TxfmParam *txfm_param";
specialize qw/av1_highbd_inv_txfm_add_16x8 sse4_1/;
add_proto qw/void av1_highbd_inv_txfm_add_8x16/, "const tran_low_t *dqcoeff, uint8_t *dst, int stride, const TxfmParam *txfm_param";
specialize qw/av1_highbd_inv_txfm_add_8x16 sse4_1/;
add_proto qw/void av1_highbd_inv_txfm_add_16x16/, "const tran_low_t *dqcoeff, uint8_t *dst, int stride, const TxfmParam *txfm_param";
specialize qw/av1_highbd_inv_txfm_add_16x16 sse4_1/;
add_proto qw/void av1_highbd_inv_txfm_add_32x32/, "const tran_low_t *dqcoeff, uint8_t *dst, int stride, const TxfmParam *txfm_param";
specialize qw/av1_highbd_inv_txfm_add_32x32 sse4_1 avx2/;
add_proto qw/void av1_highbd_iwht4x4_1_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, int bd";
add_proto qw/void av1_highbd_iwht4x4_16_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, int bd";
@ -122,9 +138,7 @@ specialize qw/av1_inv_txfm2d_add_4x4 sse4_1/;
add_proto qw/void av1_inv_txfm2d_add_8x8/, "const int32_t *input, uint16_t *output, int stride, TX_TYPE tx_type, int bd";
specialize qw/av1_inv_txfm2d_add_8x8 sse4_1/;
add_proto qw/void av1_inv_txfm2d_add_16x16/, "const int32_t *input, uint16_t *output, int stride, TX_TYPE tx_type, int bd";
specialize qw/av1_inv_txfm2d_add_16x16 sse4_1/;
add_proto qw/void av1_inv_txfm2d_add_32x32/, "const int32_t *input, uint16_t *output, int stride, TX_TYPE tx_type, int bd";
specialize qw/av1_inv_txfm2d_add_32x32 avx2/;
add_proto qw/void av1_inv_txfm2d_add_64x64/, "const int32_t *input, uint16_t *output, int stride, TX_TYPE tx_type, int bd";
add_proto qw/void av1_inv_txfm2d_add_32x64/, "const int32_t *input, uint16_t *output, int stride, TX_TYPE tx_type, int bd";
@ -132,8 +146,6 @@ add_proto qw/void av1_inv_txfm2d_add_64x32/, "const int32_t *input, uint16_t *ou
add_proto qw/void av1_inv_txfm2d_add_16x64/, "const int32_t *input, uint16_t *output, int stride, TX_TYPE tx_type, int bd";
add_proto qw/void av1_inv_txfm2d_add_64x16/, "const int32_t *input, uint16_t *output, int stride, TX_TYPE tx_type, int bd";
specialize qw/av1_inv_txfm2d_add_64x64 sse4_1/;
add_proto qw/void av1_inv_txfm2d_add_4x16/, "const int32_t *input, uint16_t *output, int stride, TX_TYPE tx_type, int bd";
add_proto qw/void av1_inv_txfm2d_add_16x4/, "const int32_t *input, uint16_t *output, int stride, TX_TYPE tx_type, int bd";
add_proto qw/void av1_inv_txfm2d_add_8x32/, "const int32_t *input, uint16_t *output, int stride, TX_TYPE tx_type, int bd";
@ -146,13 +158,13 @@ add_proto qw/void av1_highbd_dr_prediction_z3/, "uint16_t *dst, ptrdiff_t stride
# build compound seg mask functions
add_proto qw/void av1_build_compound_diffwtd_mask/, "uint8_t *mask, DIFFWTD_MASK_TYPE mask_type, const uint8_t *src0, int src0_stride, const uint8_t *src1, int src1_stride, int h, int w";
specialize qw/av1_build_compound_diffwtd_mask sse4_1/;
specialize qw/av1_build_compound_diffwtd_mask sse4_1 avx2/;
add_proto qw/void av1_build_compound_diffwtd_mask_highbd/, "uint8_t *mask, DIFFWTD_MASK_TYPE mask_type, const uint8_t *src0, int src0_stride, const uint8_t *src1, int src1_stride, int h, int w, int bd";
specialize qw/av1_build_compound_diffwtd_mask_highbd ssse3 avx2/;
add_proto qw/void av1_build_compound_diffwtd_mask_d16/, "uint8_t *mask, DIFFWTD_MASK_TYPE mask_type, const CONV_BUF_TYPE *src0, int src0_stride, const CONV_BUF_TYPE *src1, int src1_stride, int h, int w, ConvolveParams *conv_params, int bd";
specialize qw/av1_build_compound_diffwtd_mask_d16 sse4_1 neon/;
specialize qw/av1_build_compound_diffwtd_mask_d16 sse4_1 avx2 neon/;
#
# Encoder functions below this point.
@ -186,7 +198,9 @@ if (aom_config("CONFIG_AV1_ENCODER") eq "yes") {
add_proto qw/void av1_fwd_txfm2d_4x8/, "const int16_t *input, int32_t *output, int stride, TX_TYPE tx_type, int bd";
add_proto qw/void av1_fwd_txfm2d_8x4/, "const int16_t *input, int32_t *output, int stride, TX_TYPE tx_type, int bd";
add_proto qw/void av1_fwd_txfm2d_8x16/, "const int16_t *input, int32_t *output, int stride, TX_TYPE tx_type, int bd";
specialize qw/av1_fwd_txfm2d_8x16 sse4_1/;
add_proto qw/void av1_fwd_txfm2d_16x8/, "const int16_t *input, int32_t *output, int stride, TX_TYPE tx_type, int bd";
specialize qw/av1_fwd_txfm2d_16x8 sse4_1/;
add_proto qw/void av1_fwd_txfm2d_16x32/, "const int16_t *input, int32_t *output, int stride, TX_TYPE tx_type, int bd";
add_proto qw/void av1_fwd_txfm2d_32x16/, "const int16_t *input, int32_t *output, int stride, TX_TYPE tx_type, int bd";
add_proto qw/void av1_fwd_txfm2d_4x16/, "const int16_t *input, int32_t *output, int stride, TX_TYPE tx_type, int bd";
@ -203,6 +217,7 @@ if (aom_config("CONFIG_AV1_ENCODER") eq "yes") {
specialize qw/av1_fwd_txfm2d_32x32 sse4_1/;
add_proto qw/void av1_fwd_txfm2d_64x64/, "const int16_t *input, int32_t *output, int stride, TX_TYPE tx_type, int bd";
specialize qw/av1_fwd_txfm2d_64x64 sse4_1/;
add_proto qw/void av1_fwd_txfm2d_32x64/, "const int16_t *input, int32_t *output, int stride, TX_TYPE tx_type, int bd";
add_proto qw/void av1_fwd_txfm2d_64x32/, "const int16_t *input, int32_t *output, int stride, TX_TYPE tx_type, int bd";
add_proto qw/void av1_fwd_txfm2d_16x64/, "const int16_t *input, int32_t *output, int stride, TX_TYPE tx_type, int bd";
@ -218,7 +233,7 @@ if (aom_config("CONFIG_AV1_ENCODER") eq "yes") {
add_proto qw/void av1_temporal_filter_apply/, "uint8_t *frame1, unsigned int stride, uint8_t *frame2, unsigned int block_width, unsigned int block_height, int strength, int filter_weight, unsigned int *accumulator, uint16_t *count";
specialize qw/av1_temporal_filter_apply sse2 msa/;
add_proto qw/void av1_quantize_b/, "const tran_low_t *coeff_ptr, intptr_t n_coeffs, int skip_block, const int16_t *zbin_ptr, const int16_t *round_ptr, const int16_t *quant_ptr, const int16_t *quant_shift_ptr, tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr, const int16_t *dequant_ptr, uint16_t *eob_ptr, const int16_t *scan, const int16_t *iscan, const qm_val_t * qm_ptr, const qm_val_t * iqm_ptr, int log_scale";
add_proto qw/void av1_quantize_b/, "const tran_low_t *coeff_ptr, intptr_t n_coeffs, const int16_t *zbin_ptr, const int16_t *round_ptr, const int16_t *quant_ptr, const int16_t *quant_shift_ptr, tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr, const int16_t *dequant_ptr, uint16_t *eob_ptr, const int16_t *scan, const int16_t *iscan, const qm_val_t * qm_ptr, const qm_val_t * iqm_ptr, int log_scale";
# ENCODEMB INVOKE
@ -238,7 +253,7 @@ if (aom_config("CONFIG_AV1_ENCODER") eq "yes") {
add_proto qw/void av1_get_nz_map_contexts/, "const uint8_t *const levels, const int16_t *const scan, const uint16_t eob, const TX_SIZE tx_size, const TX_CLASS tx_class, int8_t *const coeff_contexts";
specialize qw/av1_get_nz_map_contexts sse2/;
add_proto qw/void av1_txb_init_levels/, "const tran_low_t *const coeff, const int width, const int height, uint8_t *const levels";
specialize qw/av1_txb_init_levels sse4_1/;
specialize qw/av1_txb_init_levels sse4_1 avx2/;
add_proto qw/uint64_t av1_wedge_sse_from_residuals/, "const int16_t *r1, const int16_t *d, const uint8_t *m, int N";
specialize qw/av1_wedge_sse_from_residuals sse2 avx2/;
@ -251,6 +266,11 @@ if (aom_config("CONFIG_AV1_ENCODER") eq "yes") {
add_proto qw/uint32_t av1_get_crc32c_value/, "void *crc_calculator, uint8_t *p, int length";
specialize qw/av1_get_crc32c_value sse4_2/;
add_proto qw/void av1_compute_stats/, "int wiener_win, const uint8_t *dgd8, const uint8_t *src8, int h_start, int h_end, int v_start, int v_end, int dgd_stride, int src_stride, double *M, double *H";
specialize qw/av1_compute_stats sse4_1 avx2/;
add_proto qw/int64_t av1_lowbd_pixel_proj_error/, " const uint8_t *src8, int width, int height, int src_stride, const uint8_t *dat8, int dat_stride, int32_t *flt0, int flt0_stride, int32_t *flt1, int flt1_stride, int xq[2], const sgr_params_type *params";
specialize qw/av1_lowbd_pixel_proj_error sse4_1 avx2/;
}
# end encoder functions
@ -275,7 +295,7 @@ if ($opts{config} !~ /libs-x86-win32-vs.*/) {
# WARPED_MOTION / GLOBAL_MOTION functions
add_proto qw/void av1_warp_affine/, "const int32_t *mat, const uint8_t *ref, int width, int height, int stride, uint8_t *pred, int p_col, int p_row, int p_width, int p_height, int p_stride, int subsampling_x, int subsampling_y, ConvolveParams *conv_params, int16_t alpha, int16_t beta, int16_t gamma, int16_t delta";
specialize qw/av1_warp_affine sse4_1/;
specialize qw/av1_warp_affine sse4_1 neon/;
add_proto qw/void av1_highbd_warp_affine/, "const int32_t *mat, const uint16_t *ref, int width, int height, int stride, uint16_t *pred, int p_col, int p_row, int p_width, int p_height, int p_stride, int subsampling_x, int subsampling_y, int bd, ConvolveParams *conv_params, int16_t alpha, int16_t beta, int16_t gamma, int16_t delta";
specialize qw/av1_highbd_warp_affine sse4_1/;
@ -290,9 +310,9 @@ if (aom_config("CONFIG_AV1_ENCODER") eq "yes") {
add_proto qw/void apply_selfguided_restoration/, "const uint8_t *dat, int width, int height, int stride, int eps, const int *xqd, uint8_t *dst, int dst_stride, int32_t *tmpbuf, int bit_depth, int highbd";
specialize qw/apply_selfguided_restoration sse4_1 avx2 neon/;
add_proto qw/void av1_selfguided_restoration/, "const uint8_t *dgd8, int width, int height,
int dgd_stride, int32_t *flt0, int32_t *flt1, int flt_stride,
int sgr_params_idx, int bit_depth, int highbd";
add_proto qw/int av1_selfguided_restoration/, "const uint8_t *dgd8, int width, int height,
int dgd_stride, int32_t *flt0, int32_t *flt1, int flt_stride,
int sgr_params_idx, int bit_depth, int highbd";
specialize qw/av1_selfguided_restoration sse4_1 avx2 neon/;
# CONVOLVE_ROUND/COMPOUND_ROUND functions

View file

@ -108,3 +108,53 @@ const int8_t av1_txfm_stage_num_list[TXFM_TYPES] = {
1, // TXFM_TYPE_IDENTITY16
1, // TXFM_TYPE_IDENTITY32
};
void av1_range_check_buf(int32_t stage, const int32_t *input,
const int32_t *buf, int32_t size, int8_t bit) {
#if CONFIG_COEFFICIENT_RANGE_CHECKING
const int64_t max_value = (1LL << (bit - 1)) - 1;
const int64_t min_value = -(1LL << (bit - 1));
int in_range = 1;
for (int i = 0; i < size; ++i) {
if (buf[i] < min_value || buf[i] > max_value) {
in_range = 0;
}
}
if (!in_range) {
fprintf(stderr, "Error: coeffs contain out-of-range values\n");
fprintf(stderr, "size: %d\n", size);
fprintf(stderr, "stage: %d\n", stage);
fprintf(stderr, "allowed range: [%" PRId64 ";%" PRId64 "]\n", min_value,
max_value);
fprintf(stderr, "coeffs: ");
fprintf(stderr, "[");
for (int j = 0; j < size; j++) {
if (j > 0) fprintf(stderr, ", ");
fprintf(stderr, "%d", input[j]);
}
fprintf(stderr, "]\n");
fprintf(stderr, " buf: ");
fprintf(stderr, "[");
for (int j = 0; j < size; j++) {
if (j > 0) fprintf(stderr, ", ");
fprintf(stderr, "%d", buf[j]);
}
fprintf(stderr, "]\n\n");
}
assert(in_range);
#else
(void)stage;
(void)input;
(void)buf;
(void)size;
(void)bit;
#endif
}

View file

@ -9,8 +9,8 @@
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AV1_TXFM_H_
#define AV1_TXFM_H_
#ifndef AOM_AV1_COMMON_AV1_TXFM_H_
#define AOM_AV1_COMMON_AV1_TXFM_H_
#include <assert.h>
#include <math.h>
@ -39,7 +39,7 @@ extern const int32_t av1_sinpi_arr_data[7][5];
static const int cos_bit_min = 10;
static const int cos_bit_max = 16;
static const int NewSqrt2Bits = 12;
#define NewSqrt2Bits ((int32_t)12)
// 2^12 * sqrt(2)
static const int32_t NewSqrt2 = 5793;
// 2^12 / sqrt(2)
@ -64,7 +64,7 @@ static INLINE int32_t range_check_value(int32_t value, int8_t bit) {
#endif // CONFIG_COEFFICIENT_RANGE_CHECKING
#if DO_RANGE_CHECK_CLAMP
bit = AOMMIN(bit, 31);
return clamp(value, (1 << (bit - 1)) - 1, -(1 << (bit - 1)));
return clamp(value, -(1 << (bit - 1)), (1 << (bit - 1)) - 1);
#endif // DO_RANGE_CHECK_CLAMP
(void)bit;
return value;
@ -78,10 +78,25 @@ static INLINE int32_t round_shift(int64_t value, int bit) {
static INLINE int32_t half_btf(int32_t w0, int32_t in0, int32_t w1, int32_t in1,
int bit) {
int64_t result_64 = (int64_t)(w0 * in0) + (int64_t)(w1 * in1);
int64_t intermediate = result_64 + (1LL << (bit - 1));
// NOTE(david.barker): The value 'result_64' may not necessarily fit
// into 32 bits. However, the result of this function is nominally
// ROUND_POWER_OF_TWO_64(result_64, bit)
// and that is required to fit into stage_range[stage] many bits
// (checked by range_check_buf()).
//
// Here we've unpacked that rounding operation, and it can be shown
// that the value of 'intermediate' here *does* fit into 32 bits
// for any conformant bitstream.
// The upshot is that, if you do all this calculation using
// wrapping 32-bit arithmetic instead of (non-wrapping) 64-bit arithmetic,
// then you'll still get the correct result.
// To provide a check on this logic, we assert that 'intermediate'
// would fit into an int32 if range checking is enabled.
#if CONFIG_COEFFICIENT_RANGE_CHECKING
assert(result_64 >= INT32_MIN && result_64 <= INT32_MAX);
assert(intermediate >= INT32_MIN && intermediate <= INT32_MAX);
#endif
return round_shift(result_64, bit);
return (int32_t)(intermediate >> bit);
}
static INLINE uint16_t highbd_clip_pixel_add(uint16_t dest, tran_high_t trans,
@ -206,9 +221,12 @@ static INLINE int get_txw_idx(TX_SIZE tx_size) {
static INLINE int get_txh_idx(TX_SIZE tx_size) {
return tx_size_high_log2[tx_size] - tx_size_high_log2[0];
}
void av1_range_check_buf(int32_t stage, const int32_t *input,
const int32_t *buf, int32_t size, int8_t bit);
#define MAX_TXWH_IDX 5
#ifdef __cplusplus
}
#endif // __cplusplus
#endif // AV1_TXFM_H_
#endif // AOM_AV1_COMMON_AV1_TXFM_H_

View file

@ -28,66 +28,6 @@ PREDICTION_MODE av1_above_block_mode(const MB_MODE_INFO *above_mi) {
return above_mi->mode;
}
void av1_foreach_transformed_block_in_plane(
const MACROBLOCKD *const xd, BLOCK_SIZE bsize, int plane,
foreach_transformed_block_visitor visit, void *arg) {
const struct macroblockd_plane *const pd = &xd->plane[plane];
// block and transform sizes, in number of 4x4 blocks log 2 ("*_b")
// 4x4=0, 8x8=2, 16x16=4, 32x32=6, 64x64=8
// transform size varies per plane, look it up in a common way.
const TX_SIZE tx_size = av1_get_tx_size(plane, xd);
const BLOCK_SIZE plane_bsize =
get_plane_block_size(bsize, pd->subsampling_x, pd->subsampling_y);
const uint8_t txw_unit = tx_size_wide_unit[tx_size];
const uint8_t txh_unit = tx_size_high_unit[tx_size];
const int step = txw_unit * txh_unit;
int i = 0, r, c;
// If mb_to_right_edge is < 0 we are in a situation in which
// the current block size extends into the UMV and we won't
// visit the sub blocks that are wholly within the UMV.
const int max_blocks_wide = max_block_wide(xd, plane_bsize, plane);
const int max_blocks_high = max_block_high(xd, plane_bsize, plane);
int blk_row, blk_col;
const BLOCK_SIZE max_unit_bsize =
get_plane_block_size(BLOCK_64X64, pd->subsampling_x, pd->subsampling_y);
int mu_blocks_wide = block_size_wide[max_unit_bsize] >> tx_size_wide_log2[0];
int mu_blocks_high = block_size_high[max_unit_bsize] >> tx_size_high_log2[0];
mu_blocks_wide = AOMMIN(max_blocks_wide, mu_blocks_wide);
mu_blocks_high = AOMMIN(max_blocks_high, mu_blocks_high);
// Keep track of the row and column of the blocks we use so that we know
// if we are in the unrestricted motion border.
for (r = 0; r < max_blocks_high; r += mu_blocks_high) {
const int unit_height = AOMMIN(mu_blocks_high + r, max_blocks_high);
// Skip visiting the sub blocks that are wholly within the UMV.
for (c = 0; c < max_blocks_wide; c += mu_blocks_wide) {
const int unit_width = AOMMIN(mu_blocks_wide + c, max_blocks_wide);
for (blk_row = r; blk_row < unit_height; blk_row += txh_unit) {
for (blk_col = c; blk_col < unit_width; blk_col += txw_unit) {
visit(plane, i, blk_row, blk_col, plane_bsize, tx_size, arg);
i += step;
}
}
}
}
}
void av1_foreach_transformed_block(const MACROBLOCKD *const xd,
BLOCK_SIZE bsize, int mi_row, int mi_col,
foreach_transformed_block_visitor visit,
void *arg, const int num_planes) {
for (int plane = 0; plane < num_planes; ++plane) {
if (!is_chroma_reference(mi_row, mi_col, bsize,
xd->plane[plane].subsampling_x,
xd->plane[plane].subsampling_y))
continue;
av1_foreach_transformed_block_in_plane(xd, bsize, plane, visit, arg);
}
}
void av1_set_contexts(const MACROBLOCKD *xd, struct macroblockd_plane *pd,
int plane, BLOCK_SIZE plane_bsize, TX_SIZE tx_size,
int has_eob, int aoff, int loff) {
@ -159,6 +99,10 @@ void av1_setup_block_planes(MACROBLOCKD *xd, int ss_x, int ss_y,
xd->plane[i].subsampling_x = i ? ss_x : 0;
xd->plane[i].subsampling_y = i ? ss_y : 0;
}
for (i = num_planes; i < MAX_MB_PLANE; i++) {
xd->plane[i].subsampling_x = 1;
xd->plane[i].subsampling_y = 1;
}
}
const int16_t dr_intra_derivative[90] = {

View file

@ -9,8 +9,8 @@
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AV1_COMMON_BLOCKD_H_
#define AV1_COMMON_BLOCKD_H_
#ifndef AOM_AV1_COMMON_BLOCKD_H_
#define AOM_AV1_COMMON_BLOCKD_H_
#include "config/aom_config.h"
@ -38,13 +38,13 @@ extern "C" {
#define MAX_DIFFWTD_MASK_BITS 1
// DIFFWTD_MASK_TYPES should not surpass 1 << MAX_DIFFWTD_MASK_BITS
typedef enum {
typedef enum ATTRIBUTE_PACKED {
DIFFWTD_38 = 0,
DIFFWTD_38_INV,
DIFFWTD_MASK_TYPES,
} DIFFWTD_MASK_TYPE;
typedef enum {
typedef enum ATTRIBUTE_PACKED {
KEY_FRAME = 0,
INTER_FRAME = 1,
INTRA_ONLY_FRAME = 2, // replaces intra-only
@ -57,7 +57,7 @@ static INLINE int is_comp_ref_allowed(BLOCK_SIZE bsize) {
}
static INLINE int is_inter_mode(PREDICTION_MODE mode) {
return mode >= NEARESTMV && mode <= NEW_NEWMV;
return mode >= INTER_MODE_START && mode < INTER_MODE_END;
}
typedef struct {
@ -66,10 +66,10 @@ typedef struct {
} BUFFER_SET;
static INLINE int is_inter_singleref_mode(PREDICTION_MODE mode) {
return mode >= NEARESTMV && mode <= NEWMV;
return mode >= SINGLE_INTER_MODE_START && mode < SINGLE_INTER_MODE_END;
}
static INLINE int is_inter_compound_mode(PREDICTION_MODE mode) {
return mode >= NEAREST_NEARESTMV && mode <= NEW_NEWMV;
return mode >= COMP_INTER_MODE_START && mode < COMP_INTER_MODE_END;
}
static INLINE PREDICTION_MODE compound_ref0_mode(PREDICTION_MODE mode) {
@ -148,10 +148,6 @@ static INLINE int have_newmv_in_inter_mode(PREDICTION_MODE mode) {
mode == NEW_NEARESTMV || mode == NEAR_NEWMV || mode == NEW_NEARMV);
}
static INLINE int use_masked_motion_search(COMPOUND_TYPE type) {
return (type == COMPOUND_WEDGE);
}
static INLINE int is_masked_compound_type(COMPOUND_TYPE type) {
return (type == COMPOUND_WEDGE || type == COMPOUND_DIFFWTD);
}
@ -267,8 +263,8 @@ typedef struct MB_MODE_INFO {
int mi_row;
int mi_col;
#endif
int num_proj_ref[2];
WarpedMotionParams wm_params[2];
int num_proj_ref;
WarpedMotionParams wm_params;
// Index of the alpha Cb and alpha Cr combination
int cfl_alpha_idx;
@ -376,7 +372,7 @@ static INLINE void mi_to_pixel_loc(int *pixel_c, int *pixel_r, int mi_col,
}
#endif
enum mv_precision { MV_PRECISION_Q3, MV_PRECISION_Q4 };
enum ATTRIBUTE_PACKED mv_precision { MV_PRECISION_Q3, MV_PRECISION_Q4 };
struct buf_2d {
uint8_t *buf;
@ -500,6 +496,8 @@ typedef struct jnt_comp_params {
int bck_offset;
} JNT_COMP_PARAMS;
// Most/all of the pointers are mere pointers to actual arrays are allocated
// elsewhere. This is mostly for coding convenience.
typedef struct macroblockd {
struct macroblockd_plane plane[MAX_MB_PLANE];
@ -544,7 +542,7 @@ typedef struct macroblockd {
SgrprojInfo sgrproj_info[MAX_MB_PLANE];
// block dimension in the unit of mode_info.
uint8_t n8_w, n8_h;
uint8_t n4_w, n4_h;
uint8_t ref_mv_count[MODE_CTX_REF_FRAMES];
CANDIDATE_MV ref_mv_stack[MODE_CTX_REF_FRAMES][MAX_REF_MV_STACK_SIZE];
@ -599,6 +597,9 @@ typedef struct macroblockd {
uint16_t cb_offset[MAX_MB_PLANE];
uint16_t txb_offset[MAX_MB_PLANE];
uint16_t color_index_map_offset[2];
CONV_BUF_TYPE *tmp_conv_dst;
uint8_t *tmp_obmc_bufs[2];
} MACROBLOCKD;
static INLINE int get_bitdepth_data_path_index(const MACROBLOCKD *xd) {
@ -623,6 +624,11 @@ static INLINE int get_sqr_bsize_idx(BLOCK_SIZE bsize) {
}
}
// For a square block size 'bsize', returns the size of the sub-blocks used by
// the given partition type. If the partition produces sub-blocks of different
// sizes, then the function returns the largest sub-block size.
// Implements the Partition_Subsize lookup table in the spec (Section 9.3.
// Conversion tables).
// Note: the input block size should be square.
// Otherwise it's considered invalid.
static INLINE BLOCK_SIZE get_partition_subsize(BLOCK_SIZE bsize,
@ -781,6 +787,8 @@ static INLINE TX_TYPE get_default_tx_type(PLANE_TYPE plane_type,
return intra_mode_to_tx_type(mbmi, plane_type);
}
// Implements the get_plane_residual_size() function in the spec (Section
// 5.11.38. Get plane residual size function).
static INLINE BLOCK_SIZE get_plane_block_size(BLOCK_SIZE bsize,
int subsampling_x,
int subsampling_y) {
@ -952,15 +960,6 @@ typedef void (*foreach_transformed_block_visitor)(int plane, int block,
BLOCK_SIZE plane_bsize,
TX_SIZE tx_size, void *arg);
void av1_foreach_transformed_block_in_plane(
const MACROBLOCKD *const xd, BLOCK_SIZE bsize, int plane,
foreach_transformed_block_visitor visit, void *arg);
void av1_foreach_transformed_block(const MACROBLOCKD *const xd,
BLOCK_SIZE bsize, int mi_row, int mi_col,
foreach_transformed_block_visitor visit,
void *arg, const int num_planes);
void av1_set_contexts(const MACROBLOCKD *xd, struct macroblockd_plane *pd,
int plane, BLOCK_SIZE plane_bsize, TX_SIZE tx_size,
int has_eob, int aoff, int loff);
@ -976,7 +975,7 @@ static INLINE int is_interintra_allowed_bsize(const BLOCK_SIZE bsize) {
}
static INLINE int is_interintra_allowed_mode(const PREDICTION_MODE mode) {
return (mode >= NEARESTMV) && (mode <= NEWMV);
return (mode >= SINGLE_INTER_MODE_START) && (mode < SINGLE_INTER_MODE_END);
}
static INLINE int is_interintra_allowed_ref(const MV_REFERENCE_FRAME rf[2]) {
@ -1045,7 +1044,7 @@ motion_mode_allowed(const WarpedMotionParams *gm_params, const MACROBLOCKD *xd,
is_motion_variation_allowed_compound(mbmi)) {
if (!check_num_overlappable_neighbors(mbmi)) return SIMPLE_TRANSLATION;
assert(!has_second_ref(mbmi));
if (mbmi->num_proj_ref[0] >= 1 &&
if (mbmi->num_proj_ref >= 1 &&
(allow_warped_motion && !av1_is_scaled(&(xd->block_refs[0]->sf)))) {
if (xd->cur_frame_force_integer_mv) {
return OBMC_CAUSAL;
@ -1174,4 +1173,4 @@ static INLINE int av1_get_max_eob(TX_SIZE tx_size) {
} // extern "C"
#endif
#endif // AV1_COMMON_BLOCKD_H_
#endif // AOM_AV1_COMMON_BLOCKD_H_

View file

@ -8,8 +8,8 @@
* Media Patent License 1.0 was not distributed with this source code in the
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AV1_COMMON_CDEF_H_
#define AV1_COMMON_CDEF_H_
#ifndef AOM_AV1_COMMON_CDEF_H_
#define AOM_AV1_COMMON_CDEF_H_
#define CDEF_STRENGTH_BITS 6
@ -48,4 +48,4 @@ void av1_cdef_search(YV12_BUFFER_CONFIG *frame, const YV12_BUFFER_CONFIG *ref,
#ifdef __cplusplus
} // extern "C"
#endif
#endif // AV1_COMMON_CDEF_H_
#endif // AOM_AV1_COMMON_CDEF_H_

View file

@ -9,8 +9,8 @@
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#if !defined(_CDEF_BLOCK_H)
#define _CDEF_BLOCK_H (1)
#ifndef AOM_AV1_COMMON_CDEF_BLOCK_H_
#define AOM_AV1_COMMON_CDEF_BLOCK_H_
#include "av1/common/odintrin.h"
@ -56,4 +56,4 @@ void cdef_filter_fb(uint8_t *dst8, uint16_t *dst16, int dstride, uint16_t *in,
cdef_list *dlist, int cdef_count, int level,
int sec_strength, int pri_damping, int sec_damping,
int coeff_shift);
#endif
#endif // AOM_AV1_COMMON_CDEF_BLOCK_H_

View file

@ -9,6 +9,9 @@
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AOM_AV1_COMMON_CDEF_BLOCK_SIMD_H_
#define AOM_AV1_COMMON_CDEF_BLOCK_SIMD_H_
#include "config/av1_rtcd.h"
#include "av1/common/cdef_block.h"
@ -913,3 +916,5 @@ void SIMD_FUNC(copy_rect8_16bit_to_16bit)(uint16_t *dst, int dstride,
}
}
}
#endif // AOM_AV1_COMMON_CDEF_BLOCK_SIMD_H_

View file

@ -9,8 +9,8 @@
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AV1_COMMON_CFL_H_
#define AV1_COMMON_CFL_H_
#ifndef AOM_AV1_COMMON_CFL_H_
#define AOM_AV1_COMMON_CFL_H_
#include "av1/common/blockd.h"
#include "av1/common/onyxc_int.h"
@ -299,4 +299,4 @@ void cfl_predict_hbd_null(const int16_t *pred_buf_q3, uint16_t *dst,
return pred[tx_size % TX_SIZES_ALL]; \
}
#endif // AV1_COMMON_CFL_H_
#endif // AOM_AV1_COMMON_CFL_H_

View file

@ -9,8 +9,8 @@
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AV1_COMMON_COMMON_H_
#define AV1_COMMON_COMMON_H_
#ifndef AOM_AV1_COMMON_COMMON_H_
#define AOM_AV1_COMMON_COMMON_H_
/* Interface header for common constant data structures and lookup tables */
@ -60,4 +60,4 @@ static INLINE int get_unsigned_bits(unsigned int num_values) {
} // extern "C"
#endif
#endif // AV1_COMMON_COMMON_H_
#endif // AOM_AV1_COMMON_COMMON_H_

View file

@ -9,8 +9,8 @@
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AV1_COMMON_COMMON_DATA_H_
#define AV1_COMMON_COMMON_DATA_H_
#ifndef AOM_AV1_COMMON_COMMON_DATA_H_
#define AOM_AV1_COMMON_COMMON_DATA_H_
#include "av1/common/enums.h"
#include "aom/aom_integer.h"
@ -20,34 +20,43 @@
extern "C" {
#endif
// Log 2 conversion lookup tables in units of mode info(4x4).
// Log 2 conversion lookup tables in units of mode info (4x4).
// The Mi_Width_Log2 table in the spec (Section 9.3. Conversion tables).
static const uint8_t mi_size_wide_log2[BLOCK_SIZES_ALL] = {
0, 0, 1, 1, 1, 2, 2, 2, 3, 3, 3, 4, 4, 4, 5, 5, 0, 2, 1, 3, 2, 4
};
// The Mi_Height_Log2 table in the spec (Section 9.3. Conversion tables).
static const uint8_t mi_size_high_log2[BLOCK_SIZES_ALL] = {
0, 1, 0, 1, 2, 1, 2, 3, 2, 3, 4, 3, 4, 5, 4, 5, 2, 0, 3, 1, 4, 2
};
// Width/height lookup tables in units of mode info (4x4).
// The Num_4x4_Blocks_Wide table in the spec (Section 9.3. Conversion tables).
static const uint8_t mi_size_wide[BLOCK_SIZES_ALL] = {
1, 1, 2, 2, 2, 4, 4, 4, 8, 8, 8, 16, 16, 16, 32, 32, 1, 4, 2, 8, 4, 16
};
// The Num_4x4_Blocks_High table in the spec (Section 9.3. Conversion tables).
static const uint8_t mi_size_high[BLOCK_SIZES_ALL] = {
1, 2, 1, 2, 4, 2, 4, 8, 4, 8, 16, 8, 16, 32, 16, 32, 4, 1, 8, 2, 16, 4
};
// Width/height lookup tables in units of various block sizes
// Width/height lookup tables in units of samples.
// The Block_Width table in the spec (Section 9.3. Conversion tables).
static const uint8_t block_size_wide[BLOCK_SIZES_ALL] = {
4, 4, 8, 8, 8, 16, 16, 16, 32, 32, 32,
64, 64, 64, 128, 128, 4, 16, 8, 32, 16, 64
};
// The Block_Height table in the spec (Section 9.3. Conversion tables).
static const uint8_t block_size_high[BLOCK_SIZES_ALL] = {
4, 8, 4, 8, 16, 8, 16, 32, 16, 32, 64,
32, 64, 128, 64, 128, 16, 4, 32, 8, 64, 16
};
// AOMMIN(3, AOMMIN(b_width_log2(bsize), b_height_log2(bsize)))
// Maps a block size to a context.
// The Size_Group table in the spec (Section 9.3. Conversion tables).
// AOMMIN(3, AOMMIN(mi_size_wide_log2(bsize), mi_size_high_log2(bsize)))
static const uint8_t size_group_lookup[BLOCK_SIZES_ALL] = {
0, 0, 0, 1, 1, 1, 2, 2, 2, 3, 3, 3, 3, 3, 3, 3, 0, 0, 1, 1, 2, 2
};
@ -56,6 +65,8 @@ static const uint8_t num_pels_log2_lookup[BLOCK_SIZES_ALL] = {
4, 5, 5, 6, 7, 7, 8, 9, 9, 10, 11, 11, 12, 13, 13, 14, 6, 6, 8, 8, 10, 10
};
// A compressed version of the Partition_Subsize table in the spec (9.3.
// Conversion tables), for square block sizes only.
/* clang-format off */
static const BLOCK_SIZE subsize_lookup[EXT_PARTITION_TYPES][SQR_BLOCK_SIZES] = {
{ // PARTITION_NONE
@ -350,34 +361,36 @@ static const TX_SIZE tx_mode_to_biggest_tx_size[TX_MODES] = {
TX_64X64, // TX_MODE_LARGEST
TX_64X64, // TX_MODE_SELECT
};
/* clang-format on */
// The Subsampled_Size table in the spec (Section 5.11.38. Get plane residual
// size function).
static const BLOCK_SIZE ss_size_lookup[BLOCK_SIZES_ALL][2][2] = {
// ss_x == 0 ss_x == 0 ss_x == 1 ss_x == 1
// ss_y == 0 ss_y == 1 ss_y == 0 ss_y == 1
{ { BLOCK_4X4, BLOCK_4X4 }, { BLOCK_4X4, BLOCK_4X4 } },
{ { BLOCK_4X8, BLOCK_4X4 }, { BLOCK_4X4, BLOCK_4X4 } },
{ { BLOCK_8X4, BLOCK_4X4 }, { BLOCK_4X4, BLOCK_4X4 } },
{ { BLOCK_8X8, BLOCK_8X4 }, { BLOCK_4X8, BLOCK_4X4 } },
{ { BLOCK_8X16, BLOCK_8X8 }, { BLOCK_4X16, BLOCK_4X8 } },
{ { BLOCK_16X8, BLOCK_16X4 }, { BLOCK_8X8, BLOCK_8X4 } },
{ { BLOCK_16X16, BLOCK_16X8 }, { BLOCK_8X16, BLOCK_8X8 } },
{ { BLOCK_16X32, BLOCK_16X16 }, { BLOCK_8X32, BLOCK_8X16 } },
{ { BLOCK_32X16, BLOCK_32X8 }, { BLOCK_16X16, BLOCK_16X8 } },
{ { BLOCK_32X32, BLOCK_32X16 }, { BLOCK_16X32, BLOCK_16X16 } },
{ { BLOCK_32X64, BLOCK_32X32 }, { BLOCK_16X64, BLOCK_16X32 } },
{ { BLOCK_64X32, BLOCK_64X16 }, { BLOCK_32X32, BLOCK_32X16 } },
{ { BLOCK_64X64, BLOCK_64X32 }, { BLOCK_32X64, BLOCK_32X32 } },
{ { BLOCK_64X128, BLOCK_64X64 }, { BLOCK_INVALID, BLOCK_32X64 } },
{ { BLOCK_128X64, BLOCK_INVALID }, { BLOCK_64X64, BLOCK_64X32 } },
{ { BLOCK_128X128, BLOCK_128X64 }, { BLOCK_64X128, BLOCK_64X64 } },
{ { BLOCK_4X16, BLOCK_4X8 }, { BLOCK_4X16, BLOCK_4X8 } },
{ { BLOCK_16X4, BLOCK_16X4 }, { BLOCK_8X4, BLOCK_8X4 } },
{ { BLOCK_8X32, BLOCK_8X16 }, { BLOCK_INVALID, BLOCK_4X16 } },
{ { BLOCK_32X8, BLOCK_INVALID }, { BLOCK_16X8, BLOCK_16X4 } },
{ { BLOCK_16X64, BLOCK_16X32 }, { BLOCK_INVALID, BLOCK_8X32 } },
{ { BLOCK_64X16, BLOCK_INVALID }, { BLOCK_32X16, BLOCK_32X8 } }
// ss_x == 0 ss_x == 0 ss_x == 1 ss_x == 1
// ss_y == 0 ss_y == 1 ss_y == 0 ss_y == 1
{ { BLOCK_4X4, BLOCK_4X4 }, { BLOCK_4X4, BLOCK_4X4 } },
{ { BLOCK_4X8, BLOCK_4X4 }, { BLOCK_INVALID, BLOCK_4X4 } },
{ { BLOCK_8X4, BLOCK_INVALID }, { BLOCK_4X4, BLOCK_4X4 } },
{ { BLOCK_8X8, BLOCK_8X4 }, { BLOCK_4X8, BLOCK_4X4 } },
{ { BLOCK_8X16, BLOCK_8X8 }, { BLOCK_INVALID, BLOCK_4X8 } },
{ { BLOCK_16X8, BLOCK_INVALID }, { BLOCK_8X8, BLOCK_8X4 } },
{ { BLOCK_16X16, BLOCK_16X8 }, { BLOCK_8X16, BLOCK_8X8 } },
{ { BLOCK_16X32, BLOCK_16X16 }, { BLOCK_INVALID, BLOCK_8X16 } },
{ { BLOCK_32X16, BLOCK_INVALID }, { BLOCK_16X16, BLOCK_16X8 } },
{ { BLOCK_32X32, BLOCK_32X16 }, { BLOCK_16X32, BLOCK_16X16 } },
{ { BLOCK_32X64, BLOCK_32X32 }, { BLOCK_INVALID, BLOCK_16X32 } },
{ { BLOCK_64X32, BLOCK_INVALID }, { BLOCK_32X32, BLOCK_32X16 } },
{ { BLOCK_64X64, BLOCK_64X32 }, { BLOCK_32X64, BLOCK_32X32 } },
{ { BLOCK_64X128, BLOCK_64X64 }, { BLOCK_INVALID, BLOCK_32X64 } },
{ { BLOCK_128X64, BLOCK_INVALID }, { BLOCK_64X64, BLOCK_64X32 } },
{ { BLOCK_128X128, BLOCK_128X64 }, { BLOCK_64X128, BLOCK_64X64 } },
{ { BLOCK_4X16, BLOCK_4X8 }, { BLOCK_INVALID, BLOCK_4X8 } },
{ { BLOCK_16X4, BLOCK_INVALID }, { BLOCK_8X4, BLOCK_8X4 } },
{ { BLOCK_8X32, BLOCK_8X16 }, { BLOCK_INVALID, BLOCK_4X16 } },
{ { BLOCK_32X8, BLOCK_INVALID }, { BLOCK_16X8, BLOCK_16X4 } },
{ { BLOCK_16X64, BLOCK_16X32 }, { BLOCK_INVALID, BLOCK_8X32 } },
{ { BLOCK_64X16, BLOCK_INVALID }, { BLOCK_32X16, BLOCK_32X8 } }
};
/* clang-format on */
// Generates 5 bit field in which each bit set to 1 represents
// a blocksize partition 11111 means we split 128x128, 64x64, 32x32, 16x16
@ -430,4 +443,4 @@ static const int quant_dist_lookup_table[2][4][2] = {
} // extern "C"
#endif
#endif // AV1_COMMON_COMMON_DATA_H_
#endif // AOM_AV1_COMMON_COMMON_DATA_H_

View file

@ -173,6 +173,7 @@ void av1_convolve_x_sr_c(const uint8_t *src, int src_stride, uint8_t *dst,
// horizontal filter
const int16_t *x_filter = av1_get_interp_filter_subpel_kernel(
filter_params_x, subpel_x_q4 & SUBPEL_MASK);
for (int y = 0; y < h; ++y) {
for (int x = 0; x < w; ++x) {
int32_t res = 0;
@ -510,31 +511,73 @@ static void convolve_2d_scale_wrapper(
y_step_qn, conv_params);
}
// TODO(huisu@google.com): bilinear filtering only needs 2 taps in general. So
// we may create optimized code to do 2-tap filtering for all bilinear filtering
// usages, not just IntraBC.
static void convolve_2d_for_intrabc(const uint8_t *src, int src_stride,
uint8_t *dst, int dst_stride, int w, int h,
int subpel_x_q4, int subpel_y_q4,
ConvolveParams *conv_params) {
const InterpFilterParams *filter_params_x =
subpel_x_q4 ? &av1_intrabc_filter_params : NULL;
const InterpFilterParams *filter_params_y =
subpel_y_q4 ? &av1_intrabc_filter_params : NULL;
if (subpel_x_q4 != 0 && subpel_y_q4 != 0) {
av1_convolve_2d_sr_c(src, src_stride, dst, dst_stride, w, h,
filter_params_x, filter_params_y, 0, 0, conv_params);
} else if (subpel_x_q4 != 0) {
av1_convolve_x_sr_c(src, src_stride, dst, dst_stride, w, h, filter_params_x,
filter_params_y, 0, 0, conv_params);
} else {
av1_convolve_y_sr_c(src, src_stride, dst, dst_stride, w, h, filter_params_x,
filter_params_y, 0, 0, conv_params);
}
}
void av1_convolve_2d_facade(const uint8_t *src, int src_stride, uint8_t *dst,
int dst_stride, int w, int h,
InterpFilters interp_filters, const int subpel_x_q4,
int x_step_q4, const int subpel_y_q4, int y_step_q4,
int scaled, ConvolveParams *conv_params,
const struct scale_factors *sf) {
const struct scale_factors *sf, int is_intrabc) {
assert(IMPLIES(is_intrabc, !scaled));
(void)x_step_q4;
(void)y_step_q4;
(void)dst;
(void)dst_stride;
InterpFilter filter_x = av1_extract_interp_filter(interp_filters, 1);
InterpFilter filter_y = av1_extract_interp_filter(interp_filters, 0);
const InterpFilterParams *filter_params_x =
av1_get_interp_filter_params_with_block_size(filter_x, w);
const InterpFilterParams *filter_params_y =
av1_get_interp_filter_params_with_block_size(filter_y, h);
if (scaled)
if (is_intrabc && (subpel_x_q4 != 0 || subpel_y_q4 != 0)) {
convolve_2d_for_intrabc(src, src_stride, dst, dst_stride, w, h, subpel_x_q4,
subpel_y_q4, conv_params);
return;
}
InterpFilter filter_x = 0;
InterpFilter filter_y = 0;
const int need_filter_params_x = (subpel_x_q4 != 0) | scaled;
const int need_filter_params_y = (subpel_y_q4 != 0) | scaled;
if (need_filter_params_x)
filter_x = av1_extract_interp_filter(interp_filters, 1);
if (need_filter_params_y)
filter_y = av1_extract_interp_filter(interp_filters, 0);
const InterpFilterParams *filter_params_x =
need_filter_params_x
? av1_get_interp_filter_params_with_block_size(filter_x, w)
: NULL;
const InterpFilterParams *filter_params_y =
need_filter_params_y
? av1_get_interp_filter_params_with_block_size(filter_y, h)
: NULL;
if (scaled) {
convolve_2d_scale_wrapper(src, src_stride, dst, dst_stride, w, h,
filter_params_x, filter_params_y, subpel_x_q4,
x_step_q4, subpel_y_q4, y_step_q4, conv_params);
else
} else {
sf->convolve[subpel_x_q4 != 0][subpel_y_q4 != 0][conv_params->is_compound](
src, src_stride, dst, dst_stride, w, h, filter_params_x,
filter_params_y, subpel_x_q4, subpel_y_q4, conv_params);
}
}
void av1_highbd_convolve_2d_copy_sr_c(
@ -964,24 +1007,68 @@ void av1_highbd_convolve_2d_scale_c(const uint16_t *src, int src_stride,
}
}
static void highbd_convolve_2d_for_intrabc(const uint16_t *src, int src_stride,
uint16_t *dst, int dst_stride, int w,
int h, int subpel_x_q4,
int subpel_y_q4,
ConvolveParams *conv_params,
int bd) {
const InterpFilterParams *filter_params_x =
subpel_x_q4 ? &av1_intrabc_filter_params : NULL;
const InterpFilterParams *filter_params_y =
subpel_y_q4 ? &av1_intrabc_filter_params : NULL;
if (subpel_x_q4 != 0 && subpel_y_q4 != 0) {
av1_highbd_convolve_2d_sr_c(src, src_stride, dst, dst_stride, w, h,
filter_params_x, filter_params_y, 0, 0,
conv_params, bd);
} else if (subpel_x_q4 != 0) {
av1_highbd_convolve_x_sr_c(src, src_stride, dst, dst_stride, w, h,
filter_params_x, filter_params_y, 0, 0,
conv_params, bd);
} else {
av1_highbd_convolve_y_sr_c(src, src_stride, dst, dst_stride, w, h,
filter_params_x, filter_params_y, 0, 0,
conv_params, bd);
}
}
void av1_highbd_convolve_2d_facade(const uint8_t *src8, int src_stride,
uint8_t *dst8, int dst_stride, int w, int h,
InterpFilters interp_filters,
const int subpel_x_q4, int x_step_q4,
const int subpel_y_q4, int y_step_q4,
int scaled, ConvolveParams *conv_params,
const struct scale_factors *sf, int bd) {
const struct scale_factors *sf,
int is_intrabc, int bd) {
assert(IMPLIES(is_intrabc, !scaled));
(void)x_step_q4;
(void)y_step_q4;
(void)dst_stride;
const uint16_t *src = CONVERT_TO_SHORTPTR(src8);
InterpFilter filter_x = av1_extract_interp_filter(interp_filters, 1);
InterpFilter filter_y = av1_extract_interp_filter(interp_filters, 0);
if (is_intrabc && (subpel_x_q4 != 0 || subpel_y_q4 != 0)) {
uint16_t *dst = CONVERT_TO_SHORTPTR(dst8);
highbd_convolve_2d_for_intrabc(src, src_stride, dst, dst_stride, w, h,
subpel_x_q4, subpel_y_q4, conv_params, bd);
return;
}
InterpFilter filter_x = 0;
InterpFilter filter_y = 0;
const int need_filter_params_x = (subpel_x_q4 != 0) | scaled;
const int need_filter_params_y = (subpel_y_q4 != 0) | scaled;
if (need_filter_params_x)
filter_x = av1_extract_interp_filter(interp_filters, 1);
if (need_filter_params_y)
filter_y = av1_extract_interp_filter(interp_filters, 0);
const InterpFilterParams *filter_params_x =
av1_get_interp_filter_params_with_block_size(filter_x, w);
need_filter_params_x
? av1_get_interp_filter_params_with_block_size(filter_x, w)
: NULL;
const InterpFilterParams *filter_params_y =
av1_get_interp_filter_params_with_block_size(filter_y, h);
need_filter_params_y
? av1_get_interp_filter_params_with_block_size(filter_y, h)
: NULL;
if (scaled) {
uint16_t *dst = CONVERT_TO_SHORTPTR(dst8);
@ -1111,7 +1198,8 @@ void av1_wiener_convolve_add_src_c(const uint8_t *src, ptrdiff_t src_stride,
uint16_t temp[WIENER_MAX_EXT_SIZE * MAX_SB_SIZE];
const int intermediate_height =
(((h - 1) * y_step_q4 + y0_q4) >> SUBPEL_BITS) + SUBPEL_TAPS;
(((h - 1) * y_step_q4 + y0_q4) >> SUBPEL_BITS) + SUBPEL_TAPS - 1;
memset(temp + (intermediate_height * MAX_SB_SIZE), 0, MAX_SB_SIZE);
assert(w <= MAX_SB_SIZE);
assert(h <= MAX_SB_SIZE);

View file

@ -9,8 +9,8 @@
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AV1_COMMON_AV1_CONVOLVE_H_
#define AV1_COMMON_AV1_CONVOLVE_H_
#ifndef AOM_AV1_COMMON_CONVOLVE_H_
#define AOM_AV1_COMMON_CONVOLVE_H_
#include "av1/common/filter.h"
#ifdef __cplusplus
@ -19,7 +19,6 @@ extern "C" {
typedef uint16_t CONV_BUF_TYPE;
typedef struct ConvolveParams {
int ref;
int do_average;
CONV_BUF_TYPE *dst;
int dst_stride;
@ -59,15 +58,13 @@ void av1_convolve_2d_facade(const uint8_t *src, int src_stride, uint8_t *dst,
InterpFilters interp_filters, const int subpel_x_q4,
int x_step_q4, const int subpel_y_q4, int y_step_q4,
int scaled, ConvolveParams *conv_params,
const struct scale_factors *sf);
const struct scale_factors *sf, int is_intrabc);
static INLINE ConvolveParams get_conv_params_no_round(int ref, int do_average,
int plane,
static INLINE ConvolveParams get_conv_params_no_round(int do_average, int plane,
CONV_BUF_TYPE *dst,
int dst_stride,
int is_compound, int bd) {
ConvolveParams conv_params;
conv_params.ref = ref;
conv_params.do_average = do_average;
assert(IMPLIES(do_average, is_compound));
conv_params.is_compound = is_compound;
@ -88,15 +85,14 @@ static INLINE ConvolveParams get_conv_params_no_round(int ref, int do_average,
return conv_params;
}
static INLINE ConvolveParams get_conv_params(int ref, int do_average, int plane,
static INLINE ConvolveParams get_conv_params(int do_average, int plane,
int bd) {
return get_conv_params_no_round(ref, do_average, plane, NULL, 0, 0, bd);
return get_conv_params_no_round(do_average, plane, NULL, 0, 0, bd);
}
static INLINE ConvolveParams get_conv_params_wiener(int bd) {
ConvolveParams conv_params;
(void)bd;
conv_params.ref = 0;
conv_params.do_average = 0;
conv_params.is_compound = 0;
conv_params.round_0 = WIENER_ROUND0_BITS;
@ -119,10 +115,11 @@ void av1_highbd_convolve_2d_facade(const uint8_t *src8, int src_stride,
const int subpel_x_q4, int x_step_q4,
const int subpel_y_q4, int y_step_q4,
int scaled, ConvolveParams *conv_params,
const struct scale_factors *sf, int bd);
const struct scale_factors *sf,
int is_intrabc, int bd);
#ifdef __cplusplus
} // extern "C"
#endif
#endif // AV1_COMMON_AV1_CONVOLVE_H_
#endif // AOM_AV1_COMMON_CONVOLVE_H_

View file

@ -9,8 +9,8 @@
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AV1_COMMON_ENTROPY_H_
#define AV1_COMMON_ENTROPY_H_
#ifndef AOM_AV1_COMMON_ENTROPY_H_
#define AOM_AV1_COMMON_ENTROPY_H_
#include "config/aom_config.h"
@ -178,4 +178,4 @@ static INLINE TX_SIZE get_txsize_entropy_ctx(TX_SIZE txsize) {
} // extern "C"
#endif
#endif // AV1_COMMON_ENTROPY_H_
#endif // AOM_AV1_COMMON_ENTROPY_H_

View file

@ -9,8 +9,8 @@
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AV1_COMMON_ENTROPYMODE_H_
#define AV1_COMMON_ENTROPYMODE_H_
#ifndef AOM_AV1_COMMON_ENTROPYMODE_H_
#define AOM_AV1_COMMON_ENTROPYMODE_H_
#include "av1/common/entropy.h"
#include "av1/common/entropymv.h"
@ -186,6 +186,8 @@ void av1_set_default_mode_deltas(int8_t *mode_deltas);
void av1_setup_frame_contexts(struct AV1Common *cm);
void av1_setup_past_independence(struct AV1Common *cm);
// Returns (int)ceil(log2(n)).
// NOTE: This implementation only works for n <= 2^30.
static INLINE int av1_ceil_log2(int n) {
if (n < 2) return 0;
int i = 1, p = 2;
@ -207,4 +209,4 @@ int av1_get_palette_color_index_context(const uint8_t *color_map, int stride,
} // extern "C"
#endif
#endif // AV1_COMMON_ENTROPYMODE_H_
#endif // AOM_AV1_COMMON_ENTROPYMODE_H_

View file

@ -60,61 +60,6 @@ static const nmv_context default_nmv_context = {
} },
};
static const uint8_t log_in_base_2[] = {
0, 0, 1, 1, 2, 2, 2, 2, 3, 3, 3, 3, 3, 3, 3, 3, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4,
4, 4, 4, 4, 4, 4, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5,
5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6,
6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6,
6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 7, 7,
7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7,
7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7,
7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7,
7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7,
7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 8, 8, 8, 8,
8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8,
8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8,
8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8,
8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8,
8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8,
8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8,
8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8,
8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8,
8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8,
8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 8, 9, 9, 9, 9, 9, 9, 9, 9,
9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9,
9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9,
9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9,
9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9,
9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9,
9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9,
9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9,
9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9,
9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9,
9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9,
9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9,
9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9,
9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9,
9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9,
9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9,
9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9,
9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9,
9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9,
9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9,
9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 10
};
static INLINE int mv_class_base(MV_CLASS_TYPE c) {
return c ? CLASS0_SIZE << (c + 2) : 0;
}
MV_CLASS_TYPE av1_get_mv_class(int z, int *offset) {
const MV_CLASS_TYPE c = (z >= CLASS0_SIZE * 4096)
? MV_CLASS_10
: (MV_CLASS_TYPE)log_in_base_2[z >> 3];
if (offset) *offset = z - mv_class_base(c);
return c;
}
void av1_init_mv_probs(AV1_COMMON *cm) {
// NB: this sets CDFs too
cm->fc->nmvc = default_nmv_context;

View file

@ -9,8 +9,8 @@
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AV1_COMMON_ENTROPYMV_H_
#define AV1_COMMON_ENTROPYMV_H_
#ifndef AOM_AV1_COMMON_ENTROPYMV_H_
#define AOM_AV1_COMMON_ENTROPYMV_H_
#include "config/aom_config.h"
@ -91,16 +91,6 @@ typedef struct {
nmv_component comps[2];
} nmv_context;
static INLINE MV_JOINT_TYPE av1_get_mv_joint(const MV *mv) {
if (mv->row == 0) {
return mv->col == 0 ? MV_JOINT_ZERO : MV_JOINT_HNZVZ;
} else {
return mv->col == 0 ? MV_JOINT_HZVNZ : MV_JOINT_HNZVNZ;
}
}
MV_CLASS_TYPE av1_get_mv_class(int z, int *offset);
typedef enum {
MV_SUBPEL_NONE = -1,
MV_SUBPEL_LOW_PRECISION = 0,
@ -111,4 +101,4 @@ typedef enum {
} // extern "C"
#endif
#endif // AV1_COMMON_ENTROPYMV_H_
#endif // AOM_AV1_COMMON_ENTROPYMV_H_

View file

@ -9,8 +9,8 @@
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AV1_COMMON_ENUMS_H_
#define AV1_COMMON_ENUMS_H_
#ifndef AOM_AV1_COMMON_ENUMS_H_
#define AOM_AV1_COMMON_ENUMS_H_
#include "config/aom_config.h"
@ -274,7 +274,7 @@ typedef enum ATTRIBUTE_PACKED {
TX_TYPES,
} TX_TYPE;
typedef enum {
typedef enum ATTRIBUTE_PACKED {
REG_REG,
REG_SMOOTH,
REG_SHARP,
@ -438,6 +438,8 @@ typedef enum ATTRIBUTE_PACKED {
COMP_INTER_MODE_START = NEAREST_NEARESTMV,
COMP_INTER_MODE_END = MB_MODE_COUNT,
COMP_INTER_MODE_NUM = COMP_INTER_MODE_END - COMP_INTER_MODE_START,
INTER_MODE_START = NEARESTMV,
INTER_MODE_END = MB_MODE_COUNT,
INTRA_MODES = PAETH_PRED + 1, // PAETH_PRED has to be the last intra mode.
INTRA_INVALID = MB_MODE_COUNT // For uv_mode in inter blocks
} PREDICTION_MODE;
@ -478,7 +480,7 @@ typedef enum ATTRIBUTE_PACKED {
INTERINTRA_MODES
} INTERINTRA_MODE;
typedef enum {
typedef enum ATTRIBUTE_PACKED {
COMPOUND_AVERAGE,
COMPOUND_WEDGE,
COMPOUND_DIFFWTD,
@ -614,4 +616,4 @@ typedef enum ATTRIBUTE_PACKED {
} // extern "C"
#endif
#endif // AV1_COMMON_ENUMS_H_
#endif // AOM_AV1_COMMON_ENUMS_H_

View file

@ -9,8 +9,8 @@
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AV1_COMMON_FILTER_H_
#define AV1_COMMON_FILTER_H_
#ifndef AOM_AV1_COMMON_FILTER_H_
#define AOM_AV1_COMMON_FILTER_H_
#include <assert.h>
@ -139,6 +139,17 @@ static const InterpFilterParams
BILINEAR }
};
// A special 2-tap bilinear filter for IntraBC chroma. IntraBC uses full pixel
// MV for luma. If sub-sampling exists, chroma may possibly use half-pel MV.
DECLARE_ALIGNED(256, static const int16_t, av1_intrabc_bilinear_filter[2]) = {
64,
64,
};
static const InterpFilterParams av1_intrabc_filter_params = {
av1_intrabc_bilinear_filter, 2, 0, BILINEAR
};
DECLARE_ALIGNED(256, static const InterpKernel,
av1_sub_pel_filters_4[SUBPEL_SHIFTS]) = {
{ 0, 0, 0, 128, 0, 0, 0, 0 }, { 0, 0, -4, 126, 8, -2, 0, 0 },
@ -181,6 +192,11 @@ av1_get_interp_filter_params_with_block_size(const InterpFilter interp_filter,
return &av1_interp_filter_params_list[interp_filter];
}
static INLINE const InterpFilterParams *av1_get_4tap_interp_filter_params(
const InterpFilter interp_filter) {
return &av1_interp_4tap[interp_filter];
}
static INLINE const int16_t *av1_get_interp_filter_kernel(
const InterpFilter interp_filter) {
return av1_interp_filter_params_list[interp_filter].filter_ptr;
@ -195,4 +211,4 @@ static INLINE const int16_t *av1_get_interp_filter_subpel_kernel(
} // extern "C"
#endif
#endif // AV1_COMMON_FILTER_H_
#endif // AOM_AV1_COMMON_FILTER_H_

View file

@ -38,6 +38,17 @@ void av1_free_internal_frame_buffers(InternalFrameBufferList *list) {
list->int_fb = NULL;
}
void av1_zero_unused_internal_frame_buffers(InternalFrameBufferList *list) {
int i;
assert(list != NULL);
for (i = 0; i < list->num_internal_frame_buffers; ++i) {
if (list->int_fb[i].data && !list->int_fb[i].in_use)
memset(list->int_fb[i].data, 0, list->int_fb[i].size);
}
}
int av1_get_frame_buffer(void *cb_priv, size_t min_size,
aom_codec_frame_buffer_t *fb) {
int i;

View file

@ -9,8 +9,8 @@
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AV1_COMMON_FRAME_BUFFERS_H_
#define AV1_COMMON_FRAME_BUFFERS_H_
#ifndef AOM_AV1_COMMON_FRAME_BUFFERS_H_
#define AOM_AV1_COMMON_FRAME_BUFFERS_H_
#include "aom/aom_frame_buffer.h"
#include "aom/aom_integer.h"
@ -36,6 +36,12 @@ int av1_alloc_internal_frame_buffers(InternalFrameBufferList *list);
// Free any data allocated to the frame buffers.
void av1_free_internal_frame_buffers(InternalFrameBufferList *list);
// Zeros all unused internal frame buffers. In particular, this zeros the
// frame borders. Call this function after a sequence header change to
// re-initialize the frame borders for the different width, height, or bit
// depth.
void av1_zero_unused_internal_frame_buffers(InternalFrameBufferList *list);
// Callback used by libaom to request an external frame buffer. |cb_priv|
// Callback private data, which points to an InternalFrameBufferList.
// |min_size| is the minimum size in bytes needed to decode the next frame.
@ -51,4 +57,4 @@ int av1_release_frame_buffer(void *cb_priv, aom_codec_frame_buffer_t *fb);
} // extern "C"
#endif
#endif // AV1_COMMON_FRAME_BUFFERS_H_
#endif // AOM_AV1_COMMON_FRAME_BUFFERS_H_

View file

@ -31,21 +31,16 @@ int av1_get_tx_scale(const TX_SIZE tx_size) {
// that input and output could be the same buffer.
// idct
static void highbd_iwht4x4_add(const tran_low_t *input, uint8_t *dest,
int stride, int eob, int bd) {
void av1_highbd_iwht4x4_add(const tran_low_t *input, uint8_t *dest, int stride,
int eob, int bd) {
if (eob > 1)
av1_highbd_iwht4x4_16_add(input, dest, stride, bd);
else
av1_highbd_iwht4x4_1_add(input, dest, stride, bd);
}
static const int32_t *cast_to_int32(const tran_low_t *input) {
assert(sizeof(int32_t) == sizeof(tran_low_t));
return (const int32_t *)input;
}
void av1_highbd_inv_txfm_add_4x4(const tran_low_t *input, uint8_t *dest,
int stride, const TxfmParam *txfm_param) {
void av1_highbd_inv_txfm_add_4x4_c(const tran_low_t *input, uint8_t *dest,
int stride, const TxfmParam *txfm_param) {
assert(av1_ext_tx_used[txfm_param->tx_set_type][txfm_param->tx_type]);
int eob = txfm_param->eob;
int bd = txfm_param->bd;
@ -54,206 +49,150 @@ void av1_highbd_inv_txfm_add_4x4(const tran_low_t *input, uint8_t *dest,
const TX_TYPE tx_type = txfm_param->tx_type;
if (lossless) {
assert(tx_type == DCT_DCT);
highbd_iwht4x4_add(input, dest, stride, eob, bd);
av1_highbd_iwht4x4_add(input, dest, stride, eob, bd);
return;
}
switch (tx_type) {
// Assembly version doesn't support some transform types, so use C version
// for those.
case V_DCT:
case H_DCT:
case V_ADST:
case H_ADST:
case V_FLIPADST:
case H_FLIPADST:
case IDTX:
av1_inv_txfm2d_add_4x4_c(src, CONVERT_TO_SHORTPTR(dest), stride, tx_type,
bd);
break;
default:
av1_inv_txfm2d_add_4x4(src, CONVERT_TO_SHORTPTR(dest), stride, tx_type,
bd);
break;
}
av1_inv_txfm2d_add_4x4_c(src, CONVERT_TO_SHORTPTR(dest), stride, tx_type, bd);
}
static void highbd_inv_txfm_add_4x8(const tran_low_t *input, uint8_t *dest,
int stride, const TxfmParam *txfm_param) {
void av1_highbd_inv_txfm_add_4x8(const tran_low_t *input, uint8_t *dest,
int stride, const TxfmParam *txfm_param) {
assert(av1_ext_tx_used[txfm_param->tx_set_type][txfm_param->tx_type]);
const int32_t *src = cast_to_int32(input);
av1_inv_txfm2d_add_4x8(src, CONVERT_TO_SHORTPTR(dest), stride,
txfm_param->tx_type, txfm_param->bd);
av1_inv_txfm2d_add_4x8_c(src, CONVERT_TO_SHORTPTR(dest), stride,
txfm_param->tx_type, txfm_param->bd);
}
static void highbd_inv_txfm_add_8x4(const tran_low_t *input, uint8_t *dest,
int stride, const TxfmParam *txfm_param) {
void av1_highbd_inv_txfm_add_8x4(const tran_low_t *input, uint8_t *dest,
int stride, const TxfmParam *txfm_param) {
assert(av1_ext_tx_used[txfm_param->tx_set_type][txfm_param->tx_type]);
const int32_t *src = cast_to_int32(input);
av1_inv_txfm2d_add_8x4(src, CONVERT_TO_SHORTPTR(dest), stride,
txfm_param->tx_type, txfm_param->bd);
}
static void highbd_inv_txfm_add_8x16(const tran_low_t *input, uint8_t *dest,
int stride, const TxfmParam *txfm_param) {
const int32_t *src = cast_to_int32(input);
av1_inv_txfm2d_add_8x16(src, CONVERT_TO_SHORTPTR(dest), stride,
txfm_param->tx_type, txfm_param->bd);
}
static void highbd_inv_txfm_add_16x8(const tran_low_t *input, uint8_t *dest,
int stride, const TxfmParam *txfm_param) {
const int32_t *src = cast_to_int32(input);
av1_inv_txfm2d_add_16x8(src, CONVERT_TO_SHORTPTR(dest), stride,
txfm_param->tx_type, txfm_param->bd);
}
static void highbd_inv_txfm_add_16x32(const tran_low_t *input, uint8_t *dest,
int stride, const TxfmParam *txfm_param) {
const int32_t *src = cast_to_int32(input);
av1_inv_txfm2d_add_16x32(src, CONVERT_TO_SHORTPTR(dest), stride,
av1_inv_txfm2d_add_8x4_c(src, CONVERT_TO_SHORTPTR(dest), stride,
txfm_param->tx_type, txfm_param->bd);
}
static void highbd_inv_txfm_add_32x16(const tran_low_t *input, uint8_t *dest,
int stride, const TxfmParam *txfm_param) {
void av1_highbd_inv_txfm_add_16x32(const tran_low_t *input, uint8_t *dest,
int stride, const TxfmParam *txfm_param) {
const int32_t *src = cast_to_int32(input);
av1_inv_txfm2d_add_32x16(src, CONVERT_TO_SHORTPTR(dest), stride,
txfm_param->tx_type, txfm_param->bd);
av1_inv_txfm2d_add_16x32_c(src, CONVERT_TO_SHORTPTR(dest), stride,
txfm_param->tx_type, txfm_param->bd);
}
static void highbd_inv_txfm_add_16x4(const tran_low_t *input, uint8_t *dest,
int stride, const TxfmParam *txfm_param) {
void av1_highbd_inv_txfm_add_32x16(const tran_low_t *input, uint8_t *dest,
int stride, const TxfmParam *txfm_param) {
const int32_t *src = cast_to_int32(input);
av1_inv_txfm2d_add_16x4(src, CONVERT_TO_SHORTPTR(dest), stride,
txfm_param->tx_type, txfm_param->bd);
av1_inv_txfm2d_add_32x16_c(src, CONVERT_TO_SHORTPTR(dest), stride,
txfm_param->tx_type, txfm_param->bd);
}
static void highbd_inv_txfm_add_4x16(const tran_low_t *input, uint8_t *dest,
int stride, const TxfmParam *txfm_param) {
void av1_highbd_inv_txfm_add_16x4(const tran_low_t *input, uint8_t *dest,
int stride, const TxfmParam *txfm_param) {
const int32_t *src = cast_to_int32(input);
av1_inv_txfm2d_add_4x16(src, CONVERT_TO_SHORTPTR(dest), stride,
txfm_param->tx_type, txfm_param->bd);
av1_inv_txfm2d_add_16x4_c(src, CONVERT_TO_SHORTPTR(dest), stride,
txfm_param->tx_type, txfm_param->bd);
}
static void highbd_inv_txfm_add_32x8(const tran_low_t *input, uint8_t *dest,
int stride, const TxfmParam *txfm_param) {
void av1_highbd_inv_txfm_add_4x16(const tran_low_t *input, uint8_t *dest,
int stride, const TxfmParam *txfm_param) {
const int32_t *src = cast_to_int32(input);
av1_inv_txfm2d_add_32x8(src, CONVERT_TO_SHORTPTR(dest), stride,
txfm_param->tx_type, txfm_param->bd);
av1_inv_txfm2d_add_4x16_c(src, CONVERT_TO_SHORTPTR(dest), stride,
txfm_param->tx_type, txfm_param->bd);
}
static void highbd_inv_txfm_add_8x32(const tran_low_t *input, uint8_t *dest,
int stride, const TxfmParam *txfm_param) {
void av1_highbd_inv_txfm_add_32x8(const tran_low_t *input, uint8_t *dest,
int stride, const TxfmParam *txfm_param) {
const int32_t *src = cast_to_int32(input);
av1_inv_txfm2d_add_8x32(src, CONVERT_TO_SHORTPTR(dest), stride,
txfm_param->tx_type, txfm_param->bd);
av1_inv_txfm2d_add_32x8_c(src, CONVERT_TO_SHORTPTR(dest), stride,
txfm_param->tx_type, txfm_param->bd);
}
static void highbd_inv_txfm_add_32x64(const tran_low_t *input, uint8_t *dest,
int stride, const TxfmParam *txfm_param) {
void av1_highbd_inv_txfm_add_8x32(const tran_low_t *input, uint8_t *dest,
int stride, const TxfmParam *txfm_param) {
const int32_t *src = cast_to_int32(input);
av1_inv_txfm2d_add_32x64(src, CONVERT_TO_SHORTPTR(dest), stride,
txfm_param->tx_type, txfm_param->bd);
av1_inv_txfm2d_add_8x32_c(src, CONVERT_TO_SHORTPTR(dest), stride,
txfm_param->tx_type, txfm_param->bd);
}
static void highbd_inv_txfm_add_64x32(const tran_low_t *input, uint8_t *dest,
int stride, const TxfmParam *txfm_param) {
void av1_highbd_inv_txfm_add_32x64(const tran_low_t *input, uint8_t *dest,
int stride, const TxfmParam *txfm_param) {
const int32_t *src = cast_to_int32(input);
av1_inv_txfm2d_add_64x32(src, CONVERT_TO_SHORTPTR(dest), stride,
txfm_param->tx_type, txfm_param->bd);
av1_inv_txfm2d_add_32x64_c(src, CONVERT_TO_SHORTPTR(dest), stride,
txfm_param->tx_type, txfm_param->bd);
}
static void highbd_inv_txfm_add_16x64(const tran_low_t *input, uint8_t *dest,
int stride, const TxfmParam *txfm_param) {
void av1_highbd_inv_txfm_add_64x32(const tran_low_t *input, uint8_t *dest,
int stride, const TxfmParam *txfm_param) {
const int32_t *src = cast_to_int32(input);
av1_inv_txfm2d_add_16x64(src, CONVERT_TO_SHORTPTR(dest), stride,
txfm_param->tx_type, txfm_param->bd);
av1_inv_txfm2d_add_64x32_c(src, CONVERT_TO_SHORTPTR(dest), stride,
txfm_param->tx_type, txfm_param->bd);
}
static void highbd_inv_txfm_add_64x16(const tran_low_t *input, uint8_t *dest,
int stride, const TxfmParam *txfm_param) {
void av1_highbd_inv_txfm_add_16x64(const tran_low_t *input, uint8_t *dest,
int stride, const TxfmParam *txfm_param) {
const int32_t *src = cast_to_int32(input);
av1_inv_txfm2d_add_64x16(src, CONVERT_TO_SHORTPTR(dest), stride,
txfm_param->tx_type, txfm_param->bd);
av1_inv_txfm2d_add_16x64_c(src, CONVERT_TO_SHORTPTR(dest), stride,
txfm_param->tx_type, txfm_param->bd);
}
static void highbd_inv_txfm_add_8x8(const tran_low_t *input, uint8_t *dest,
int stride, const TxfmParam *txfm_param) {
void av1_highbd_inv_txfm_add_64x16(const tran_low_t *input, uint8_t *dest,
int stride, const TxfmParam *txfm_param) {
const int32_t *src = cast_to_int32(input);
av1_inv_txfm2d_add_64x16_c(src, CONVERT_TO_SHORTPTR(dest), stride,
txfm_param->tx_type, txfm_param->bd);
}
void av1_highbd_inv_txfm_add_8x8_c(const tran_low_t *input, uint8_t *dest,
int stride, const TxfmParam *txfm_param) {
int bd = txfm_param->bd;
const TX_TYPE tx_type = txfm_param->tx_type;
const int32_t *src = cast_to_int32(input);
switch (tx_type) {
// Assembly version doesn't support some transform types, so use C version
// for those.
case V_DCT:
case H_DCT:
case V_ADST:
case H_ADST:
case V_FLIPADST:
case H_FLIPADST:
case IDTX:
av1_inv_txfm2d_add_8x8_c(src, CONVERT_TO_SHORTPTR(dest), stride, tx_type,
bd);
break;
default:
av1_inv_txfm2d_add_8x8(src, CONVERT_TO_SHORTPTR(dest), stride, tx_type,
bd);
break;
}
av1_inv_txfm2d_add_8x8_c(src, CONVERT_TO_SHORTPTR(dest), stride, tx_type, bd);
}
static void highbd_inv_txfm_add_16x16(const tran_low_t *input, uint8_t *dest,
int stride, const TxfmParam *txfm_param) {
void av1_highbd_inv_txfm_add_16x16_c(const tran_low_t *input, uint8_t *dest,
int stride, const TxfmParam *txfm_param) {
int bd = txfm_param->bd;
const TX_TYPE tx_type = txfm_param->tx_type;
const int32_t *src = cast_to_int32(input);
switch (tx_type) {
// Assembly version doesn't support some transform types, so use C version
// for those.
case V_DCT:
case H_DCT:
case V_ADST:
case H_ADST:
case V_FLIPADST:
case H_FLIPADST:
case IDTX:
av1_inv_txfm2d_add_16x16_c(src, CONVERT_TO_SHORTPTR(dest), stride,
tx_type, bd);
break;
default:
av1_inv_txfm2d_add_16x16(src, CONVERT_TO_SHORTPTR(dest), stride, tx_type,
bd);
break;
}
av1_inv_txfm2d_add_16x16_c(src, CONVERT_TO_SHORTPTR(dest), stride, tx_type,
bd);
}
static void highbd_inv_txfm_add_32x32(const tran_low_t *input, uint8_t *dest,
int stride, const TxfmParam *txfm_param) {
void av1_highbd_inv_txfm_add_8x16_c(const tran_low_t *input, uint8_t *dest,
int stride, const TxfmParam *txfm_param) {
const int32_t *src = cast_to_int32(input);
av1_inv_txfm2d_add_8x16_c(src, CONVERT_TO_SHORTPTR(dest), stride,
txfm_param->tx_type, txfm_param->bd);
}
void av1_highbd_inv_txfm_add_16x8_c(const tran_low_t *input, uint8_t *dest,
int stride, const TxfmParam *txfm_param) {
const int32_t *src = cast_to_int32(input);
av1_inv_txfm2d_add_16x8_c(src, CONVERT_TO_SHORTPTR(dest), stride,
txfm_param->tx_type, txfm_param->bd);
}
void av1_highbd_inv_txfm_add_32x32_c(const tran_low_t *input, uint8_t *dest,
int stride, const TxfmParam *txfm_param) {
const int bd = txfm_param->bd;
const TX_TYPE tx_type = txfm_param->tx_type;
const int32_t *src = cast_to_int32(input);
switch (tx_type) {
case DCT_DCT:
av1_inv_txfm2d_add_32x32(src, CONVERT_TO_SHORTPTR(dest), stride, tx_type,
bd);
break;
// Assembly version doesn't support IDTX, so use C version for it.
case IDTX:
av1_inv_txfm2d_add_32x32_c(src, CONVERT_TO_SHORTPTR(dest), stride,
tx_type, bd);
break;
default: assert(0);
}
av1_inv_txfm2d_add_32x32_c(src, CONVERT_TO_SHORTPTR(dest), stride, tx_type,
bd);
}
static void highbd_inv_txfm_add_64x64(const tran_low_t *input, uint8_t *dest,
int stride, const TxfmParam *txfm_param) {
void av1_highbd_inv_txfm_add_64x64_c(const tran_low_t *input, uint8_t *dest,
int stride, const TxfmParam *txfm_param) {
const int bd = txfm_param->bd;
const TX_TYPE tx_type = txfm_param->tx_type;
const int32_t *src = cast_to_int32(input);
assert(tx_type == DCT_DCT);
av1_inv_txfm2d_add_64x64(src, CONVERT_TO_SHORTPTR(dest), stride, tx_type, bd);
av1_inv_txfm2d_add_64x64_c(src, CONVERT_TO_SHORTPTR(dest), stride, tx_type,
bd);
}
static void init_txfm_param(const MACROBLOCKD *xd, int plane, TX_SIZE tx_size,
@ -270,70 +209,70 @@ static void init_txfm_param(const MACROBLOCKD *xd, int plane, TX_SIZE tx_size,
txfm_param->tx_size, is_inter_block(xd->mi[0]), reduced_tx_set);
}
static void highbd_inv_txfm_add(const tran_low_t *input, uint8_t *dest,
int stride, const TxfmParam *txfm_param) {
void av1_highbd_inv_txfm_add_c(const tran_low_t *input, uint8_t *dest,
int stride, const TxfmParam *txfm_param) {
assert(av1_ext_tx_used[txfm_param->tx_set_type][txfm_param->tx_type]);
const TX_SIZE tx_size = txfm_param->tx_size;
switch (tx_size) {
case TX_32X32:
highbd_inv_txfm_add_32x32(input, dest, stride, txfm_param);
av1_highbd_inv_txfm_add_32x32_c(input, dest, stride, txfm_param);
break;
case TX_16X16:
highbd_inv_txfm_add_16x16(input, dest, stride, txfm_param);
av1_highbd_inv_txfm_add_16x16_c(input, dest, stride, txfm_param);
break;
case TX_8X8:
highbd_inv_txfm_add_8x8(input, dest, stride, txfm_param);
av1_highbd_inv_txfm_add_8x8_c(input, dest, stride, txfm_param);
break;
case TX_4X8:
highbd_inv_txfm_add_4x8(input, dest, stride, txfm_param);
av1_highbd_inv_txfm_add_4x8(input, dest, stride, txfm_param);
break;
case TX_8X4:
highbd_inv_txfm_add_8x4(input, dest, stride, txfm_param);
av1_highbd_inv_txfm_add_8x4(input, dest, stride, txfm_param);
break;
case TX_8X16:
highbd_inv_txfm_add_8x16(input, dest, stride, txfm_param);
av1_highbd_inv_txfm_add_8x16_c(input, dest, stride, txfm_param);
break;
case TX_16X8:
highbd_inv_txfm_add_16x8(input, dest, stride, txfm_param);
av1_highbd_inv_txfm_add_16x8_c(input, dest, stride, txfm_param);
break;
case TX_16X32:
highbd_inv_txfm_add_16x32(input, dest, stride, txfm_param);
av1_highbd_inv_txfm_add_16x32(input, dest, stride, txfm_param);
break;
case TX_32X16:
highbd_inv_txfm_add_32x16(input, dest, stride, txfm_param);
av1_highbd_inv_txfm_add_32x16(input, dest, stride, txfm_param);
break;
case TX_64X64:
highbd_inv_txfm_add_64x64(input, dest, stride, txfm_param);
av1_highbd_inv_txfm_add_64x64_c(input, dest, stride, txfm_param);
break;
case TX_32X64:
highbd_inv_txfm_add_32x64(input, dest, stride, txfm_param);
av1_highbd_inv_txfm_add_32x64(input, dest, stride, txfm_param);
break;
case TX_64X32:
highbd_inv_txfm_add_64x32(input, dest, stride, txfm_param);
av1_highbd_inv_txfm_add_64x32(input, dest, stride, txfm_param);
break;
case TX_16X64:
highbd_inv_txfm_add_16x64(input, dest, stride, txfm_param);
av1_highbd_inv_txfm_add_16x64(input, dest, stride, txfm_param);
break;
case TX_64X16:
highbd_inv_txfm_add_64x16(input, dest, stride, txfm_param);
av1_highbd_inv_txfm_add_64x16(input, dest, stride, txfm_param);
break;
case TX_4X4:
// this is like av1_short_idct4x4 but has a special case around eob<=1
// which is significant (not just an optimization) for the lossless
// case.
av1_highbd_inv_txfm_add_4x4(input, dest, stride, txfm_param);
av1_highbd_inv_txfm_add_4x4_c(input, dest, stride, txfm_param);
break;
case TX_16X4:
highbd_inv_txfm_add_16x4(input, dest, stride, txfm_param);
av1_highbd_inv_txfm_add_16x4(input, dest, stride, txfm_param);
break;
case TX_4X16:
highbd_inv_txfm_add_4x16(input, dest, stride, txfm_param);
av1_highbd_inv_txfm_add_4x16(input, dest, stride, txfm_param);
break;
case TX_8X32:
highbd_inv_txfm_add_8x32(input, dest, stride, txfm_param);
av1_highbd_inv_txfm_add_8x32(input, dest, stride, txfm_param);
break;
case TX_32X8:
highbd_inv_txfm_add_32x8(input, dest, stride, txfm_param);
av1_highbd_inv_txfm_add_32x8(input, dest, stride, txfm_param);
break;
default: assert(0 && "Invalid transform size"); break;
}
@ -352,7 +291,8 @@ void av1_inv_txfm_add_c(const tran_low_t *dqcoeff, uint8_t *dst, int stride,
}
}
highbd_inv_txfm_add(dqcoeff, CONVERT_TO_BYTEPTR(tmp), tmp_stride, txfm_param);
av1_highbd_inv_txfm_add(dqcoeff, CONVERT_TO_BYTEPTR(tmp), tmp_stride,
txfm_param);
for (int r = 0; r < h; ++r) {
for (int c = 0; c < w; ++c) {
@ -375,7 +315,7 @@ void av1_inverse_transform_block(const MACROBLOCKD *xd,
assert(av1_ext_tx_used[txfm_param.tx_set_type][txfm_param.tx_type]);
if (txfm_param.is_hbd) {
highbd_inv_txfm_add(dqcoeff, dst, stride, &txfm_param);
av1_highbd_inv_txfm_add(dqcoeff, dst, stride, &txfm_param);
} else {
av1_inv_txfm_add(dqcoeff, dst, stride, &txfm_param);
}

View file

@ -9,8 +9,8 @@
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AV1_COMMON_IDCT_H_
#define AV1_COMMON_IDCT_H_
#ifndef AOM_AV1_COMMON_IDCT_H_
#define AOM_AV1_COMMON_IDCT_H_
#include "config/aom_config.h"
@ -36,11 +36,32 @@ void av1_inverse_transform_block(const MACROBLOCKD *xd,
const tran_low_t *dqcoeff, int plane,
TX_TYPE tx_type, TX_SIZE tx_size, uint8_t *dst,
int stride, int eob, int reduced_tx_set);
void av1_highbd_iwht4x4_add(const tran_low_t *input, uint8_t *dest, int stride,
int eob, int bd);
static INLINE const int32_t *cast_to_int32(const tran_low_t *input) {
assert(sizeof(int32_t) == sizeof(tran_low_t));
return (const int32_t *)input;
}
typedef void(highbd_inv_txfm_add)(const tran_low_t *input, uint8_t *dest,
int stride, const TxfmParam *param);
highbd_inv_txfm_add av1_highbd_inv_txfm_add_4x8;
highbd_inv_txfm_add av1_highbd_inv_txfm_add_8x4;
highbd_inv_txfm_add av1_highbd_inv_txfm_add_16x32;
highbd_inv_txfm_add av1_highbd_inv_txfm_add_32x16;
highbd_inv_txfm_add av1_highbd_inv_txfm_add_32x64;
highbd_inv_txfm_add av1_highbd_inv_txfm_add_64x32;
highbd_inv_txfm_add av1_highbd_inv_txfm_add_16x64;
highbd_inv_txfm_add av1_highbd_inv_txfm_add_64x16;
highbd_inv_txfm_add av1_highbd_inv_txfm_add_16x4;
highbd_inv_txfm_add av1_highbd_inv_txfm_add_4x16;
highbd_inv_txfm_add av1_highbd_inv_txfm_add_8x32;
highbd_inv_txfm_add av1_highbd_inv_txfm_add_32x8;
void av1_highbd_inv_txfm_add_4x4(const tran_low_t *input, uint8_t *dest,
int stride, const TxfmParam *param);
#ifdef __cplusplus
} // extern "C"
#endif
#endif // AV1_COMMON_IDCT_H_
#endif // AOM_AV1_COMMON_IDCT_H_

View file

@ -9,8 +9,8 @@
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AV1_COMMON_MV_H_
#define AV1_COMMON_MV_H_
#ifndef AOM_AV1_COMMON_MV_H_
#define AOM_AV1_COMMON_MV_H_
#include "av1/common/common.h"
#include "av1/common/common_data.h"
@ -56,7 +56,7 @@ typedef struct mv32 {
#define WARPEDDIFF_PREC_BITS (WARPEDMODEL_PREC_BITS - WARPEDPIXEL_PREC_BITS)
/* clang-format off */
typedef enum {
typedef enum ATTRIBUTE_PACKED {
IDENTITY = 0, // identity transformation, 0-parameter
TRANSLATION = 1, // translational motion 2-parameter
ROTZOOM = 2, // simplified affine with rotation + zoom only, 4-parameter
@ -298,4 +298,4 @@ static INLINE void clamp_mv(MV *mv, int min_col, int max_col, int min_row,
} // extern "C"
#endif
#endif // AV1_COMMON_MV_H_
#endif // AOM_AV1_COMMON_MV_H_

View file

@ -27,16 +27,19 @@ static void get_mv_projection(MV *output, MV ref, int num, int den) {
den = AOMMIN(den, MAX_FRAME_DISTANCE);
num = num > 0 ? AOMMIN(num, MAX_FRAME_DISTANCE)
: AOMMAX(num, -MAX_FRAME_DISTANCE);
int mv_row = ROUND_POWER_OF_TWO_SIGNED(ref.row * num * div_mult[den], 14);
int mv_col = ROUND_POWER_OF_TWO_SIGNED(ref.col * num * div_mult[den], 14);
const int mv_row =
ROUND_POWER_OF_TWO_SIGNED(ref.row * num * div_mult[den], 14);
const int mv_col =
ROUND_POWER_OF_TWO_SIGNED(ref.col * num * div_mult[den], 14);
const int clamp_max = MV_UPP - 1;
const int clamp_min = MV_LOW + 1;
output->row = (int16_t)clamp(mv_row, clamp_min, clamp_max);
output->col = (int16_t)clamp(mv_col, clamp_min, clamp_max);
}
void av1_copy_frame_mvs(const AV1_COMMON *const cm, MB_MODE_INFO *mi,
int mi_row, int mi_col, int x_mis, int y_mis) {
void av1_copy_frame_mvs(const AV1_COMMON *const cm,
const MB_MODE_INFO *const mi, int mi_row, int mi_col,
int x_mis, int y_mis) {
const int frame_mvs_stride = ROUND_POWER_OF_TWO(cm->mi_cols, 1);
MV_REF *frame_mvs =
cm->cur_frame->mvs + (mi_row >> 1) * frame_mvs_stride + (mi_col >> 1);
@ -141,38 +144,37 @@ static void scan_row_mbmi(const AV1_COMMON *cm, const MACROBLOCKD *xd,
uint8_t *ref_match_count, uint8_t *newmv_count,
int_mv *gm_mv_candidates, int max_row_offset,
int *processed_rows) {
int end_mi = AOMMIN(xd->n8_w, cm->mi_cols - mi_col);
int end_mi = AOMMIN(xd->n4_w, cm->mi_cols - mi_col);
end_mi = AOMMIN(end_mi, mi_size_wide[BLOCK_64X64]);
const int n8_w_8 = mi_size_wide[BLOCK_8X8];
const int n8_w_16 = mi_size_wide[BLOCK_16X16];
int i;
int col_offset = 0;
const int shift = 0;
// TODO(jingning): Revisit this part after cb4x4 is stable.
if (abs(row_offset) > 1) {
col_offset = 1;
if ((mi_col & 0x01) && xd->n8_w < n8_w_8) --col_offset;
if ((mi_col & 0x01) && xd->n4_w < n8_w_8) --col_offset;
}
const int use_step_16 = (xd->n8_w >= 16);
const int use_step_16 = (xd->n4_w >= 16);
MB_MODE_INFO **const candidate_mi0 = xd->mi + row_offset * xd->mi_stride;
(void)mi_row;
for (i = 0; i < end_mi;) {
const MB_MODE_INFO *const candidate = candidate_mi0[col_offset + i];
const int candidate_bsize = candidate->sb_type;
const int n8_w = mi_size_wide[candidate_bsize];
int len = AOMMIN(xd->n8_w, n8_w);
const int n4_w = mi_size_wide[candidate_bsize];
int len = AOMMIN(xd->n4_w, n4_w);
if (use_step_16)
len = AOMMAX(n8_w_16, len);
else if (abs(row_offset) > 1)
len = AOMMAX(len, n8_w_8);
int weight = 2;
if (xd->n8_w >= n8_w_8 && xd->n8_w <= n8_w) {
if (xd->n4_w >= n8_w_8 && xd->n4_w <= n4_w) {
int inc = AOMMIN(-max_row_offset + row_offset + 1,
mi_size_high[candidate_bsize]);
// Obtain range used in weight calculation.
weight = AOMMAX(weight, (inc << shift));
weight = AOMMAX(weight, inc);
// Update processed rows.
*processed_rows = inc - row_offset - 1;
}
@ -192,37 +194,36 @@ static void scan_col_mbmi(const AV1_COMMON *cm, const MACROBLOCKD *xd,
uint8_t *ref_match_count, uint8_t *newmv_count,
int_mv *gm_mv_candidates, int max_col_offset,
int *processed_cols) {
int end_mi = AOMMIN(xd->n8_h, cm->mi_rows - mi_row);
int end_mi = AOMMIN(xd->n4_h, cm->mi_rows - mi_row);
end_mi = AOMMIN(end_mi, mi_size_high[BLOCK_64X64]);
const int n8_h_8 = mi_size_high[BLOCK_8X8];
const int n8_h_16 = mi_size_high[BLOCK_16X16];
int i;
int row_offset = 0;
const int shift = 0;
if (abs(col_offset) > 1) {
row_offset = 1;
if ((mi_row & 0x01) && xd->n8_h < n8_h_8) --row_offset;
if ((mi_row & 0x01) && xd->n4_h < n8_h_8) --row_offset;
}
const int use_step_16 = (xd->n8_h >= 16);
const int use_step_16 = (xd->n4_h >= 16);
(void)mi_col;
for (i = 0; i < end_mi;) {
const MB_MODE_INFO *const candidate =
xd->mi[(row_offset + i) * xd->mi_stride + col_offset];
const int candidate_bsize = candidate->sb_type;
const int n8_h = mi_size_high[candidate_bsize];
int len = AOMMIN(xd->n8_h, n8_h);
const int n4_h = mi_size_high[candidate_bsize];
int len = AOMMIN(xd->n4_h, n4_h);
if (use_step_16)
len = AOMMAX(n8_h_16, len);
else if (abs(col_offset) > 1)
len = AOMMAX(len, n8_h_8);
int weight = 2;
if (xd->n8_h >= n8_h_8 && xd->n8_h <= n8_h) {
if (xd->n4_h >= n8_h_8 && xd->n4_h <= n4_h) {
int inc = AOMMIN(-max_col_offset + col_offset + 1,
mi_size_wide[candidate_bsize]);
// Obtain range used in weight calculation.
weight = AOMMAX(weight, (inc << shift));
weight = AOMMAX(weight, inc);
// Update processed cols.
*processed_cols = inc - col_offset - 1;
}
@ -248,7 +249,7 @@ static void scan_blk_mbmi(const AV1_COMMON *cm, const MACROBLOCKD *xd,
mi_pos.row = row_offset;
mi_pos.col = col_offset;
if (is_inside(tile, mi_col, mi_row, cm->mi_rows, &mi_pos)) {
if (is_inside(tile, mi_col, mi_row, &mi_pos)) {
const MB_MODE_INFO *const candidate =
xd->mi[mi_pos.row * xd->mi_stride + mi_pos.col];
const int len = mi_size_wide[BLOCK_8X8];
@ -290,19 +291,19 @@ static int has_top_right(const AV1_COMMON *cm, const MACROBLOCKD *xd,
// The left hand of two vertical rectangles always has a top right (as the
// block above will have been decoded)
if (xd->n8_w < xd->n8_h)
if (xd->n4_w < xd->n4_h)
if (!xd->is_sec_rect) has_tr = 1;
// The bottom of two horizontal rectangles never has a top right (as the block
// to the right won't have been decoded)
if (xd->n8_w > xd->n8_h)
if (xd->n4_w > xd->n4_h)
if (xd->is_sec_rect) has_tr = 0;
// The bottom left square of a Vertical A (in the old format) does
// not have a top right as it is decoded before the right hand
// rectangle of the partition
if (xd->mi[0]->partition == PARTITION_VERT_A) {
if (xd->n8_w == xd->n8_h)
if (xd->n4_w == xd->n4_h)
if (mask_row & bs) has_tr = 0;
}
@ -335,7 +336,7 @@ static int add_tpl_ref_mv(const AV1_COMMON *cm, const MACROBLOCKD *xd,
mi_pos.row = (mi_row & 0x01) ? blk_row : blk_row + 1;
mi_pos.col = (mi_col & 0x01) ? blk_col : blk_col + 1;
if (!is_inside(&xd->tile, mi_col, mi_row, cm->mi_rows, &mi_pos)) return 0;
if (!is_inside(&xd->tile, mi_col, mi_row, &mi_pos)) return 0;
const TPL_MV_REF *prev_frame_mvs =
cm->tpl_mvs + ((mi_row + mi_pos.row) >> 1) * (cm->mi_stride >> 1) +
@ -430,20 +431,75 @@ static int add_tpl_ref_mv(const AV1_COMMON *cm, const MACROBLOCKD *xd,
return 0;
}
static void process_compound_ref_mv_candidate(
const MB_MODE_INFO *const candidate, const AV1_COMMON *const cm,
const MV_REFERENCE_FRAME *const rf, int_mv ref_id[2][2],
int ref_id_count[2], int_mv ref_diff[2][2], int ref_diff_count[2]) {
for (int rf_idx = 0; rf_idx < 2; ++rf_idx) {
MV_REFERENCE_FRAME can_rf = candidate->ref_frame[rf_idx];
for (int cmp_idx = 0; cmp_idx < 2; ++cmp_idx) {
if (can_rf == rf[cmp_idx] && ref_id_count[cmp_idx] < 2) {
ref_id[cmp_idx][ref_id_count[cmp_idx]] = candidate->mv[rf_idx];
++ref_id_count[cmp_idx];
} else if (can_rf > INTRA_FRAME && ref_diff_count[cmp_idx] < 2) {
int_mv this_mv = candidate->mv[rf_idx];
if (cm->ref_frame_sign_bias[can_rf] !=
cm->ref_frame_sign_bias[rf[cmp_idx]]) {
this_mv.as_mv.row = -this_mv.as_mv.row;
this_mv.as_mv.col = -this_mv.as_mv.col;
}
ref_diff[cmp_idx][ref_diff_count[cmp_idx]] = this_mv;
++ref_diff_count[cmp_idx];
}
}
}
}
static void process_single_ref_mv_candidate(
const MB_MODE_INFO *const candidate, const AV1_COMMON *const cm,
MV_REFERENCE_FRAME ref_frame, uint8_t refmv_count[MODE_CTX_REF_FRAMES],
CANDIDATE_MV ref_mv_stack[][MAX_REF_MV_STACK_SIZE]) {
for (int rf_idx = 0; rf_idx < 2; ++rf_idx) {
if (candidate->ref_frame[rf_idx] > INTRA_FRAME) {
int_mv this_mv = candidate->mv[rf_idx];
if (cm->ref_frame_sign_bias[candidate->ref_frame[rf_idx]] !=
cm->ref_frame_sign_bias[ref_frame]) {
this_mv.as_mv.row = -this_mv.as_mv.row;
this_mv.as_mv.col = -this_mv.as_mv.col;
}
int stack_idx;
for (stack_idx = 0; stack_idx < refmv_count[ref_frame]; ++stack_idx) {
const int_mv stack_mv = ref_mv_stack[ref_frame][stack_idx].this_mv;
if (this_mv.as_int == stack_mv.as_int) break;
}
if (stack_idx == refmv_count[ref_frame]) {
ref_mv_stack[ref_frame][stack_idx].this_mv = this_mv;
// TODO(jingning): Set an arbitrary small number here. The weight
// doesn't matter as long as it is properly initialized.
ref_mv_stack[ref_frame][stack_idx].weight = 2;
++refmv_count[ref_frame];
}
}
}
}
static void setup_ref_mv_list(
const AV1_COMMON *cm, const MACROBLOCKD *xd, MV_REFERENCE_FRAME ref_frame,
uint8_t refmv_count[MODE_CTX_REF_FRAMES],
CANDIDATE_MV ref_mv_stack[][MAX_REF_MV_STACK_SIZE],
int_mv mv_ref_list[][MAX_MV_REF_CANDIDATES], int_mv *gm_mv_candidates,
int mi_row, int mi_col, int16_t *mode_context) {
const int bs = AOMMAX(xd->n8_w, xd->n8_h);
const int bs = AOMMAX(xd->n4_w, xd->n4_h);
const int has_tr = has_top_right(cm, xd, mi_row, mi_col, bs);
MV_REFERENCE_FRAME rf[2];
const TileInfo *const tile = &xd->tile;
int max_row_offset = 0, max_col_offset = 0;
const int row_adj = (xd->n8_h < mi_size_high[BLOCK_8X8]) && (mi_row & 0x01);
const int col_adj = (xd->n8_w < mi_size_wide[BLOCK_8X8]) && (mi_col & 0x01);
const int row_adj = (xd->n4_h < mi_size_high[BLOCK_8X8]) && (mi_row & 0x01);
const int col_adj = (xd->n4_w < mi_size_wide[BLOCK_8X8]) && (mi_col & 0x01);
int processed_rows = 0;
int processed_cols = 0;
@ -455,17 +511,16 @@ static void setup_ref_mv_list(
if (xd->up_available) {
max_row_offset = -(MVREF_ROW_COLS << 1) + row_adj;
if (xd->n8_h < mi_size_high[BLOCK_8X8])
if (xd->n4_h < mi_size_high[BLOCK_8X8])
max_row_offset = -(2 << 1) + row_adj;
max_row_offset =
find_valid_row_offset(tile, mi_row, cm->mi_rows, max_row_offset);
max_row_offset = find_valid_row_offset(tile, mi_row, max_row_offset);
}
if (xd->left_available) {
max_col_offset = -(MVREF_ROW_COLS << 1) + col_adj;
if (xd->n8_w < mi_size_wide[BLOCK_8X8])
if (xd->n4_w < mi_size_wide[BLOCK_8X8])
max_col_offset = -(2 << 1) + col_adj;
max_col_offset = find_valid_col_offset(tile, mi_col, max_col_offset);
@ -487,12 +542,12 @@ static void setup_ref_mv_list(
gm_mv_candidates, max_col_offset, &processed_cols);
// Check top-right boundary
if (has_tr)
scan_blk_mbmi(cm, xd, mi_row, mi_col, rf, -1, xd->n8_w,
scan_blk_mbmi(cm, xd, mi_row, mi_col, rf, -1, xd->n4_w,
ref_mv_stack[ref_frame], &row_match_count, &newmv_count,
gm_mv_candidates, &refmv_count[ref_frame]);
uint8_t nearest_match = (row_match_count > 0) + (col_match_count > 0);
uint8_t nearest_refmv_count = refmv_count[ref_frame];
const uint8_t nearest_match = (row_match_count > 0) + (col_match_count > 0);
const uint8_t nearest_refmv_count = refmv_count[ref_frame];
// TODO(yunqing): for comp_search, do it for all 3 cases.
for (int idx = 0; idx < nearest_refmv_count; ++idx)
@ -500,27 +555,27 @@ static void setup_ref_mv_list(
if (cm->allow_ref_frame_mvs) {
int is_available = 0;
const int voffset = AOMMAX(mi_size_high[BLOCK_8X8], xd->n8_h);
const int hoffset = AOMMAX(mi_size_wide[BLOCK_8X8], xd->n8_w);
const int blk_row_end = AOMMIN(xd->n8_h, mi_size_high[BLOCK_64X64]);
const int blk_col_end = AOMMIN(xd->n8_w, mi_size_wide[BLOCK_64X64]);
const int voffset = AOMMAX(mi_size_high[BLOCK_8X8], xd->n4_h);
const int hoffset = AOMMAX(mi_size_wide[BLOCK_8X8], xd->n4_w);
const int blk_row_end = AOMMIN(xd->n4_h, mi_size_high[BLOCK_64X64]);
const int blk_col_end = AOMMIN(xd->n4_w, mi_size_wide[BLOCK_64X64]);
const int tpl_sample_pos[3][2] = {
{ voffset, -2 },
{ voffset, hoffset },
{ voffset - 2, hoffset },
};
const int allow_extension = (xd->n8_h >= mi_size_high[BLOCK_8X8]) &&
(xd->n8_h < mi_size_high[BLOCK_64X64]) &&
(xd->n8_w >= mi_size_wide[BLOCK_8X8]) &&
(xd->n8_w < mi_size_wide[BLOCK_64X64]);
const int allow_extension = (xd->n4_h >= mi_size_high[BLOCK_8X8]) &&
(xd->n4_h < mi_size_high[BLOCK_64X64]) &&
(xd->n4_w >= mi_size_wide[BLOCK_8X8]) &&
(xd->n4_w < mi_size_wide[BLOCK_64X64]);
int step_h = (xd->n8_h >= mi_size_high[BLOCK_64X64])
? mi_size_high[BLOCK_16X16]
: mi_size_high[BLOCK_8X8];
int step_w = (xd->n8_w >= mi_size_wide[BLOCK_64X64])
? mi_size_wide[BLOCK_16X16]
: mi_size_wide[BLOCK_8X8];
const int step_h = (xd->n4_h >= mi_size_high[BLOCK_64X64])
? mi_size_high[BLOCK_16X16]
: mi_size_high[BLOCK_8X8];
const int step_w = (xd->n4_w >= mi_size_wide[BLOCK_64X64])
? mi_size_wide[BLOCK_16X16]
: mi_size_wide[BLOCK_8X8];
for (int blk_row = 0; blk_row < blk_row_end; blk_row += step_h) {
for (int blk_col = 0; blk_col < blk_col_end; blk_col += step_w) {
@ -569,7 +624,7 @@ static void setup_ref_mv_list(
max_col_offset, &processed_cols);
}
uint8_t ref_match_count = (row_match_count > 0) + (col_match_count > 0);
const uint8_t ref_match_count = (row_match_count > 0) + (col_match_count > 0);
switch (nearest_match) {
case 0:
@ -636,62 +691,24 @@ static void setup_ref_mv_list(
int_mv ref_id[2][2], ref_diff[2][2];
int ref_id_count[2] = { 0 }, ref_diff_count[2] = { 0 };
int mi_width = AOMMIN(mi_size_wide[BLOCK_64X64], xd->n8_w);
int mi_width = AOMMIN(mi_size_wide[BLOCK_64X64], xd->n4_w);
mi_width = AOMMIN(mi_width, cm->mi_cols - mi_col);
int mi_height = AOMMIN(mi_size_high[BLOCK_64X64], xd->n8_h);
int mi_height = AOMMIN(mi_size_high[BLOCK_64X64], xd->n4_h);
mi_height = AOMMIN(mi_height, cm->mi_rows - mi_row);
int mi_size = AOMMIN(mi_width, mi_height);
for (int idx = 0; abs(max_row_offset) >= 1 && idx < mi_size;) {
const MB_MODE_INFO *const candidate = xd->mi[-xd->mi_stride + idx];
const int candidate_bsize = candidate->sb_type;
for (int rf_idx = 0; rf_idx < 2; ++rf_idx) {
MV_REFERENCE_FRAME can_rf = candidate->ref_frame[rf_idx];
for (int cmp_idx = 0; cmp_idx < 2; ++cmp_idx) {
if (can_rf == rf[cmp_idx] && ref_id_count[cmp_idx] < 2) {
ref_id[cmp_idx][ref_id_count[cmp_idx]] = candidate->mv[rf_idx];
++ref_id_count[cmp_idx];
} else if (can_rf > INTRA_FRAME && ref_diff_count[cmp_idx] < 2) {
int_mv this_mv = candidate->mv[rf_idx];
if (cm->ref_frame_sign_bias[can_rf] !=
cm->ref_frame_sign_bias[rf[cmp_idx]]) {
this_mv.as_mv.row = -this_mv.as_mv.row;
this_mv.as_mv.col = -this_mv.as_mv.col;
}
ref_diff[cmp_idx][ref_diff_count[cmp_idx]] = this_mv;
++ref_diff_count[cmp_idx];
}
}
}
idx += mi_size_wide[candidate_bsize];
process_compound_ref_mv_candidate(
candidate, cm, rf, ref_id, ref_id_count, ref_diff, ref_diff_count);
idx += mi_size_wide[candidate->sb_type];
}
for (int idx = 0; abs(max_col_offset) >= 1 && idx < mi_size;) {
const MB_MODE_INFO *const candidate = xd->mi[idx * xd->mi_stride - 1];
const int candidate_bsize = candidate->sb_type;
for (int rf_idx = 0; rf_idx < 2; ++rf_idx) {
MV_REFERENCE_FRAME can_rf = candidate->ref_frame[rf_idx];
for (int cmp_idx = 0; cmp_idx < 2; ++cmp_idx) {
if (can_rf == rf[cmp_idx] && ref_id_count[cmp_idx] < 2) {
ref_id[cmp_idx][ref_id_count[cmp_idx]] = candidate->mv[rf_idx];
++ref_id_count[cmp_idx];
} else if (can_rf > INTRA_FRAME && ref_diff_count[cmp_idx] < 2) {
int_mv this_mv = candidate->mv[rf_idx];
if (cm->ref_frame_sign_bias[can_rf] !=
cm->ref_frame_sign_bias[rf[cmp_idx]]) {
this_mv.as_mv.row = -this_mv.as_mv.row;
this_mv.as_mv.col = -this_mv.as_mv.col;
}
ref_diff[cmp_idx][ref_diff_count[cmp_idx]] = this_mv;
++ref_diff_count[cmp_idx];
}
}
}
idx += mi_size_high[candidate_bsize];
process_compound_ref_mv_candidate(
candidate, cm, rf, ref_id, ref_id_count, ref_diff, ref_diff_count);
idx += mi_size_high[candidate->sb_type];
}
// Build up the compound mv predictor
@ -743,87 +760,37 @@ static void setup_ref_mv_list(
for (int idx = 0; idx < refmv_count[ref_frame]; ++idx) {
clamp_mv_ref(&ref_mv_stack[ref_frame][idx].this_mv.as_mv,
xd->n8_w << MI_SIZE_LOG2, xd->n8_h << MI_SIZE_LOG2, xd);
xd->n4_w << MI_SIZE_LOG2, xd->n4_h << MI_SIZE_LOG2, xd);
clamp_mv_ref(&ref_mv_stack[ref_frame][idx].comp_mv.as_mv,
xd->n8_w << MI_SIZE_LOG2, xd->n8_h << MI_SIZE_LOG2, xd);
xd->n4_w << MI_SIZE_LOG2, xd->n4_h << MI_SIZE_LOG2, xd);
}
} else {
// Handle single reference frame extension
int mi_width = AOMMIN(mi_size_wide[BLOCK_64X64], xd->n8_w);
int mi_width = AOMMIN(mi_size_wide[BLOCK_64X64], xd->n4_w);
mi_width = AOMMIN(mi_width, cm->mi_cols - mi_col);
int mi_height = AOMMIN(mi_size_high[BLOCK_64X64], xd->n8_h);
int mi_height = AOMMIN(mi_size_high[BLOCK_64X64], xd->n4_h);
mi_height = AOMMIN(mi_height, cm->mi_rows - mi_row);
int mi_size = AOMMIN(mi_width, mi_height);
for (int idx = 0; abs(max_row_offset) >= 1 && idx < mi_size &&
refmv_count[ref_frame] < MAX_MV_REF_CANDIDATES;) {
const MB_MODE_INFO *const candidate = xd->mi[-xd->mi_stride + idx];
const int candidate_bsize = candidate->sb_type;
// TODO(jingning): Refactor the following code.
for (int rf_idx = 0; rf_idx < 2; ++rf_idx) {
if (candidate->ref_frame[rf_idx] > INTRA_FRAME) {
int_mv this_mv = candidate->mv[rf_idx];
if (cm->ref_frame_sign_bias[candidate->ref_frame[rf_idx]] !=
cm->ref_frame_sign_bias[ref_frame]) {
this_mv.as_mv.row = -this_mv.as_mv.row;
this_mv.as_mv.col = -this_mv.as_mv.col;
}
int stack_idx;
for (stack_idx = 0; stack_idx < refmv_count[ref_frame]; ++stack_idx) {
int_mv stack_mv = ref_mv_stack[ref_frame][stack_idx].this_mv;
if (this_mv.as_int == stack_mv.as_int) break;
}
if (stack_idx == refmv_count[ref_frame]) {
ref_mv_stack[ref_frame][stack_idx].this_mv = this_mv;
// TODO(jingning): Set an arbitrary small number here. The weight
// doesn't matter as long as it is properly initialized.
ref_mv_stack[ref_frame][stack_idx].weight = 2;
++refmv_count[ref_frame];
}
}
}
idx += mi_size_wide[candidate_bsize];
process_single_ref_mv_candidate(candidate, cm, ref_frame, refmv_count,
ref_mv_stack);
idx += mi_size_wide[candidate->sb_type];
}
for (int idx = 0; abs(max_col_offset) >= 1 && idx < mi_size &&
refmv_count[ref_frame] < MAX_MV_REF_CANDIDATES;) {
const MB_MODE_INFO *const candidate = xd->mi[idx * xd->mi_stride - 1];
const int candidate_bsize = candidate->sb_type;
// TODO(jingning): Refactor the following code.
for (int rf_idx = 0; rf_idx < 2; ++rf_idx) {
if (candidate->ref_frame[rf_idx] > INTRA_FRAME) {
int_mv this_mv = candidate->mv[rf_idx];
if (cm->ref_frame_sign_bias[candidate->ref_frame[rf_idx]] !=
cm->ref_frame_sign_bias[ref_frame]) {
this_mv.as_mv.row = -this_mv.as_mv.row;
this_mv.as_mv.col = -this_mv.as_mv.col;
}
int stack_idx;
for (stack_idx = 0; stack_idx < refmv_count[ref_frame]; ++stack_idx) {
int_mv stack_mv = ref_mv_stack[ref_frame][stack_idx].this_mv;
if (this_mv.as_int == stack_mv.as_int) break;
}
if (stack_idx == refmv_count[ref_frame]) {
ref_mv_stack[ref_frame][stack_idx].this_mv = this_mv;
// TODO(jingning): Set an arbitrary small number here. The weight
// doesn't matter as long as it is properly initialized.
ref_mv_stack[ref_frame][stack_idx].weight = 2;
++refmv_count[ref_frame];
}
}
}
idx += mi_size_high[candidate_bsize];
process_single_ref_mv_candidate(candidate, cm, ref_frame, refmv_count,
ref_mv_stack);
idx += mi_size_high[candidate->sb_type];
}
for (int idx = 0; idx < refmv_count[ref_frame]; ++idx) {
clamp_mv_ref(&ref_mv_stack[ref_frame][idx].this_mv.as_mv,
xd->n8_w << MI_SIZE_LOG2, xd->n8_h << MI_SIZE_LOG2, xd);
xd->n4_w << MI_SIZE_LOG2, xd->n4_h << MI_SIZE_LOG2, xd);
}
if (mv_ref_list != NULL) {
@ -936,8 +903,10 @@ static int get_block_position(AV1_COMMON *cm, int *mi_r, int *mi_c, int blk_row,
const int col_offset = (mv.col >= 0) ? (mv.col >> (4 + MI_SIZE_LOG2))
: -((-mv.col) >> (4 + MI_SIZE_LOG2));
int row = (sign_bias == 1) ? blk_row - row_offset : blk_row + row_offset;
int col = (sign_bias == 1) ? blk_col - col_offset : blk_col + col_offset;
const int row =
(sign_bias == 1) ? blk_row - row_offset : blk_row + row_offset;
const int col =
(sign_bias == 1) ? blk_col - col_offset : blk_col + col_offset;
if (row < 0 || row >= (cm->mi_rows >> 1) || col < 0 ||
col >= (cm->mi_cols >> 1))
@ -955,37 +924,44 @@ static int get_block_position(AV1_COMMON *cm, int *mi_r, int *mi_c, int blk_row,
return 1;
}
static int motion_field_projection(AV1_COMMON *cm, MV_REFERENCE_FRAME ref_frame,
int dir) {
// Note: motion_filed_projection finds motion vectors of current frame's
// reference frame, and projects them to current frame. To make it clear,
// let's call current frame's reference frame as start frame.
// Call Start frame's reference frames as reference frames.
// Call ref_offset as frame distances between start frame and its reference
// frames.
static int motion_field_projection(AV1_COMMON *cm,
MV_REFERENCE_FRAME start_frame, int dir) {
TPL_MV_REF *tpl_mvs_base = cm->tpl_mvs;
int ref_offset[REF_FRAMES] = { 0 };
(void)dir;
int ref_frame_idx = cm->frame_refs[FWD_RF_OFFSET(ref_frame)].idx;
if (ref_frame_idx < 0) return 0;
const int start_frame_idx = cm->frame_refs[FWD_RF_OFFSET(start_frame)].idx;
if (start_frame_idx < 0) return 0;
if (cm->buffer_pool->frame_bufs[ref_frame_idx].intra_only) return 0;
if (cm->buffer_pool->frame_bufs[start_frame_idx].intra_only) return 0;
if (cm->buffer_pool->frame_bufs[ref_frame_idx].mi_rows != cm->mi_rows ||
cm->buffer_pool->frame_bufs[ref_frame_idx].mi_cols != cm->mi_cols)
if (cm->buffer_pool->frame_bufs[start_frame_idx].mi_rows != cm->mi_rows ||
cm->buffer_pool->frame_bufs[start_frame_idx].mi_cols != cm->mi_cols)
return 0;
int ref_frame_index =
cm->buffer_pool->frame_bufs[ref_frame_idx].cur_frame_offset;
unsigned int *ref_rf_idx =
&cm->buffer_pool->frame_bufs[ref_frame_idx].ref_frame_offset[0];
int cur_frame_index = cm->cur_frame->cur_frame_offset;
int ref_to_cur = get_relative_dist(cm, ref_frame_index, cur_frame_index);
const int start_frame_offset =
cm->buffer_pool->frame_bufs[start_frame_idx].cur_frame_offset;
const unsigned int *const ref_frame_offsets =
&cm->buffer_pool->frame_bufs[start_frame_idx].ref_frame_offset[0];
const int cur_frame_offset = cm->cur_frame->cur_frame_offset;
int start_to_current_frame_offset =
get_relative_dist(cm, start_frame_offset, cur_frame_offset);
for (MV_REFERENCE_FRAME rf = LAST_FRAME; rf <= INTER_REFS_PER_FRAME; ++rf) {
ref_offset[rf] =
get_relative_dist(cm, ref_frame_index, ref_rf_idx[rf - LAST_FRAME]);
ref_offset[rf] = get_relative_dist(cm, start_frame_offset,
ref_frame_offsets[rf - LAST_FRAME]);
}
if (dir == 2) ref_to_cur = -ref_to_cur;
if (dir == 2) start_to_current_frame_offset = -start_to_current_frame_offset;
MV_REF *mv_ref_base = cm->buffer_pool->frame_bufs[ref_frame_idx].mvs;
MV_REF *mv_ref_base = cm->buffer_pool->frame_bufs[start_frame_idx].mvs;
const int mvs_rows = (cm->mi_rows + 1) >> 1;
const int mvs_cols = (cm->mi_cols + 1) >> 1;
@ -999,19 +975,20 @@ static int motion_field_projection(AV1_COMMON *cm, MV_REFERENCE_FRAME ref_frame,
int mi_r, mi_c;
const int ref_frame_offset = ref_offset[mv_ref->ref_frame];
int pos_valid = abs(ref_frame_offset) <= MAX_FRAME_DISTANCE &&
ref_frame_offset > 0 &&
abs(ref_to_cur) <= MAX_FRAME_DISTANCE;
int pos_valid =
abs(ref_frame_offset) <= MAX_FRAME_DISTANCE &&
ref_frame_offset > 0 &&
abs(start_to_current_frame_offset) <= MAX_FRAME_DISTANCE;
if (pos_valid) {
get_mv_projection(&this_mv.as_mv, fwd_mv, ref_to_cur,
ref_frame_offset);
get_mv_projection(&this_mv.as_mv, fwd_mv,
start_to_current_frame_offset, ref_frame_offset);
pos_valid = get_block_position(cm, &mi_r, &mi_c, blk_row, blk_col,
this_mv.as_mv, dir >> 1);
}
if (pos_valid) {
int mi_offset = mi_r * (cm->mi_stride >> 1) + mi_c;
const int mi_offset = mi_r * (cm->mi_stride >> 1) + mi_c;
tpl_mvs_base[mi_offset].mfmv0.as_mv.row = fwd_mv.row;
tpl_mvs_base[mi_offset].mfmv0.as_mv.col = fwd_mv.col;
@ -1167,14 +1144,14 @@ int findSamples(const AV1_COMMON *cm, MACROBLOCKD *xd, int mi_row, int mi_col,
if (up_available) {
int mi_row_offset = -1;
MB_MODE_INFO *mbmi = xd->mi[mi_row_offset * xd->mi_stride];
uint8_t n8_w = mi_size_wide[mbmi->sb_type];
uint8_t n4_w = mi_size_wide[mbmi->sb_type];
if (xd->n8_w <= n8_w) {
if (xd->n4_w <= n4_w) {
// Handle "current block width <= above block width" case.
int col_offset = -mi_col % n8_w;
int col_offset = -mi_col % n4_w;
if (col_offset < 0) do_tl = 0;
if (col_offset + n8_w > xd->n8_w) do_tr = 0;
if (col_offset + n4_w > xd->n4_w) do_tr = 0;
if (mbmi->ref_frame[0] == ref_frame && mbmi->ref_frame[1] == NONE_FRAME) {
record_samples(mbmi, pts, pts_inref, 0, -1, col_offset, 1);
@ -1185,11 +1162,11 @@ int findSamples(const AV1_COMMON *cm, MACROBLOCKD *xd, int mi_row, int mi_col,
}
} else {
// Handle "current block width > above block width" case.
for (i = 0; i < AOMMIN(xd->n8_w, cm->mi_cols - mi_col); i += mi_step) {
for (i = 0; i < AOMMIN(xd->n4_w, cm->mi_cols - mi_col); i += mi_step) {
int mi_col_offset = i;
mbmi = xd->mi[mi_col_offset + mi_row_offset * xd->mi_stride];
n8_w = mi_size_wide[mbmi->sb_type];
mi_step = AOMMIN(xd->n8_w, n8_w);
n4_w = mi_size_wide[mbmi->sb_type];
mi_step = AOMMIN(xd->n4_w, n4_w);
if (mbmi->ref_frame[0] == ref_frame &&
mbmi->ref_frame[1] == NONE_FRAME) {
@ -1209,11 +1186,11 @@ int findSamples(const AV1_COMMON *cm, MACROBLOCKD *xd, int mi_row, int mi_col,
int mi_col_offset = -1;
MB_MODE_INFO *mbmi = xd->mi[mi_col_offset];
uint8_t n8_h = mi_size_high[mbmi->sb_type];
uint8_t n4_h = mi_size_high[mbmi->sb_type];
if (xd->n8_h <= n8_h) {
if (xd->n4_h <= n4_h) {
// Handle "current block height <= above block height" case.
int row_offset = -mi_row % n8_h;
int row_offset = -mi_row % n4_h;
if (row_offset < 0) do_tl = 0;
@ -1226,11 +1203,11 @@ int findSamples(const AV1_COMMON *cm, MACROBLOCKD *xd, int mi_row, int mi_col,
}
} else {
// Handle "current block height > above block height" case.
for (i = 0; i < AOMMIN(xd->n8_h, cm->mi_rows - mi_row); i += mi_step) {
for (i = 0; i < AOMMIN(xd->n4_h, cm->mi_rows - mi_row); i += mi_step) {
int mi_row_offset = i;
mbmi = xd->mi[mi_col_offset + mi_row_offset * xd->mi_stride];
n8_h = mi_size_high[mbmi->sb_type];
mi_step = AOMMIN(xd->n8_h, n8_h);
n4_h = mi_size_high[mbmi->sb_type];
mi_step = AOMMIN(xd->n4_h, n4_h);
if (mbmi->ref_frame[0] == ref_frame &&
mbmi->ref_frame[1] == NONE_FRAME) {
@ -1264,18 +1241,18 @@ int findSamples(const AV1_COMMON *cm, MACROBLOCKD *xd, int mi_row, int mi_col,
// Top-right block
if (do_tr &&
has_top_right(cm, xd, mi_row, mi_col, AOMMAX(xd->n8_w, xd->n8_h))) {
POSITION trb_pos = { -1, xd->n8_w };
has_top_right(cm, xd, mi_row, mi_col, AOMMAX(xd->n4_w, xd->n4_h))) {
POSITION trb_pos = { -1, xd->n4_w };
if (is_inside(tile, mi_col, mi_row, cm->mi_rows, &trb_pos)) {
if (is_inside(tile, mi_col, mi_row, &trb_pos)) {
int mi_row_offset = -1;
int mi_col_offset = xd->n8_w;
int mi_col_offset = xd->n4_w;
MB_MODE_INFO *mbmi =
xd->mi[mi_col_offset + mi_row_offset * xd->mi_stride];
if (mbmi->ref_frame[0] == ref_frame && mbmi->ref_frame[1] == NONE_FRAME) {
record_samples(mbmi, pts, pts_inref, 0, -1, xd->n8_w, 1);
record_samples(mbmi, pts, pts_inref, 0, -1, xd->n4_w, 1);
np++;
if (np >= LEAST_SQUARES_SAMPLES_MAX) return LEAST_SQUARES_SAMPLES_MAX;
}
@ -1372,7 +1349,7 @@ static int compare_ref_frame_info(const void *arg_a, const void *arg_b) {
static void set_ref_frame_info(AV1_COMMON *const cm, int frame_idx,
REF_FRAME_INFO *ref_info) {
assert(frame_idx >= 0 && frame_idx <= INTER_REFS_PER_FRAME);
assert(frame_idx >= 0 && frame_idx < INTER_REFS_PER_FRAME);
const int buf_idx = ref_info->buf_idx;

View file

@ -8,8 +8,8 @@
* Media Patent License 1.0 was not distributed with this source code in the
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AV1_COMMON_MVREF_COMMON_H_
#define AV1_COMMON_MVREF_COMMON_H_
#ifndef AOM_AV1_COMMON_MVREF_COMMON_H_
#define AOM_AV1_COMMON_MVREF_COMMON_H_
#include "av1/common/onyxc_int.h"
#include "av1/common/blockd.h"
@ -85,29 +85,17 @@ static INLINE int_mv scale_mv(const MB_MODE_INFO *mbmi, int ref,
// Checks that the given mi_row, mi_col and search point
// are inside the borders of the tile.
static INLINE int is_inside(const TileInfo *const tile, int mi_col, int mi_row,
int mi_rows, const POSITION *mi_pos) {
const int dependent_horz_tile_flag = 0;
if (dependent_horz_tile_flag && !tile->tg_horz_boundary) {
return !(mi_row + mi_pos->row < 0 ||
mi_col + mi_pos->col < tile->mi_col_start ||
mi_row + mi_pos->row >= mi_rows ||
mi_col + mi_pos->col >= tile->mi_col_end);
} else {
return !(mi_row + mi_pos->row < tile->mi_row_start ||
mi_col + mi_pos->col < tile->mi_col_start ||
mi_row + mi_pos->row >= tile->mi_row_end ||
mi_col + mi_pos->col >= tile->mi_col_end);
}
const POSITION *mi_pos) {
return !(mi_row + mi_pos->row < tile->mi_row_start ||
mi_col + mi_pos->col < tile->mi_col_start ||
mi_row + mi_pos->row >= tile->mi_row_end ||
mi_col + mi_pos->col >= tile->mi_col_end);
}
static INLINE int find_valid_row_offset(const TileInfo *const tile, int mi_row,
int mi_rows, int row_offset) {
const int dependent_horz_tile_flag = 0;
if (dependent_horz_tile_flag && !tile->tg_horz_boundary)
return clamp(row_offset, -mi_row, mi_rows - mi_row - 1);
else
return clamp(row_offset, tile->mi_row_start - mi_row,
tile->mi_row_end - mi_row - 1);
int row_offset) {
return clamp(row_offset, tile->mi_row_start - mi_row,
tile->mi_row_end - mi_row - 1);
}
static INLINE int find_valid_col_offset(const TileInfo *const tile, int mi_col,
@ -263,8 +251,9 @@ static INLINE void av1_collect_neighbors_ref_counts(MACROBLOCKD *const xd) {
}
}
void av1_copy_frame_mvs(const AV1_COMMON *const cm, MB_MODE_INFO *mi,
int mi_row, int mi_col, int x_mis, int y_mis);
void av1_copy_frame_mvs(const AV1_COMMON *const cm,
const MB_MODE_INFO *const mi, int mi_row, int mi_col,
int x_mis, int y_mis);
void av1_find_mv_refs(const AV1_COMMON *cm, const MACROBLOCKD *xd,
MB_MODE_INFO *mi, MV_REFERENCE_FRAME ref_frame,
@ -286,7 +275,6 @@ int findSamples(const AV1_COMMON *cm, MACROBLOCKD *xd, int mi_row, int mi_col,
#define INTRABC_DELAY_PIXELS 256 // Delay of 256 pixels
#define INTRABC_DELAY_SB64 (INTRABC_DELAY_PIXELS / 64)
#define USE_WAVE_FRONT 1 // Use only top left area of frame for reference.
static INLINE void av1_find_ref_dv(int_mv *ref_dv, const TileInfo *const tile,
int mib_size, int mi_row, int mi_col) {
@ -356,13 +344,12 @@ static INLINE int av1_is_dv_valid(const MV dv, const AV1_COMMON *cm,
const int src_sb64 = src_sb_row * total_sb64_per_row + src_sb64_col;
if (src_sb64 >= active_sb64 - INTRABC_DELAY_SB64) return 0;
#if USE_WAVE_FRONT
// Wavefront constraint: use only top left area of frame for reference.
const int gradient = 1 + INTRABC_DELAY_SB64 + (sb_size > 64);
const int wf_offset = gradient * (active_sb_row - src_sb_row);
if (src_sb_row > active_sb_row ||
src_sb64_col >= active_sb64_col - INTRABC_DELAY_SB64 + wf_offset)
return 0;
#endif
return 1;
}
@ -371,4 +358,4 @@ static INLINE int av1_is_dv_valid(const MV dv, const AV1_COMMON *cm,
} // extern "C"
#endif
#endif // AV1_COMMON_MVREF_COMMON_H_
#endif // AOM_AV1_COMMON_MVREF_COMMON_H_

View file

@ -9,8 +9,8 @@
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AV1_COMMON_OBMC_H_
#define AV1_COMMON_OBMC_H_
#ifndef AOM_AV1_COMMON_OBMC_H_
#define AOM_AV1_COMMON_OBMC_H_
typedef void (*overlappable_nb_visitor_t)(MACROBLOCKD *xd, int rel_mi_pos,
uint8_t nb_mi_size,
@ -30,7 +30,7 @@ static INLINE void foreach_overlappable_nb_above(const AV1_COMMON *cm,
// prev_row_mi points into the mi array, starting at the beginning of the
// previous row.
MB_MODE_INFO **prev_row_mi = xd->mi - mi_col - 1 * xd->mi_stride;
const int end_col = AOMMIN(mi_col + xd->n8_w, cm->mi_cols);
const int end_col = AOMMIN(mi_col + xd->n4_w, cm->mi_cols);
uint8_t mi_step;
for (int above_mi_col = mi_col; above_mi_col < end_col && nb_count < nb_max;
above_mi_col += mi_step) {
@ -49,7 +49,7 @@ static INLINE void foreach_overlappable_nb_above(const AV1_COMMON *cm,
}
if (is_neighbor_overlappable(*above_mi)) {
++nb_count;
fun(xd, above_mi_col - mi_col, AOMMIN(xd->n8_w, mi_step), *above_mi,
fun(xd, above_mi_col - mi_col, AOMMIN(xd->n4_w, mi_step), *above_mi,
fun_ctxt, num_planes);
}
}
@ -68,7 +68,7 @@ static INLINE void foreach_overlappable_nb_left(const AV1_COMMON *cm,
// prev_col_mi points into the mi array, starting at the top of the
// previous column
MB_MODE_INFO **prev_col_mi = xd->mi - 1 - mi_row * xd->mi_stride;
const int end_row = AOMMIN(mi_row + xd->n8_h, cm->mi_rows);
const int end_row = AOMMIN(mi_row + xd->n4_h, cm->mi_rows);
uint8_t mi_step;
for (int left_mi_row = mi_row; left_mi_row < end_row && nb_count < nb_max;
left_mi_row += mi_step) {
@ -82,10 +82,10 @@ static INLINE void foreach_overlappable_nb_left(const AV1_COMMON *cm,
}
if (is_neighbor_overlappable(*left_mi)) {
++nb_count;
fun(xd, left_mi_row - mi_row, AOMMIN(xd->n8_h, mi_step), *left_mi,
fun(xd, left_mi_row - mi_row, AOMMIN(xd->n4_h, mi_step), *left_mi,
fun_ctxt, num_planes);
}
}
}
#endif // AV1_COMMON_OBMC_H_
#endif // AOM_AV1_COMMON_OBMC_H_

147
third_party/aom/av1/common/obu_util.c vendored Normal file
View file

@ -0,0 +1,147 @@
/*
* Copyright (c) 2018, Alliance for Open Media. All rights reserved
*
* This source code is subject to the terms of the BSD 2 Clause License and
* the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
* was not distributed with this source code in the LICENSE file, you can
* obtain it at www.aomedia.org/license/software. If the Alliance for Open
* Media Patent License 1.0 was not distributed with this source code in the
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#include "av1/common/obu_util.h"
#include "aom_dsp/bitreader_buffer.h"
// Returns 1 when OBU type is valid, and 0 otherwise.
static int valid_obu_type(int obu_type) {
int valid_type = 0;
switch (obu_type) {
case OBU_SEQUENCE_HEADER:
case OBU_TEMPORAL_DELIMITER:
case OBU_FRAME_HEADER:
case OBU_TILE_GROUP:
case OBU_METADATA:
case OBU_FRAME:
case OBU_REDUNDANT_FRAME_HEADER:
case OBU_TILE_LIST:
case OBU_PADDING: valid_type = 1; break;
default: break;
}
return valid_type;
}
static aom_codec_err_t read_obu_size(const uint8_t *data,
size_t bytes_available,
size_t *const obu_size,
size_t *const length_field_size) {
uint64_t u_obu_size = 0;
if (aom_uleb_decode(data, bytes_available, &u_obu_size, length_field_size) !=
0) {
return AOM_CODEC_CORRUPT_FRAME;
}
if (u_obu_size > UINT32_MAX) return AOM_CODEC_CORRUPT_FRAME;
*obu_size = (size_t)u_obu_size;
return AOM_CODEC_OK;
}
// Parses OBU header and stores values in 'header'.
static aom_codec_err_t read_obu_header(struct aom_read_bit_buffer *rb,
int is_annexb, ObuHeader *header) {
if (!rb || !header) return AOM_CODEC_INVALID_PARAM;
const ptrdiff_t bit_buffer_byte_length = rb->bit_buffer_end - rb->bit_buffer;
if (bit_buffer_byte_length < 1) return AOM_CODEC_CORRUPT_FRAME;
header->size = 1;
if (aom_rb_read_bit(rb) != 0) {
// Forbidden bit. Must not be set.
return AOM_CODEC_CORRUPT_FRAME;
}
header->type = (OBU_TYPE)aom_rb_read_literal(rb, 4);
if (!valid_obu_type(header->type)) return AOM_CODEC_CORRUPT_FRAME;
header->has_extension = aom_rb_read_bit(rb);
header->has_size_field = aom_rb_read_bit(rb);
if (!header->has_size_field && !is_annexb) {
// section 5 obu streams must have obu_size field set.
return AOM_CODEC_UNSUP_BITSTREAM;
}
if (aom_rb_read_bit(rb) != 0) {
// obu_reserved_1bit must be set to 0.
return AOM_CODEC_CORRUPT_FRAME;
}
if (header->has_extension) {
if (bit_buffer_byte_length == 1) return AOM_CODEC_CORRUPT_FRAME;
header->size += 1;
header->temporal_layer_id = aom_rb_read_literal(rb, 3);
header->spatial_layer_id = aom_rb_read_literal(rb, 2);
if (aom_rb_read_literal(rb, 3) != 0) {
// extension_header_reserved_3bits must be set to 0.
return AOM_CODEC_CORRUPT_FRAME;
}
}
return AOM_CODEC_OK;
}
aom_codec_err_t aom_read_obu_header(uint8_t *buffer, size_t buffer_length,
size_t *consumed, ObuHeader *header,
int is_annexb) {
if (buffer_length < 1 || !consumed || !header) return AOM_CODEC_INVALID_PARAM;
// TODO(tomfinegan): Set the error handler here and throughout this file, and
// confirm parsing work done via aom_read_bit_buffer is successful.
struct aom_read_bit_buffer rb = { buffer, buffer + buffer_length, 0, NULL,
NULL };
aom_codec_err_t parse_result = read_obu_header(&rb, is_annexb, header);
if (parse_result == AOM_CODEC_OK) *consumed = header->size;
return parse_result;
}
aom_codec_err_t aom_read_obu_header_and_size(const uint8_t *data,
size_t bytes_available,
int is_annexb,
ObuHeader *obu_header,
size_t *const payload_size,
size_t *const bytes_read) {
size_t length_field_size = 0, obu_size = 0;
aom_codec_err_t status;
if (is_annexb) {
// Size field comes before the OBU header, and includes the OBU header
status =
read_obu_size(data, bytes_available, &obu_size, &length_field_size);
if (status != AOM_CODEC_OK) return status;
}
struct aom_read_bit_buffer rb = { data + length_field_size,
data + bytes_available, 0, NULL, NULL };
status = read_obu_header(&rb, is_annexb, obu_header);
if (status != AOM_CODEC_OK) return status;
if (is_annexb) {
// Derive the payload size from the data we've already read
if (obu_size < obu_header->size) return AOM_CODEC_CORRUPT_FRAME;
*payload_size = obu_size - obu_header->size;
} else {
// Size field comes after the OBU header, and is just the payload size
status = read_obu_size(data + obu_header->size,
bytes_available - obu_header->size, payload_size,
&length_field_size);
if (status != AOM_CODEC_OK) return status;
}
*bytes_read = length_field_size + obu_header->size;
return AOM_CODEC_OK;
}

47
third_party/aom/av1/common/obu_util.h vendored Normal file
View file

@ -0,0 +1,47 @@
/*
* Copyright (c) 2018, Alliance for Open Media. All rights reserved
*
* This source code is subject to the terms of the BSD 2 Clause License and
* the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
* was not distributed with this source code in the LICENSE file, you can
* obtain it at www.aomedia.org/license/software. If the Alliance for Open
* Media Patent License 1.0 was not distributed with this source code in the
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AOM_AV1_COMMON_OBU_UTIL_H_
#define AOM_AV1_COMMON_OBU_UTIL_H_
#include "aom/aom_codec.h"
#ifdef __cplusplus
extern "C" {
#endif
typedef struct {
size_t size; // Size (1 or 2 bytes) of the OBU header (including the
// optional OBU extension header) in the bitstream.
OBU_TYPE type;
int has_size_field;
int has_extension;
// The following fields come from the OBU extension header and therefore are
// only used if has_extension is true.
int temporal_layer_id;
int spatial_layer_id;
} ObuHeader;
aom_codec_err_t aom_read_obu_header(uint8_t *buffer, size_t buffer_length,
size_t *consumed, ObuHeader *header,
int is_annexb);
aom_codec_err_t aom_read_obu_header_and_size(const uint8_t *data,
size_t bytes_available,
int is_annexb,
ObuHeader *obu_header,
size_t *const payload_size,
size_t *const bytes_read);
#ifdef __cplusplus
} // extern "C"
#endif
#endif // AOM_AV1_COMMON_OBU_UTIL_H_

View file

@ -11,8 +11,8 @@
/* clang-format off */
#ifndef AV1_COMMON_ODINTRIN_H_
#define AV1_COMMON_ODINTRIN_H_
#ifndef AOM_AV1_COMMON_ODINTRIN_H_
#define AOM_AV1_COMMON_ODINTRIN_H_
#include <stdlib.h>
#include <string.h>
@ -46,9 +46,9 @@ extern uint32_t OD_DIVU_SMALL_CONSTS[OD_DIVU_DMAX][2];
#define OD_MAXI AOMMAX
#define OD_CLAMPI(min, val, max) (OD_MAXI(min, OD_MINI(val, max)))
#define OD_CLZ0 (1)
#define OD_CLZ(x) (-get_msb(x))
#define OD_ILOG_NZ(x) (OD_CLZ0 - OD_CLZ(x))
/*Integer logarithm (base 2) of a nonzero unsigned 32-bit integer.
OD_ILOG_NZ(x) = (int)floor(log2(x)) + 1.*/
#define OD_ILOG_NZ(x) (1 + get_msb(x))
/*Enable special features for gcc and compatible compilers.*/
#if defined(__GNUC__) && defined(__GNUC_MINOR__) && defined(__GNUC_PATCHLEVEL__)
@ -93,4 +93,4 @@ extern uint32_t OD_DIVU_SMALL_CONSTS[OD_DIVU_DMAX][2];
} // extern "C"
#endif
#endif // AV1_COMMON_ODINTRIN_H_
#endif // AOM_AV1_COMMON_ODINTRIN_H_

View file

@ -9,8 +9,8 @@
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AV1_COMMON_ONYXC_INT_H_
#define AV1_COMMON_ONYXC_INT_H_
#ifndef AOM_AV1_COMMON_ONYXC_INT_H_
#define AOM_AV1_COMMON_ONYXC_INT_H_
#include "config/aom_config.h"
#include "config/av1_rtcd.h"
@ -480,6 +480,7 @@ typedef struct AV1Common {
int byte_alignment;
int skip_loop_filter;
int skip_film_grain;
// Private data associated with the frame buffer callbacks.
void *cb_priv;
@ -823,18 +824,18 @@ static INLINE void set_mi_row_col(MACROBLOCKD *xd, const TileInfo *const tile,
xd->chroma_left_mbmi = chroma_left_mi;
}
xd->n8_h = bh;
xd->n8_w = bw;
xd->n4_h = bh;
xd->n4_w = bw;
xd->is_sec_rect = 0;
if (xd->n8_w < xd->n8_h) {
if (xd->n4_w < xd->n4_h) {
// Only mark is_sec_rect as 1 for the last block.
// For PARTITION_VERT_4, it would be (0, 0, 0, 1);
// For other partitions, it would be (0, 1).
if (!((mi_col + xd->n8_w) & (xd->n8_h - 1))) xd->is_sec_rect = 1;
if (!((mi_col + xd->n4_w) & (xd->n4_h - 1))) xd->is_sec_rect = 1;
}
if (xd->n8_w > xd->n8_h)
if (mi_row & (xd->n8_w - 1)) xd->is_sec_rect = 1;
if (xd->n4_w > xd->n4_h)
if (mi_row & (xd->n4_w - 1)) xd->is_sec_rect = 1;
}
static INLINE aom_cdf_prob *get_y_mode_cdf(FRAME_CONTEXT *tile_ctx,
@ -1115,18 +1116,18 @@ static INLINE void set_txfm_ctx(TXFM_CONTEXT *txfm_ctx, uint8_t txs, int len) {
for (i = 0; i < len; ++i) txfm_ctx[i] = txs;
}
static INLINE void set_txfm_ctxs(TX_SIZE tx_size, int n8_w, int n8_h, int skip,
static INLINE void set_txfm_ctxs(TX_SIZE tx_size, int n4_w, int n4_h, int skip,
const MACROBLOCKD *xd) {
uint8_t bw = tx_size_wide[tx_size];
uint8_t bh = tx_size_high[tx_size];
if (skip) {
bw = n8_w * MI_SIZE;
bh = n8_h * MI_SIZE;
bw = n4_w * MI_SIZE;
bh = n4_h * MI_SIZE;
}
set_txfm_ctx(xd->above_txfm_context, bw, n8_w);
set_txfm_ctx(xd->left_txfm_context, bh, n8_h);
set_txfm_ctx(xd->above_txfm_context, bw, n4_w);
set_txfm_ctx(xd->left_txfm_context, bh, n4_h);
}
static INLINE void txfm_partition_update(TXFM_CONTEXT *above_ctx,
@ -1338,4 +1339,4 @@ static INLINE uint8_t major_minor_to_seq_level_idx(BitstreamLevel bl) {
} // extern "C"
#endif
#endif // AV1_COMMON_ONYXC_INT_H_
#endif // AOM_AV1_COMMON_ONYXC_INT_H_

View file

@ -24,19 +24,21 @@
#define CFL_LINE_2 128
#define CFL_LINE_3 192
typedef vector int8_t int8x16_t;
typedef vector uint8_t uint8x16_t;
typedef vector int16_t int16x8_t;
typedef vector uint16_t uint16x8_t;
typedef vector int32_t int32x4_t;
typedef vector uint32_t uint32x4_t;
typedef vector uint64_t uint64x2_t;
typedef vector signed char int8x16_t; // NOLINT(runtime/int)
typedef vector unsigned char uint8x16_t; // NOLINT(runtime/int)
typedef vector signed short int16x8_t; // NOLINT(runtime/int)
typedef vector unsigned short uint16x8_t; // NOLINT(runtime/int)
typedef vector signed int int32x4_t; // NOLINT(runtime/int)
typedef vector unsigned int uint32x4_t; // NOLINT(runtime/int)
typedef vector unsigned long long uint64x2_t; // NOLINT(runtime/int)
static INLINE void subtract_average_vsx(int16_t *pred_buf, int width,
int height, int round_offset,
static INLINE void subtract_average_vsx(const uint16_t *src_ptr, int16_t *dst,
int width, int height, int round_offset,
int num_pel_log2) {
const int16_t *end = pred_buf + height * CFL_BUF_LINE;
const int16_t *sum_buf = pred_buf;
// int16_t *dst = dst_ptr;
const int16_t *dst_end = dst + height * CFL_BUF_LINE;
const int16_t *sum_buf = (const int16_t *)src_ptr;
const int16_t *end = sum_buf + height * CFL_BUF_LINE;
const uint32x4_t div_shift = vec_splats((uint32_t)num_pel_log2);
const uint8x16_t mask_64 = { 0x08, 0x09, 0x0A, 0x0B, 0x0C, 0x0D, 0x0E, 0x0F,
0x00, 0x01, 0x02, 0x03, 0x04, 0x05, 0x06, 0x07 };
@ -71,43 +73,40 @@ static INLINE void subtract_average_vsx(int16_t *pred_buf, int width,
const int32x4_t avg = vec_sr(sum_32x4, div_shift);
const int16x8_t vec_avg = vec_pack(avg, avg);
do {
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_0, pred_buf), vec_avg), OFF_0, pred_buf);
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_0 + CFL_LINE_1, pred_buf), vec_avg),
OFF_0 + CFL_BUF_LINE_BYTES, pred_buf);
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_0 + CFL_LINE_2, pred_buf), vec_avg),
OFF_0 + CFL_LINE_2, pred_buf);
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_0 + CFL_LINE_3, pred_buf), vec_avg),
OFF_0 + CFL_LINE_3, pred_buf);
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_0, dst), vec_avg), OFF_0, dst);
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_0 + CFL_LINE_1, dst), vec_avg),
OFF_0 + CFL_BUF_LINE_BYTES, dst);
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_0 + CFL_LINE_2, dst), vec_avg),
OFF_0 + CFL_LINE_2, dst);
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_0 + CFL_LINE_3, dst), vec_avg),
OFF_0 + CFL_LINE_3, dst);
if (width >= 16) {
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_1, pred_buf), vec_avg), OFF_1,
pred_buf);
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_1 + CFL_LINE_1, pred_buf), vec_avg),
OFF_1 + CFL_LINE_1, pred_buf);
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_1 + CFL_LINE_2, pred_buf), vec_avg),
OFF_1 + CFL_LINE_2, pred_buf);
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_1 + CFL_LINE_3, pred_buf), vec_avg),
OFF_1 + CFL_LINE_3, pred_buf);
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_1, dst), vec_avg), OFF_1, dst);
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_1 + CFL_LINE_1, dst), vec_avg),
OFF_1 + CFL_LINE_1, dst);
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_1 + CFL_LINE_2, dst), vec_avg),
OFF_1 + CFL_LINE_2, dst);
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_1 + CFL_LINE_3, dst), vec_avg),
OFF_1 + CFL_LINE_3, dst);
}
if (width == 32) {
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_2, pred_buf), vec_avg), OFF_2,
pred_buf);
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_2 + CFL_LINE_1, pred_buf), vec_avg),
OFF_2 + CFL_LINE_1, pred_buf);
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_2 + CFL_LINE_2, pred_buf), vec_avg),
OFF_2 + CFL_LINE_2, pred_buf);
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_2 + CFL_LINE_3, pred_buf), vec_avg),
OFF_2 + CFL_LINE_3, pred_buf);
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_2, dst), vec_avg), OFF_2, dst);
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_2 + CFL_LINE_1, dst), vec_avg),
OFF_2 + CFL_LINE_1, dst);
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_2 + CFL_LINE_2, dst), vec_avg),
OFF_2 + CFL_LINE_2, dst);
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_2 + CFL_LINE_3, dst), vec_avg),
OFF_2 + CFL_LINE_3, dst);
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_3, pred_buf), vec_avg), OFF_3,
pred_buf);
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_3 + CFL_LINE_1, pred_buf), vec_avg),
OFF_3 + CFL_LINE_1, pred_buf);
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_3 + CFL_LINE_2, pred_buf), vec_avg),
OFF_3 + CFL_LINE_2, pred_buf);
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_3 + CFL_LINE_3, pred_buf), vec_avg),
OFF_3 + CFL_LINE_3, pred_buf);
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_3, dst), vec_avg), OFF_3, dst);
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_3 + CFL_LINE_1, dst), vec_avg),
OFF_3 + CFL_LINE_1, dst);
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_3 + CFL_LINE_2, dst), vec_avg),
OFF_3 + CFL_LINE_2, dst);
vec_vsx_st(vec_sub(vec_vsx_ld(OFF_3 + CFL_LINE_3, dst), vec_avg),
OFF_3 + CFL_LINE_3, dst);
}
} while ((pred_buf += CFL_BUF_LINE * 4) < end);
} while ((dst += CFL_BUF_LINE * 4) < dst_end);
}
// Declare wrappers for VSX sizes

View file

@ -31,8 +31,8 @@ int av1_get_pred_context_switchable_interp(const MACROBLOCKD *xd, int dir) {
const MB_MODE_INFO *const mbmi = xd->mi[0];
const int ctx_offset =
(mbmi->ref_frame[1] > INTRA_FRAME) * INTER_FILTER_COMP_OFFSET;
MV_REFERENCE_FRAME ref_frame =
(dir < 2) ? mbmi->ref_frame[0] : mbmi->ref_frame[1];
assert(dir == 0 || dir == 1);
const MV_REFERENCE_FRAME ref_frame = mbmi->ref_frame[0];
// Note:
// The mode info data structure has a one element border above and to the
// left of the entries corresponding to real macroblocks.

View file

@ -9,8 +9,8 @@
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AV1_COMMON_PRED_COMMON_H_
#define AV1_COMMON_PRED_COMMON_H_
#ifndef AOM_AV1_COMMON_PRED_COMMON_H_
#define AOM_AV1_COMMON_PRED_COMMON_H_
#include "av1/common/blockd.h"
#include "av1/common/mvref_common.h"
@ -357,4 +357,4 @@ static INLINE int get_tx_size_context(const MACROBLOCKD *xd) {
} // extern "C"
#endif
#endif // AV1_COMMON_PRED_COMMON_H_
#endif // AOM_AV1_COMMON_PRED_COMMON_H_

View file

@ -9,8 +9,8 @@
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AV1_COMMON_QUANT_COMMON_H_
#define AV1_COMMON_QUANT_COMMON_H_
#ifndef AOM_AV1_COMMON_QUANT_COMMON_H_
#define AOM_AV1_COMMON_QUANT_COMMON_H_
#include "aom/aom_codec.h"
#include "av1/common/seg_common.h"
@ -60,4 +60,4 @@ const qm_val_t *av1_qmatrix(struct AV1Common *cm, int qindex, int comp,
} // extern "C"
#endif
#endif // AV1_COMMON_QUANT_COMMON_H_
#endif // AOM_AV1_COMMON_QUANT_COMMON_H_

View file

@ -44,10 +44,9 @@ int av1_allow_warp(const MB_MODE_INFO *const mbmi,
if (build_for_obmc) return 0;
if (warp_types->local_warp_allowed && !mbmi->wm_params[0].invalid) {
if (warp_types->local_warp_allowed && !mbmi->wm_params.invalid) {
if (final_warp_params != NULL)
memcpy(final_warp_params, &mbmi->wm_params[0],
sizeof(*final_warp_params));
memcpy(final_warp_params, &mbmi->wm_params, sizeof(*final_warp_params));
return 1;
} else if (warp_types->global_warp_allowed && !gm_params->invalid) {
if (final_warp_params != NULL)
@ -78,6 +77,9 @@ void av1_make_inter_predictor(const uint8_t *src, int src_stride, uint8_t *dst,
av1_allow_warp(mi, warp_types, &xd->global_motion[mi->ref_frame[ref]],
build_for_obmc, subpel_params->xs, subpel_params->ys,
&final_warp_params));
const int is_intrabc = mi->use_intrabc;
assert(IMPLIES(is_intrabc, !do_warp));
if (do_warp && xd->cur_frame_force_integer_mv == 0) {
const struct macroblockd_plane *const pd = &xd->plane[plane];
const struct buf_2d *const pre_buf = &pd->pre[ref];
@ -88,10 +90,11 @@ void av1_make_inter_predictor(const uint8_t *src, int src_stride, uint8_t *dst,
pd->subsampling_x, pd->subsampling_y, conv_params);
} else if (xd->cur_buf->flags & YV12_FLAG_HIGHBITDEPTH) {
highbd_inter_predictor(src, src_stride, dst, dst_stride, subpel_params, sf,
w, h, conv_params, interp_filters, xd->bd);
w, h, conv_params, interp_filters, is_intrabc,
xd->bd);
} else {
inter_predictor(src, src_stride, dst, dst_stride, subpel_params, sf, w, h,
conv_params, interp_filters);
conv_params, interp_filters, is_intrabc);
}
}
@ -574,37 +577,6 @@ static void build_masked_compound_no_round(
h, subw, subh, conv_params);
}
static void build_masked_compound(
uint8_t *dst, int dst_stride, const uint8_t *src0, int src0_stride,
const uint8_t *src1, int src1_stride,
const INTERINTER_COMPOUND_DATA *const comp_data, BLOCK_SIZE sb_type, int h,
int w) {
// Derive subsampling from h and w passed in. May be refactored to
// pass in subsampling factors directly.
const int subh = (2 << mi_size_high_log2[sb_type]) == h;
const int subw = (2 << mi_size_wide_log2[sb_type]) == w;
const uint8_t *mask = av1_get_compound_type_mask(comp_data, sb_type);
aom_blend_a64_mask(dst, dst_stride, src0, src0_stride, src1, src1_stride,
mask, block_size_wide[sb_type], w, h, subw, subh);
}
static void build_masked_compound_highbd(
uint8_t *dst_8, int dst_stride, const uint8_t *src0_8, int src0_stride,
const uint8_t *src1_8, int src1_stride,
const INTERINTER_COMPOUND_DATA *const comp_data, BLOCK_SIZE sb_type, int h,
int w, int bd) {
// Derive subsampling from h and w passed in. May be refactored to
// pass in subsampling factors directly.
const int subh = (2 << mi_size_high_log2[sb_type]) == h;
const int subw = (2 << mi_size_wide_log2[sb_type]) == w;
const uint8_t *mask = av1_get_compound_type_mask(comp_data, sb_type);
// const uint8_t *mask =
// av1_get_contiguous_soft_mask(wedge_index, wedge_sign, sb_type);
aom_highbd_blend_a64_mask(dst_8, dst_stride, src0_8, src0_stride, src1_8,
src1_stride, mask, block_size_wide[sb_type], w, h,
subw, subh, bd);
}
void av1_make_masked_inter_predictor(
const uint8_t *pre, int pre_stride, uint8_t *dst, int dst_stride,
const SubpelParams *subpel_params, const struct scale_factors *sf, int w,
@ -653,63 +625,6 @@ void av1_make_masked_inter_predictor(
mi->sb_type, h, w, conv_params, xd);
}
// TODO(sarahparker) av1_highbd_build_inter_predictor and
// av1_build_inter_predictor should be combined with
// av1_make_inter_predictor
void av1_highbd_build_inter_predictor(
const uint8_t *src, int src_stride, uint8_t *dst, int dst_stride,
const MV *src_mv, const struct scale_factors *sf, int w, int h, int ref,
InterpFilters interp_filters, const WarpTypesAllowed *warp_types, int p_col,
int p_row, int plane, enum mv_precision precision, int x, int y,
const MACROBLOCKD *xd, int can_use_previous) {
const int is_q4 = precision == MV_PRECISION_Q4;
const MV mv_q4 = { is_q4 ? src_mv->row : src_mv->row * 2,
is_q4 ? src_mv->col : src_mv->col * 2 };
MV32 mv = av1_scale_mv(&mv_q4, x, y, sf);
mv.col += SCALE_EXTRA_OFF;
mv.row += SCALE_EXTRA_OFF;
const SubpelParams subpel_params = { sf->x_step_q4, sf->y_step_q4,
mv.col & SCALE_SUBPEL_MASK,
mv.row & SCALE_SUBPEL_MASK };
ConvolveParams conv_params = get_conv_params(ref, 0, plane, xd->bd);
src += (mv.row >> SCALE_SUBPEL_BITS) * src_stride +
(mv.col >> SCALE_SUBPEL_BITS);
av1_make_inter_predictor(src, src_stride, dst, dst_stride, &subpel_params, sf,
w, h, &conv_params, interp_filters, warp_types,
p_col, p_row, plane, ref, xd->mi[0], 0, xd,
can_use_previous);
}
void av1_build_inter_predictor(const uint8_t *src, int src_stride, uint8_t *dst,
int dst_stride, const MV *src_mv,
const struct scale_factors *sf, int w, int h,
ConvolveParams *conv_params,
InterpFilters interp_filters,
const WarpTypesAllowed *warp_types, int p_col,
int p_row, int plane, int ref,
enum mv_precision precision, int x, int y,
const MACROBLOCKD *xd, int can_use_previous) {
const int is_q4 = precision == MV_PRECISION_Q4;
const MV mv_q4 = { is_q4 ? src_mv->row : src_mv->row * 2,
is_q4 ? src_mv->col : src_mv->col * 2 };
MV32 mv = av1_scale_mv(&mv_q4, x, y, sf);
mv.col += SCALE_EXTRA_OFF;
mv.row += SCALE_EXTRA_OFF;
const SubpelParams subpel_params = { sf->x_step_q4, sf->y_step_q4,
mv.col & SCALE_SUBPEL_MASK,
mv.row & SCALE_SUBPEL_MASK };
src += (mv.row >> SCALE_SUBPEL_BITS) * src_stride +
(mv.col >> SCALE_SUBPEL_BITS);
av1_make_inter_predictor(src, src_stride, dst, dst_stride, &subpel_params, sf,
w, h, conv_params, interp_filters, warp_types, p_col,
p_row, plane, ref, xd->mi[0], 0, xd,
can_use_previous);
}
void av1_jnt_comp_weight_assign(const AV1_COMMON *cm, const MB_MODE_INFO *mbmi,
int order_idx, int *fwd_offset, int *bck_offset,
int *use_jnt_comp_avg, int is_compound) {
@ -759,279 +674,6 @@ void av1_jnt_comp_weight_assign(const AV1_COMMON *cm, const MB_MODE_INFO *mbmi,
*bck_offset = quant_dist_lookup_table[order_idx][i][1 - order];
}
static INLINE void calc_subpel_params(
MACROBLOCKD *xd, const struct scale_factors *const sf, const MV mv,
int plane, const int pre_x, const int pre_y, int x, int y,
struct buf_2d *const pre_buf, uint8_t **pre, SubpelParams *subpel_params,
int bw, int bh) {
struct macroblockd_plane *const pd = &xd->plane[plane];
const int is_scaled = av1_is_scaled(sf);
if (is_scaled) {
int ssx = pd->subsampling_x;
int ssy = pd->subsampling_y;
int orig_pos_y = (pre_y + y) << SUBPEL_BITS;
orig_pos_y += mv.row * (1 << (1 - ssy));
int orig_pos_x = (pre_x + x) << SUBPEL_BITS;
orig_pos_x += mv.col * (1 << (1 - ssx));
int pos_y = sf->scale_value_y(orig_pos_y, sf);
int pos_x = sf->scale_value_x(orig_pos_x, sf);
pos_x += SCALE_EXTRA_OFF;
pos_y += SCALE_EXTRA_OFF;
const int top = -AOM_LEFT_TOP_MARGIN_SCALED(ssy);
const int left = -AOM_LEFT_TOP_MARGIN_SCALED(ssx);
const int bottom = (pre_buf->height + AOM_INTERP_EXTEND)
<< SCALE_SUBPEL_BITS;
const int right = (pre_buf->width + AOM_INTERP_EXTEND) << SCALE_SUBPEL_BITS;
pos_y = clamp(pos_y, top, bottom);
pos_x = clamp(pos_x, left, right);
*pre = pre_buf->buf0 + (pos_y >> SCALE_SUBPEL_BITS) * pre_buf->stride +
(pos_x >> SCALE_SUBPEL_BITS);
subpel_params->subpel_x = pos_x & SCALE_SUBPEL_MASK;
subpel_params->subpel_y = pos_y & SCALE_SUBPEL_MASK;
subpel_params->xs = sf->x_step_q4;
subpel_params->ys = sf->y_step_q4;
} else {
const MV mv_q4 = clamp_mv_to_umv_border_sb(
xd, &mv, bw, bh, pd->subsampling_x, pd->subsampling_y);
subpel_params->xs = subpel_params->ys = SCALE_SUBPEL_SHIFTS;
subpel_params->subpel_x = (mv_q4.col & SUBPEL_MASK) << SCALE_EXTRA_BITS;
subpel_params->subpel_y = (mv_q4.row & SUBPEL_MASK) << SCALE_EXTRA_BITS;
*pre = pre_buf->buf + (y + (mv_q4.row >> SUBPEL_BITS)) * pre_buf->stride +
(x + (mv_q4.col >> SUBPEL_BITS));
}
}
static INLINE void build_inter_predictors(const AV1_COMMON *cm, MACROBLOCKD *xd,
int plane, const MB_MODE_INFO *mi,
int build_for_obmc, int bw, int bh,
int mi_x, int mi_y) {
struct macroblockd_plane *const pd = &xd->plane[plane];
int is_compound = has_second_ref(mi);
int ref;
const int is_intrabc = is_intrabc_block(mi);
assert(IMPLIES(is_intrabc, !is_compound));
int is_global[2] = { 0, 0 };
for (ref = 0; ref < 1 + is_compound; ++ref) {
const WarpedMotionParams *const wm = &xd->global_motion[mi->ref_frame[ref]];
is_global[ref] = is_global_mv_block(mi, wm->wmtype);
}
const BLOCK_SIZE bsize = mi->sb_type;
const int ss_x = pd->subsampling_x;
const int ss_y = pd->subsampling_y;
int sub8x8_inter = (block_size_wide[bsize] < 8 && ss_x) ||
(block_size_high[bsize] < 8 && ss_y);
if (is_intrabc) sub8x8_inter = 0;
// For sub8x8 chroma blocks, we may be covering more than one luma block's
// worth of pixels. Thus (mi_x, mi_y) may not be the correct coordinates for
// the top-left corner of the prediction source - the correct top-left corner
// is at (pre_x, pre_y).
const int row_start =
(block_size_high[bsize] == 4) && ss_y && !build_for_obmc ? -1 : 0;
const int col_start =
(block_size_wide[bsize] == 4) && ss_x && !build_for_obmc ? -1 : 0;
const int pre_x = (mi_x + MI_SIZE * col_start) >> ss_x;
const int pre_y = (mi_y + MI_SIZE * row_start) >> ss_y;
sub8x8_inter = sub8x8_inter && !build_for_obmc;
if (sub8x8_inter) {
for (int row = row_start; row <= 0 && sub8x8_inter; ++row) {
for (int col = col_start; col <= 0; ++col) {
const MB_MODE_INFO *this_mbmi = xd->mi[row * xd->mi_stride + col];
if (!is_inter_block(this_mbmi)) sub8x8_inter = 0;
if (is_intrabc_block(this_mbmi)) sub8x8_inter = 0;
}
}
}
if (sub8x8_inter) {
// block size
const int b4_w = block_size_wide[bsize] >> ss_x;
const int b4_h = block_size_high[bsize] >> ss_y;
const BLOCK_SIZE plane_bsize = scale_chroma_bsize(bsize, ss_x, ss_y);
const int b8_w = block_size_wide[plane_bsize] >> ss_x;
const int b8_h = block_size_high[plane_bsize] >> ss_y;
assert(!is_compound);
const struct buf_2d orig_pred_buf[2] = { pd->pre[0], pd->pre[1] };
int row = row_start;
for (int y = 0; y < b8_h; y += b4_h) {
int col = col_start;
for (int x = 0; x < b8_w; x += b4_w) {
MB_MODE_INFO *this_mbmi = xd->mi[row * xd->mi_stride + col];
is_compound = has_second_ref(this_mbmi);
DECLARE_ALIGNED(32, CONV_BUF_TYPE, tmp_dst[8 * 8]);
int tmp_dst_stride = 8;
assert(bw < 8 || bh < 8);
ConvolveParams conv_params = get_conv_params_no_round(
0, 0, plane, tmp_dst, tmp_dst_stride, is_compound, xd->bd);
conv_params.use_jnt_comp_avg = 0;
struct buf_2d *const dst_buf = &pd->dst;
uint8_t *dst = dst_buf->buf + dst_buf->stride * y + x;
ref = 0;
const RefBuffer *ref_buf =
&cm->frame_refs[this_mbmi->ref_frame[ref] - LAST_FRAME];
pd->pre[ref].buf0 =
(plane == 1) ? ref_buf->buf->u_buffer : ref_buf->buf->v_buffer;
pd->pre[ref].buf =
pd->pre[ref].buf0 + scaled_buffer_offset(pre_x, pre_y,
ref_buf->buf->uv_stride,
&ref_buf->sf);
pd->pre[ref].width = ref_buf->buf->uv_crop_width;
pd->pre[ref].height = ref_buf->buf->uv_crop_height;
pd->pre[ref].stride = ref_buf->buf->uv_stride;
const struct scale_factors *const sf =
is_intrabc ? &cm->sf_identity : &ref_buf->sf;
struct buf_2d *const pre_buf = is_intrabc ? dst_buf : &pd->pre[ref];
const MV mv = this_mbmi->mv[ref].as_mv;
uint8_t *pre;
SubpelParams subpel_params;
WarpTypesAllowed warp_types;
warp_types.global_warp_allowed = is_global[ref];
warp_types.local_warp_allowed = this_mbmi->motion_mode == WARPED_CAUSAL;
calc_subpel_params(xd, sf, mv, plane, pre_x, pre_y, x, y, pre_buf, &pre,
&subpel_params, bw, bh);
conv_params.ref = ref;
conv_params.do_average = ref;
if (is_masked_compound_type(mi->interinter_comp.type)) {
// masked compound type has its own average mechanism
conv_params.do_average = 0;
}
av1_make_inter_predictor(
pre, pre_buf->stride, dst, dst_buf->stride, &subpel_params, sf,
b4_w, b4_h, &conv_params, this_mbmi->interp_filters, &warp_types,
(mi_x >> pd->subsampling_x) + x, (mi_y >> pd->subsampling_y) + y,
plane, ref, mi, build_for_obmc, xd, cm->allow_warped_motion);
++col;
}
++row;
}
for (ref = 0; ref < 2; ++ref) pd->pre[ref] = orig_pred_buf[ref];
return;
}
{
DECLARE_ALIGNED(32, uint16_t, tmp_dst[MAX_SB_SIZE * MAX_SB_SIZE]);
ConvolveParams conv_params = get_conv_params_no_round(
0, 0, plane, tmp_dst, MAX_SB_SIZE, is_compound, xd->bd);
av1_jnt_comp_weight_assign(cm, mi, 0, &conv_params.fwd_offset,
&conv_params.bck_offset,
&conv_params.use_jnt_comp_avg, is_compound);
struct buf_2d *const dst_buf = &pd->dst;
uint8_t *const dst = dst_buf->buf;
for (ref = 0; ref < 1 + is_compound; ++ref) {
const struct scale_factors *const sf =
is_intrabc ? &cm->sf_identity : &xd->block_refs[ref]->sf;
struct buf_2d *const pre_buf = is_intrabc ? dst_buf : &pd->pre[ref];
const MV mv = mi->mv[ref].as_mv;
uint8_t *pre;
SubpelParams subpel_params;
calc_subpel_params(xd, sf, mv, plane, pre_x, pre_y, 0, 0, pre_buf, &pre,
&subpel_params, bw, bh);
WarpTypesAllowed warp_types;
warp_types.global_warp_allowed = is_global[ref];
warp_types.local_warp_allowed = mi->motion_mode == WARPED_CAUSAL;
conv_params.ref = ref;
if (ref && is_masked_compound_type(mi->interinter_comp.type)) {
// masked compound type has its own average mechanism
conv_params.do_average = 0;
av1_make_masked_inter_predictor(
pre, pre_buf->stride, dst, dst_buf->stride, &subpel_params, sf, bw,
bh, &conv_params, mi->interp_filters, plane, &warp_types,
mi_x >> pd->subsampling_x, mi_y >> pd->subsampling_y, ref, xd,
cm->allow_warped_motion);
} else {
conv_params.do_average = ref;
av1_make_inter_predictor(
pre, pre_buf->stride, dst, dst_buf->stride, &subpel_params, sf, bw,
bh, &conv_params, mi->interp_filters, &warp_types,
mi_x >> pd->subsampling_x, mi_y >> pd->subsampling_y, plane, ref,
mi, build_for_obmc, xd, cm->allow_warped_motion);
}
}
}
}
static void build_inter_predictors_for_planes(const AV1_COMMON *cm,
MACROBLOCKD *xd, BLOCK_SIZE bsize,
int mi_row, int mi_col,
int plane_from, int plane_to) {
int plane;
const int mi_x = mi_col * MI_SIZE;
const int mi_y = mi_row * MI_SIZE;
for (plane = plane_from; plane <= plane_to; ++plane) {
const struct macroblockd_plane *pd = &xd->plane[plane];
const int bw = pd->width;
const int bh = pd->height;
if (!is_chroma_reference(mi_row, mi_col, bsize, pd->subsampling_x,
pd->subsampling_y))
continue;
build_inter_predictors(cm, xd, plane, xd->mi[0], 0, bw, bh, mi_x, mi_y);
}
}
void av1_build_inter_predictors_sby(const AV1_COMMON *cm, MACROBLOCKD *xd,
int mi_row, int mi_col, BUFFER_SET *ctx,
BLOCK_SIZE bsize) {
build_inter_predictors_for_planes(cm, xd, bsize, mi_row, mi_col, 0, 0);
if (is_interintra_pred(xd->mi[0])) {
BUFFER_SET default_ctx = { { xd->plane[0].dst.buf, NULL, NULL },
{ xd->plane[0].dst.stride, 0, 0 } };
if (!ctx) ctx = &default_ctx;
av1_build_interintra_predictors_sbp(cm, xd, xd->plane[0].dst.buf,
xd->plane[0].dst.stride, ctx, 0, bsize);
}
}
void av1_build_inter_predictors_sbuv(const AV1_COMMON *cm, MACROBLOCKD *xd,
int mi_row, int mi_col, BUFFER_SET *ctx,
BLOCK_SIZE bsize) {
build_inter_predictors_for_planes(cm, xd, bsize, mi_row, mi_col, 1,
MAX_MB_PLANE - 1);
if (is_interintra_pred(xd->mi[0])) {
BUFFER_SET default_ctx = {
{ NULL, xd->plane[1].dst.buf, xd->plane[2].dst.buf },
{ 0, xd->plane[1].dst.stride, xd->plane[2].dst.stride }
};
if (!ctx) ctx = &default_ctx;
av1_build_interintra_predictors_sbuv(
cm, xd, xd->plane[1].dst.buf, xd->plane[2].dst.buf,
xd->plane[1].dst.stride, xd->plane[2].dst.stride, ctx, bsize);
}
}
void av1_build_inter_predictors_sb(const AV1_COMMON *cm, MACROBLOCKD *xd,
int mi_row, int mi_col, BUFFER_SET *ctx,
BLOCK_SIZE bsize) {
const int num_planes = av1_num_planes(cm);
av1_build_inter_predictors_sby(cm, xd, mi_row, mi_col, ctx, bsize);
if (num_planes > 1)
av1_build_inter_predictors_sbuv(cm, xd, mi_row, mi_col, ctx, bsize);
}
void av1_setup_dst_planes(struct macroblockd_plane *planes, BLOCK_SIZE bsize,
const YV12_BUFFER_CONFIG *src, int mi_row, int mi_col,
const int plane_start, const int plane_end) {
@ -1292,63 +934,7 @@ void av1_setup_build_prediction_by_above_pred(
xd->mb_to_left_edge = 8 * MI_SIZE * (-above_mi_col);
xd->mb_to_right_edge = ctxt->mb_to_far_edge +
(xd->n8_w - rel_mi_col - above_mi_width) * MI_SIZE * 8;
}
static INLINE void build_prediction_by_above_pred(
MACROBLOCKD *xd, int rel_mi_col, uint8_t above_mi_width,
MB_MODE_INFO *above_mbmi, void *fun_ctxt, const int num_planes) {
struct build_prediction_ctxt *ctxt = (struct build_prediction_ctxt *)fun_ctxt;
const int above_mi_col = ctxt->mi_col + rel_mi_col;
int mi_x, mi_y;
MB_MODE_INFO backup_mbmi = *above_mbmi;
av1_setup_build_prediction_by_above_pred(xd, rel_mi_col, above_mi_width,
above_mbmi, ctxt, num_planes);
mi_x = above_mi_col << MI_SIZE_LOG2;
mi_y = ctxt->mi_row << MI_SIZE_LOG2;
const BLOCK_SIZE bsize = xd->mi[0]->sb_type;
for (int j = 0; j < num_planes; ++j) {
const struct macroblockd_plane *pd = &xd->plane[j];
int bw = (above_mi_width * MI_SIZE) >> pd->subsampling_x;
int bh = clamp(block_size_high[bsize] >> (pd->subsampling_y + 1), 4,
block_size_high[BLOCK_64X64] >> (pd->subsampling_y + 1));
if (av1_skip_u4x4_pred_in_obmc(bsize, pd, 0)) continue;
build_inter_predictors(ctxt->cm, xd, j, above_mbmi, 1, bw, bh, mi_x, mi_y);
}
*above_mbmi = backup_mbmi;
}
void av1_build_prediction_by_above_preds(const AV1_COMMON *cm, MACROBLOCKD *xd,
int mi_row, int mi_col,
uint8_t *tmp_buf[MAX_MB_PLANE],
int tmp_width[MAX_MB_PLANE],
int tmp_height[MAX_MB_PLANE],
int tmp_stride[MAX_MB_PLANE]) {
if (!xd->up_available) return;
// Adjust mb_to_bottom_edge to have the correct value for the OBMC
// prediction block. This is half the height of the original block,
// except for 128-wide blocks, where we only use a height of 32.
int this_height = xd->n8_h * MI_SIZE;
int pred_height = AOMMIN(this_height / 2, 32);
xd->mb_to_bottom_edge += (this_height - pred_height) * 8;
struct build_prediction_ctxt ctxt = { cm, mi_row,
mi_col, tmp_buf,
tmp_width, tmp_height,
tmp_stride, xd->mb_to_right_edge };
BLOCK_SIZE bsize = xd->mi[0]->sb_type;
foreach_overlappable_nb_above(cm, xd, mi_col,
max_neighbor_obmc[mi_size_wide_log2[bsize]],
build_prediction_by_above_pred, &ctxt);
xd->mb_to_left_edge = -((mi_col * MI_SIZE) * 8);
xd->mb_to_right_edge = ctxt.mb_to_far_edge;
xd->mb_to_bottom_edge -= (this_height - pred_height) * 8;
(xd->n4_w - rel_mi_col - above_mi_width) * MI_SIZE * 8;
}
void av1_setup_build_prediction_by_left_pred(MACROBLOCKD *xd, int rel_mi_row,
@ -1386,101 +972,7 @@ void av1_setup_build_prediction_by_left_pred(MACROBLOCKD *xd, int rel_mi_row,
xd->mb_to_top_edge = 8 * MI_SIZE * (-left_mi_row);
xd->mb_to_bottom_edge =
ctxt->mb_to_far_edge +
(xd->n8_h - rel_mi_row - left_mi_height) * MI_SIZE * 8;
}
static INLINE void build_prediction_by_left_pred(
MACROBLOCKD *xd, int rel_mi_row, uint8_t left_mi_height,
MB_MODE_INFO *left_mbmi, void *fun_ctxt, const int num_planes) {
struct build_prediction_ctxt *ctxt = (struct build_prediction_ctxt *)fun_ctxt;
const int left_mi_row = ctxt->mi_row + rel_mi_row;
int mi_x, mi_y;
MB_MODE_INFO backup_mbmi = *left_mbmi;
av1_setup_build_prediction_by_left_pred(xd, rel_mi_row, left_mi_height,
left_mbmi, ctxt, num_planes);
mi_x = ctxt->mi_col << MI_SIZE_LOG2;
mi_y = left_mi_row << MI_SIZE_LOG2;
const BLOCK_SIZE bsize = xd->mi[0]->sb_type;
for (int j = 0; j < num_planes; ++j) {
const struct macroblockd_plane *pd = &xd->plane[j];
int bw = clamp(block_size_wide[bsize] >> (pd->subsampling_x + 1), 4,
block_size_wide[BLOCK_64X64] >> (pd->subsampling_x + 1));
int bh = (left_mi_height << MI_SIZE_LOG2) >> pd->subsampling_y;
if (av1_skip_u4x4_pred_in_obmc(bsize, pd, 1)) continue;
build_inter_predictors(ctxt->cm, xd, j, left_mbmi, 1, bw, bh, mi_x, mi_y);
}
*left_mbmi = backup_mbmi;
}
void av1_build_prediction_by_left_preds(const AV1_COMMON *cm, MACROBLOCKD *xd,
int mi_row, int mi_col,
uint8_t *tmp_buf[MAX_MB_PLANE],
int tmp_width[MAX_MB_PLANE],
int tmp_height[MAX_MB_PLANE],
int tmp_stride[MAX_MB_PLANE]) {
if (!xd->left_available) return;
// Adjust mb_to_right_edge to have the correct value for the OBMC
// prediction block. This is half the width of the original block,
// except for 128-wide blocks, where we only use a width of 32.
int this_width = xd->n8_w * MI_SIZE;
int pred_width = AOMMIN(this_width / 2, 32);
xd->mb_to_right_edge += (this_width - pred_width) * 8;
struct build_prediction_ctxt ctxt = { cm, mi_row,
mi_col, tmp_buf,
tmp_width, tmp_height,
tmp_stride, xd->mb_to_bottom_edge };
BLOCK_SIZE bsize = xd->mi[0]->sb_type;
foreach_overlappable_nb_left(cm, xd, mi_row,
max_neighbor_obmc[mi_size_high_log2[bsize]],
build_prediction_by_left_pred, &ctxt);
xd->mb_to_top_edge = -((mi_row * MI_SIZE) * 8);
xd->mb_to_right_edge -= (this_width - pred_width) * 8;
xd->mb_to_bottom_edge = ctxt.mb_to_far_edge;
}
void av1_build_obmc_inter_predictors_sb(const AV1_COMMON *cm, MACROBLOCKD *xd,
int mi_row, int mi_col) {
const int num_planes = av1_num_planes(cm);
DECLARE_ALIGNED(16, uint8_t, tmp_buf1[2 * MAX_MB_PLANE * MAX_SB_SQUARE]);
DECLARE_ALIGNED(16, uint8_t, tmp_buf2[2 * MAX_MB_PLANE * MAX_SB_SQUARE]);
uint8_t *dst_buf1[MAX_MB_PLANE], *dst_buf2[MAX_MB_PLANE];
int dst_stride1[MAX_MB_PLANE] = { MAX_SB_SIZE, MAX_SB_SIZE, MAX_SB_SIZE };
int dst_stride2[MAX_MB_PLANE] = { MAX_SB_SIZE, MAX_SB_SIZE, MAX_SB_SIZE };
int dst_width1[MAX_MB_PLANE] = { MAX_SB_SIZE, MAX_SB_SIZE, MAX_SB_SIZE };
int dst_width2[MAX_MB_PLANE] = { MAX_SB_SIZE, MAX_SB_SIZE, MAX_SB_SIZE };
int dst_height1[MAX_MB_PLANE] = { MAX_SB_SIZE, MAX_SB_SIZE, MAX_SB_SIZE };
int dst_height2[MAX_MB_PLANE] = { MAX_SB_SIZE, MAX_SB_SIZE, MAX_SB_SIZE };
if (xd->cur_buf->flags & YV12_FLAG_HIGHBITDEPTH) {
int len = sizeof(uint16_t);
dst_buf1[0] = CONVERT_TO_BYTEPTR(tmp_buf1);
dst_buf1[1] = CONVERT_TO_BYTEPTR(tmp_buf1 + MAX_SB_SQUARE * len);
dst_buf1[2] = CONVERT_TO_BYTEPTR(tmp_buf1 + MAX_SB_SQUARE * 2 * len);
dst_buf2[0] = CONVERT_TO_BYTEPTR(tmp_buf2);
dst_buf2[1] = CONVERT_TO_BYTEPTR(tmp_buf2 + MAX_SB_SQUARE * len);
dst_buf2[2] = CONVERT_TO_BYTEPTR(tmp_buf2 + MAX_SB_SQUARE * 2 * len);
} else {
dst_buf1[0] = tmp_buf1;
dst_buf1[1] = tmp_buf1 + MAX_SB_SQUARE;
dst_buf1[2] = tmp_buf1 + MAX_SB_SQUARE * 2;
dst_buf2[0] = tmp_buf2;
dst_buf2[1] = tmp_buf2 + MAX_SB_SQUARE;
dst_buf2[2] = tmp_buf2 + MAX_SB_SQUARE * 2;
}
av1_build_prediction_by_above_preds(cm, xd, mi_row, mi_col, dst_buf1,
dst_width1, dst_height1, dst_stride1);
av1_build_prediction_by_left_preds(cm, xd, mi_row, mi_col, dst_buf2,
dst_width2, dst_height2, dst_stride2);
av1_setup_dst_planes(xd->plane, xd->mi[0]->sb_type, get_frame_new_buffer(cm),
mi_row, mi_col, 0, num_planes);
av1_build_obmc_inter_prediction(cm, xd, mi_row, mi_col, dst_buf1, dst_stride1,
dst_buf2, dst_stride2);
(xd->n4_h - rel_mi_row - left_mi_height) * MI_SIZE * 8;
}
/* clang-format off */
@ -1668,127 +1160,3 @@ void av1_build_interintra_predictors_sbuv(const AV1_COMMON *cm, MACROBLOCKD *xd,
av1_build_interintra_predictors_sbp(cm, xd, upred, ustride, ctx, 1, bsize);
av1_build_interintra_predictors_sbp(cm, xd, vpred, vstride, ctx, 2, bsize);
}
void av1_build_interintra_predictors(const AV1_COMMON *cm, MACROBLOCKD *xd,
uint8_t *ypred, uint8_t *upred,
uint8_t *vpred, int ystride, int ustride,
int vstride, BUFFER_SET *ctx,
BLOCK_SIZE bsize) {
av1_build_interintra_predictors_sbp(cm, xd, ypred, ystride, ctx, 0, bsize);
av1_build_interintra_predictors_sbuv(cm, xd, upred, vpred, ustride, vstride,
ctx, bsize);
}
// Builds the inter-predictor for the single ref case
// for use in the encoder to search the wedges efficiently.
static void build_inter_predictors_single_buf(MACROBLOCKD *xd, int plane,
int bw, int bh, int x, int y,
int w, int h, int mi_x, int mi_y,
int ref, uint8_t *const ext_dst,
int ext_dst_stride,
int can_use_previous) {
struct macroblockd_plane *const pd = &xd->plane[plane];
const MB_MODE_INFO *mi = xd->mi[0];
const struct scale_factors *const sf = &xd->block_refs[ref]->sf;
struct buf_2d *const pre_buf = &pd->pre[ref];
uint8_t *const dst = get_buf_by_bd(xd, ext_dst) + ext_dst_stride * y + x;
const MV mv = mi->mv[ref].as_mv;
ConvolveParams conv_params = get_conv_params(ref, 0, plane, xd->bd);
WarpTypesAllowed warp_types;
const WarpedMotionParams *const wm = &xd->global_motion[mi->ref_frame[ref]];
warp_types.global_warp_allowed = is_global_mv_block(mi, wm->wmtype);
warp_types.local_warp_allowed = mi->motion_mode == WARPED_CAUSAL;
const int pre_x = (mi_x) >> pd->subsampling_x;
const int pre_y = (mi_y) >> pd->subsampling_y;
uint8_t *pre;
SubpelParams subpel_params;
calc_subpel_params(xd, sf, mv, plane, pre_x, pre_y, x, y, pre_buf, &pre,
&subpel_params, bw, bh);
av1_make_inter_predictor(pre, pre_buf->stride, dst, ext_dst_stride,
&subpel_params, sf, w, h, &conv_params,
mi->interp_filters, &warp_types, pre_x + x,
pre_y + y, plane, ref, mi, 0, xd, can_use_previous);
}
void av1_build_inter_predictors_for_planes_single_buf(
MACROBLOCKD *xd, BLOCK_SIZE bsize, int plane_from, int plane_to, int mi_row,
int mi_col, int ref, uint8_t *ext_dst[3], int ext_dst_stride[3],
int can_use_previous) {
int plane;
const int mi_x = mi_col * MI_SIZE;
const int mi_y = mi_row * MI_SIZE;
for (plane = plane_from; plane <= plane_to; ++plane) {
const BLOCK_SIZE plane_bsize = get_plane_block_size(
bsize, xd->plane[plane].subsampling_x, xd->plane[plane].subsampling_y);
const int bw = block_size_wide[plane_bsize];
const int bh = block_size_high[plane_bsize];
build_inter_predictors_single_buf(xd, plane, bw, bh, 0, 0, bw, bh, mi_x,
mi_y, ref, ext_dst[plane],
ext_dst_stride[plane], can_use_previous);
}
}
static void build_wedge_inter_predictor_from_buf(
MACROBLOCKD *xd, int plane, int x, int y, int w, int h, uint8_t *ext_dst0,
int ext_dst_stride0, uint8_t *ext_dst1, int ext_dst_stride1) {
MB_MODE_INFO *const mbmi = xd->mi[0];
const int is_compound = has_second_ref(mbmi);
MACROBLOCKD_PLANE *const pd = &xd->plane[plane];
struct buf_2d *const dst_buf = &pd->dst;
uint8_t *const dst = dst_buf->buf + dst_buf->stride * y + x;
mbmi->interinter_comp.seg_mask = xd->seg_mask;
const INTERINTER_COMPOUND_DATA *comp_data = &mbmi->interinter_comp;
if (is_compound && is_masked_compound_type(comp_data->type)) {
if (!plane && comp_data->type == COMPOUND_DIFFWTD) {
if (xd->cur_buf->flags & YV12_FLAG_HIGHBITDEPTH)
av1_build_compound_diffwtd_mask_highbd(
comp_data->seg_mask, comp_data->mask_type,
CONVERT_TO_BYTEPTR(ext_dst0), ext_dst_stride0,
CONVERT_TO_BYTEPTR(ext_dst1), ext_dst_stride1, h, w, xd->bd);
else
av1_build_compound_diffwtd_mask(
comp_data->seg_mask, comp_data->mask_type, ext_dst0,
ext_dst_stride0, ext_dst1, ext_dst_stride1, h, w);
}
if (xd->cur_buf->flags & YV12_FLAG_HIGHBITDEPTH)
build_masked_compound_highbd(
dst, dst_buf->stride, CONVERT_TO_BYTEPTR(ext_dst0), ext_dst_stride0,
CONVERT_TO_BYTEPTR(ext_dst1), ext_dst_stride1, comp_data,
mbmi->sb_type, h, w, xd->bd);
else
build_masked_compound(dst, dst_buf->stride, ext_dst0, ext_dst_stride0,
ext_dst1, ext_dst_stride1, comp_data, mbmi->sb_type,
h, w);
} else {
if (xd->cur_buf->flags & YV12_FLAG_HIGHBITDEPTH)
aom_highbd_convolve_copy(CONVERT_TO_BYTEPTR(ext_dst0), ext_dst_stride0,
dst, dst_buf->stride, NULL, 0, NULL, 0, w, h,
xd->bd);
else
aom_convolve_copy(ext_dst0, ext_dst_stride0, dst, dst_buf->stride, NULL,
0, NULL, 0, w, h);
}
}
void av1_build_wedge_inter_predictor_from_buf(MACROBLOCKD *xd, BLOCK_SIZE bsize,
int plane_from, int plane_to,
uint8_t *ext_dst0[3],
int ext_dst_stride0[3],
uint8_t *ext_dst1[3],
int ext_dst_stride1[3]) {
int plane;
for (plane = plane_from; plane <= plane_to; ++plane) {
const BLOCK_SIZE plane_bsize = get_plane_block_size(
bsize, xd->plane[plane].subsampling_x, xd->plane[plane].subsampling_y);
const int bw = block_size_wide[plane_bsize];
const int bh = block_size_high[plane_bsize];
build_wedge_inter_predictor_from_buf(
xd, plane, 0, 0, bw, bh, ext_dst0[plane], ext_dst_stride0[plane],
ext_dst1[plane], ext_dst_stride1[plane]);
}
}

View file

@ -9,8 +9,8 @@
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AV1_COMMON_RECONINTER_H_
#define AV1_COMMON_RECONINTER_H_
#ifndef AOM_AV1_COMMON_RECONINTER_H_
#define AOM_AV1_COMMON_RECONINTER_H_
#include "av1/common/filter.h"
#include "av1/common/onyxc_int.h"
@ -113,40 +113,48 @@ static INLINE void inter_predictor(const uint8_t *src, int src_stride,
const SubpelParams *subpel_params,
const struct scale_factors *sf, int w, int h,
ConvolveParams *conv_params,
InterpFilters interp_filters) {
InterpFilters interp_filters,
int is_intrabc) {
assert(conv_params->do_average == 0 || conv_params->do_average == 1);
assert(sf);
if (has_scale(subpel_params->xs, subpel_params->ys)) {
const int is_scaled = has_scale(subpel_params->xs, subpel_params->ys);
assert(IMPLIES(is_intrabc, !is_scaled));
if (is_scaled) {
av1_convolve_2d_facade(src, src_stride, dst, dst_stride, w, h,
interp_filters, subpel_params->subpel_x,
subpel_params->xs, subpel_params->subpel_y,
subpel_params->ys, 1, conv_params, sf);
subpel_params->ys, 1, conv_params, sf, is_intrabc);
} else {
SubpelParams sp = *subpel_params;
revert_scale_extra_bits(&sp);
av1_convolve_2d_facade(src, src_stride, dst, dst_stride, w, h,
interp_filters, sp.subpel_x, sp.xs, sp.subpel_y,
sp.ys, 0, conv_params, sf);
sp.ys, 0, conv_params, sf, is_intrabc);
}
}
static INLINE void highbd_inter_predictor(
const uint8_t *src, int src_stride, uint8_t *dst, int dst_stride,
const SubpelParams *subpel_params, const struct scale_factors *sf, int w,
int h, ConvolveParams *conv_params, InterpFilters interp_filters, int bd) {
static INLINE void highbd_inter_predictor(const uint8_t *src, int src_stride,
uint8_t *dst, int dst_stride,
const SubpelParams *subpel_params,
const struct scale_factors *sf, int w,
int h, ConvolveParams *conv_params,
InterpFilters interp_filters,
int is_intrabc, int bd) {
assert(conv_params->do_average == 0 || conv_params->do_average == 1);
assert(sf);
if (has_scale(subpel_params->xs, subpel_params->ys)) {
av1_highbd_convolve_2d_facade(src, src_stride, dst, dst_stride, w, h,
interp_filters, subpel_params->subpel_x,
subpel_params->xs, subpel_params->subpel_y,
subpel_params->ys, 1, conv_params, sf, bd);
const int is_scaled = has_scale(subpel_params->xs, subpel_params->ys);
assert(IMPLIES(is_intrabc, !is_scaled));
if (is_scaled) {
av1_highbd_convolve_2d_facade(
src, src_stride, dst, dst_stride, w, h, interp_filters,
subpel_params->subpel_x, subpel_params->xs, subpel_params->subpel_y,
subpel_params->ys, 1, conv_params, sf, is_intrabc, bd);
} else {
SubpelParams sp = *subpel_params;
revert_scale_extra_bits(&sp);
av1_highbd_convolve_2d_facade(src, src_stride, dst, dst_stride, w, h,
interp_filters, sp.subpel_x, sp.xs,
sp.subpel_y, sp.ys, 0, conv_params, sf, bd);
av1_highbd_convolve_2d_facade(
src, src_stride, dst, dst_stride, w, h, interp_filters, sp.subpel_x,
sp.xs, sp.subpel_y, sp.ys, 0, conv_params, sf, is_intrabc, bd);
}
}
@ -237,35 +245,6 @@ static INLINE MV clamp_mv_to_umv_border_sb(const MACROBLOCKD *xd,
return clamped_mv;
}
void av1_build_inter_predictors_sby(const AV1_COMMON *cm, MACROBLOCKD *xd,
int mi_row, int mi_col, BUFFER_SET *ctx,
BLOCK_SIZE bsize);
void av1_build_inter_predictors_sbuv(const AV1_COMMON *cm, MACROBLOCKD *xd,
int mi_row, int mi_col, BUFFER_SET *ctx,
BLOCK_SIZE bsize);
void av1_build_inter_predictors_sb(const AV1_COMMON *cm, MACROBLOCKD *xd,
int mi_row, int mi_col, BUFFER_SET *ctx,
BLOCK_SIZE bsize);
void av1_build_inter_predictor(const uint8_t *src, int src_stride, uint8_t *dst,
int dst_stride, const MV *src_mv,
const struct scale_factors *sf, int w, int h,
ConvolveParams *conv_params,
InterpFilters interp_filters,
const WarpTypesAllowed *warp_types, int p_col,
int p_row, int plane, int ref,
enum mv_precision precision, int x, int y,
const MACROBLOCKD *xd, int can_use_previous);
void av1_highbd_build_inter_predictor(
const uint8_t *src, int src_stride, uint8_t *dst, int dst_stride,
const MV *mv_q3, const struct scale_factors *sf, int w, int h, int do_avg,
InterpFilters interp_filters, const WarpTypesAllowed *warp_types, int p_col,
int p_row, int plane, enum mv_precision precision, int x, int y,
const MACROBLOCKD *xd, int can_use_previous);
static INLINE int scaled_buffer_offset(int x_offset, int y_offset, int stride,
const struct scale_factors *sf) {
const int x =
@ -303,32 +282,6 @@ void av1_setup_pre_planes(MACROBLOCKD *xd, int idx,
const YV12_BUFFER_CONFIG *src, int mi_row, int mi_col,
const struct scale_factors *sf, const int num_planes);
// Detect if the block have sub-pixel level motion vectors
// per component.
#define CHECK_SUBPEL 0
static INLINE int has_subpel_mv_component(const MB_MODE_INFO *const mbmi,
const MACROBLOCKD *const xd,
int dir) {
#if CHECK_SUBPEL
const BLOCK_SIZE bsize = mbmi->sb_type;
int plane;
int ref = (dir >> 1);
if (dir & 0x01) {
if (mbmi->mv[ref].as_mv.col & SUBPEL_MASK) return 1;
} else {
if (mbmi->mv[ref].as_mv.row & SUBPEL_MASK) return 1;
}
return 0;
#else
(void)mbmi;
(void)xd;
(void)dir;
return 1;
#endif
}
static INLINE void set_default_interp_filters(
MB_MODE_INFO *const mbmi, InterpFilter frame_interp_filter) {
mbmi->interp_filters =
@ -343,21 +296,6 @@ static INLINE int av1_is_interp_needed(const MACROBLOCKD *const xd) {
return 1;
}
static INLINE int av1_is_interp_search_needed(const MACROBLOCKD *const xd) {
MB_MODE_INFO *const mi = xd->mi[0];
const int is_compound = has_second_ref(mi);
int ref;
for (ref = 0; ref < 1 + is_compound; ++ref) {
int row_col;
for (row_col = 0; row_col < 2; ++row_col) {
const int dir = (ref << 1) + row_col;
if (has_subpel_mv_component(mi, xd, dir)) {
return 1;
}
}
}
return 0;
}
void av1_setup_build_prediction_by_above_pred(
MACROBLOCKD *xd, int rel_mi_col, uint8_t above_mi_width,
MB_MODE_INFO *above_mbmi, struct build_prediction_ctxt *ctxt,
@ -367,18 +305,6 @@ void av1_setup_build_prediction_by_left_pred(MACROBLOCKD *xd, int rel_mi_row,
MB_MODE_INFO *left_mbmi,
struct build_prediction_ctxt *ctxt,
const int num_planes);
void av1_build_prediction_by_above_preds(const AV1_COMMON *cm, MACROBLOCKD *xd,
int mi_row, int mi_col,
uint8_t *tmp_buf[MAX_MB_PLANE],
int tmp_width[MAX_MB_PLANE],
int tmp_height[MAX_MB_PLANE],
int tmp_stride[MAX_MB_PLANE]);
void av1_build_prediction_by_left_preds(const AV1_COMMON *cm, MACROBLOCKD *xd,
int mi_row, int mi_col,
uint8_t *tmp_buf[MAX_MB_PLANE],
int tmp_width[MAX_MB_PLANE],
int tmp_height[MAX_MB_PLANE],
int tmp_stride[MAX_MB_PLANE]);
void av1_build_obmc_inter_prediction(const AV1_COMMON *cm, MACROBLOCKD *xd,
int mi_row, int mi_col,
uint8_t *above[MAX_MB_PLANE],
@ -389,8 +315,6 @@ void av1_build_obmc_inter_prediction(const AV1_COMMON *cm, MACROBLOCKD *xd,
const uint8_t *av1_get_obmc_mask(int length);
void av1_count_overlappable_neighbors(const AV1_COMMON *cm, MACROBLOCKD *xd,
int mi_row, int mi_col);
void av1_build_obmc_inter_predictors_sb(const AV1_COMMON *cm, MACROBLOCKD *xd,
int mi_row, int mi_col);
#define MASK_MASTER_SIZE ((MAX_WEDGE_SIZE) << 1)
#define MASK_MASTER_STRIDE (MASK_MASTER_SIZE)
@ -406,12 +330,6 @@ static INLINE const uint8_t *av1_get_contiguous_soft_mask(int wedge_index,
const uint8_t *av1_get_compound_type_mask(
const INTERINTER_COMPOUND_DATA *const comp_data, BLOCK_SIZE sb_type);
void av1_build_interintra_predictors(const AV1_COMMON *cm, MACROBLOCKD *xd,
uint8_t *ypred, uint8_t *upred,
uint8_t *vpred, int ystride, int ustride,
int vstride, BUFFER_SET *ctx,
BLOCK_SIZE bsize);
// build interintra_predictors for one plane
void av1_build_interintra_predictors_sbp(const AV1_COMMON *cm, MACROBLOCKD *xd,
uint8_t *pred, int stride,
@ -431,18 +349,6 @@ void av1_combine_interintra(MACROBLOCKD *xd, BLOCK_SIZE bsize, int plane,
const uint8_t *inter_pred, int inter_stride,
const uint8_t *intra_pred, int intra_stride);
// Encoder only
void av1_build_inter_predictors_for_planes_single_buf(
MACROBLOCKD *xd, BLOCK_SIZE bsize, int plane_from, int plane_to, int mi_row,
int mi_col, int ref, uint8_t *ext_dst[3], int ext_dst_stride[3],
int can_use_previous);
void av1_build_wedge_inter_predictor_from_buf(MACROBLOCKD *xd, BLOCK_SIZE bsize,
int plane_from, int plane_to,
uint8_t *ext_dst0[3],
int ext_dst_stride0[3],
uint8_t *ext_dst1[3],
int ext_dst_stride1[3]);
void av1_jnt_comp_weight_assign(const AV1_COMMON *cm, const MB_MODE_INFO *mbmi,
int order_idx, int *fwd_offset, int *bck_offset,
int *use_jnt_comp_avg, int is_compound);
@ -456,4 +362,4 @@ int av1_allow_warp(const MB_MODE_INFO *const mbmi,
} // extern "C"
#endif
#endif // AV1_COMMON_RECONINTER_H_
#endif // AOM_AV1_COMMON_RECONINTER_H_

View file

@ -9,8 +9,8 @@
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AV1_COMMON_RECONINTRA_H_
#define AV1_COMMON_RECONINTRA_H_
#ifndef AOM_AV1_COMMON_RECONINTRA_H_
#define AOM_AV1_COMMON_RECONINTRA_H_
#include <stdlib.h>
@ -116,4 +116,4 @@ static INLINE int av1_use_intra_edge_upsample(int bs0, int bs1, int delta,
#ifdef __cplusplus
} // extern "C"
#endif
#endif // AV1_COMMON_RECONINTRA_H_
#endif // AOM_AV1_COMMON_RECONINTRA_H_

View file

@ -170,42 +170,6 @@ static const InterpKernel filteredinterp_filters875[(1 << RS_SUBPEL_BITS)] = {
{ -1, 3, -9, 17, 112, 10, -7, 3 }, { -1, 3, -8, 15, 112, 12, -7, 2 },
};
// Filters for interpolation (full-band) - no filtering for integer pixels
static const InterpKernel filteredinterp_filters1000[(1 << RS_SUBPEL_BITS)] = {
{ 0, 0, 0, 128, 0, 0, 0, 0 }, { 0, 0, -1, 128, 2, -1, 0, 0 },
{ 0, 1, -3, 127, 4, -2, 1, 0 }, { 0, 1, -4, 127, 6, -3, 1, 0 },
{ 0, 2, -6, 126, 8, -3, 1, 0 }, { 0, 2, -7, 125, 11, -4, 1, 0 },
{ -1, 2, -8, 125, 13, -5, 2, 0 }, { -1, 3, -9, 124, 15, -6, 2, 0 },
{ -1, 3, -10, 123, 18, -6, 2, -1 }, { -1, 3, -11, 122, 20, -7, 3, -1 },
{ -1, 4, -12, 121, 22, -8, 3, -1 }, { -1, 4, -13, 120, 25, -9, 3, -1 },
{ -1, 4, -14, 118, 28, -9, 3, -1 }, { -1, 4, -15, 117, 30, -10, 4, -1 },
{ -1, 5, -16, 116, 32, -11, 4, -1 }, { -1, 5, -16, 114, 35, -12, 4, -1 },
{ -1, 5, -17, 112, 38, -12, 4, -1 }, { -1, 5, -18, 111, 40, -13, 5, -1 },
{ -1, 5, -18, 109, 43, -14, 5, -1 }, { -1, 6, -19, 107, 45, -14, 5, -1 },
{ -1, 6, -19, 105, 48, -15, 5, -1 }, { -1, 6, -19, 103, 51, -16, 5, -1 },
{ -1, 6, -20, 101, 53, -16, 6, -1 }, { -1, 6, -20, 99, 56, -17, 6, -1 },
{ -1, 6, -20, 97, 58, -17, 6, -1 }, { -1, 6, -20, 95, 61, -18, 6, -1 },
{ -2, 7, -20, 93, 64, -18, 6, -2 }, { -2, 7, -20, 91, 66, -19, 6, -1 },
{ -2, 7, -20, 88, 69, -19, 6, -1 }, { -2, 7, -20, 86, 71, -19, 6, -1 },
{ -2, 7, -20, 84, 74, -20, 7, -2 }, { -2, 7, -20, 81, 76, -20, 7, -1 },
{ -2, 7, -20, 79, 79, -20, 7, -2 }, { -1, 7, -20, 76, 81, -20, 7, -2 },
{ -2, 7, -20, 74, 84, -20, 7, -2 }, { -1, 6, -19, 71, 86, -20, 7, -2 },
{ -1, 6, -19, 69, 88, -20, 7, -2 }, { -1, 6, -19, 66, 91, -20, 7, -2 },
{ -2, 6, -18, 64, 93, -20, 7, -2 }, { -1, 6, -18, 61, 95, -20, 6, -1 },
{ -1, 6, -17, 58, 97, -20, 6, -1 }, { -1, 6, -17, 56, 99, -20, 6, -1 },
{ -1, 6, -16, 53, 101, -20, 6, -1 }, { -1, 5, -16, 51, 103, -19, 6, -1 },
{ -1, 5, -15, 48, 105, -19, 6, -1 }, { -1, 5, -14, 45, 107, -19, 6, -1 },
{ -1, 5, -14, 43, 109, -18, 5, -1 }, { -1, 5, -13, 40, 111, -18, 5, -1 },
{ -1, 4, -12, 38, 112, -17, 5, -1 }, { -1, 4, -12, 35, 114, -16, 5, -1 },
{ -1, 4, -11, 32, 116, -16, 5, -1 }, { -1, 4, -10, 30, 117, -15, 4, -1 },
{ -1, 3, -9, 28, 118, -14, 4, -1 }, { -1, 3, -9, 25, 120, -13, 4, -1 },
{ -1, 3, -8, 22, 121, -12, 4, -1 }, { -1, 3, -7, 20, 122, -11, 3, -1 },
{ -1, 2, -6, 18, 123, -10, 3, -1 }, { 0, 2, -6, 15, 124, -9, 3, -1 },
{ 0, 2, -5, 13, 125, -8, 2, -1 }, { 0, 1, -4, 11, 125, -7, 2, 0 },
{ 0, 1, -3, 8, 126, -6, 2, 0 }, { 0, 1, -3, 6, 127, -4, 1, 0 },
{ 0, 1, -2, 4, 127, -3, 1, 0 }, { 0, 0, -1, 2, 128, -1, 0, 0 },
};
const int16_t av1_resize_filter_normative[(
1 << RS_SUBPEL_BITS)][UPSCALE_NORMATIVE_TAPS] = {
#if UPSCALE_NORMATIVE_TAPS == 8
@ -246,6 +210,9 @@ const int16_t av1_resize_filter_normative[(
#endif // UPSCALE_NORMATIVE_TAPS == 8
};
// Filters for interpolation (full-band) - no filtering for integer pixels
#define filteredinterp_filters1000 av1_resize_filter_normative
// Filters for factor of 2 downsampling.
static const int16_t av1_down2_symeven_half_filter[] = { 56, 12, -3, -1 };
static const int16_t av1_down2_symodd_half_filter[] = { 64, 35, 0, -3 };

View file

@ -9,8 +9,8 @@
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AV1_ENCODER_RESIZE_H_
#define AV1_ENCODER_RESIZE_H_
#ifndef AOM_AV1_COMMON_RESIZE_H_
#define AOM_AV1_COMMON_RESIZE_H_
#include <stdio.h>
#include "aom/aom_integer.h"
@ -109,4 +109,4 @@ int32_t av1_get_upscale_convolve_step(int in_length, int out_length);
} // extern "C"
#endif
#endif // AV1_ENCODER_RESIZE_H_
#endif // AOM_AV1_COMMON_RESIZE_H_

View file

@ -661,9 +661,10 @@ const int32_t one_by_x[MAX_NELEM] = {
293, 273, 256, 241, 228, 216, 205, 195, 186, 178, 171, 164,
};
static void selfguided_restoration_fast_internal(
int32_t *dgd, int width, int height, int dgd_stride, int32_t *dst,
int dst_stride, int bit_depth, int sgr_params_idx, int radius_idx) {
static void calculate_intermediate_result(int32_t *dgd, int width, int height,
int dgd_stride, int bit_depth,
int sgr_params_idx, int radius_idx,
int pass, int32_t *A, int32_t *B) {
const sgr_params_type *const params = &sgr_params[sgr_params_idx];
const int r = params->r[radius_idx];
const int width_ext = width + 2 * SGRPROJ_BORDER_HORZ;
@ -673,10 +674,7 @@ static void selfguided_restoration_fast_internal(
// We also align the stride to a multiple of 16 bytes, for consistency
// with the SIMD version of this function.
int buf_stride = ((width_ext + 3) & ~3) + 16;
int32_t A_[RESTORATION_PROC_UNIT_PELS];
int32_t B_[RESTORATION_PROC_UNIT_PELS];
int32_t *A = A_;
int32_t *B = B_;
const int step = pass == 0 ? 1 : 2;
int i, j;
assert(r <= MAX_RADIUS && "Need MAX_RADIUS >= r");
@ -691,7 +689,7 @@ static void selfguided_restoration_fast_internal(
B += SGRPROJ_BORDER_VERT * buf_stride + SGRPROJ_BORDER_HORZ;
// Calculate the eventual A[] and B[] arrays. Include a 1-pixel border - ie,
// for a 64x64 processing unit, we calculate 66x66 pixels of A[] and B[].
for (i = -1; i < height + 1; i += 2) {
for (i = -1; i < height + 1; i += step) {
for (j = -1; j < width + 1; ++j) {
const int k = i * buf_stride + j;
const int n = (2 * r + 1) * (2 * r + 1);
@ -754,7 +752,31 @@ static void selfguided_restoration_fast_internal(
SGRPROJ_RECIP_BITS);
}
}
}
static void selfguided_restoration_fast_internal(
int32_t *dgd, int width, int height, int dgd_stride, int32_t *dst,
int dst_stride, int bit_depth, int sgr_params_idx, int radius_idx) {
const sgr_params_type *const params = &sgr_params[sgr_params_idx];
const int r = params->r[radius_idx];
const int width_ext = width + 2 * SGRPROJ_BORDER_HORZ;
// Adjusting the stride of A and B here appears to avoid bad cache effects,
// leading to a significant speed improvement.
// We also align the stride to a multiple of 16 bytes, for consistency
// with the SIMD version of this function.
int buf_stride = ((width_ext + 3) & ~3) + 16;
int32_t A_[RESTORATION_PROC_UNIT_PELS];
int32_t B_[RESTORATION_PROC_UNIT_PELS];
int32_t *A = A_;
int32_t *B = B_;
int i, j;
calculate_intermediate_result(dgd, width, height, dgd_stride, bit_depth,
sgr_params_idx, radius_idx, 1, A, B);
A += SGRPROJ_BORDER_VERT * buf_stride + SGRPROJ_BORDER_HORZ;
B += SGRPROJ_BORDER_VERT * buf_stride + SGRPROJ_BORDER_HORZ;
// Use the A[] and B[] arrays to calculate the filtered image
(void)r;
assert(r == 2);
for (i = 0; i < height; ++i) {
if (!(i & 1)) { // even row
@ -796,10 +818,7 @@ static void selfguided_restoration_internal(int32_t *dgd, int width, int height,
int dst_stride, int bit_depth,
int sgr_params_idx,
int radius_idx) {
const sgr_params_type *const params = &sgr_params[sgr_params_idx];
const int r = params->r[radius_idx];
const int width_ext = width + 2 * SGRPROJ_BORDER_HORZ;
const int height_ext = height + 2 * SGRPROJ_BORDER_VERT;
// Adjusting the stride of A and B here appears to avoid bad cache effects,
// leading to a significant speed improvement.
// We also align the stride to a multiple of 16 bytes, for consistency
@ -810,82 +829,11 @@ static void selfguided_restoration_internal(int32_t *dgd, int width, int height,
int32_t *A = A_;
int32_t *B = B_;
int i, j;
assert(r <= MAX_RADIUS && "Need MAX_RADIUS >= r");
assert(r <= SGRPROJ_BORDER_VERT - 1 && r <= SGRPROJ_BORDER_HORZ - 1 &&
"Need SGRPROJ_BORDER_* >= r+1");
boxsum(dgd - dgd_stride * SGRPROJ_BORDER_VERT - SGRPROJ_BORDER_HORZ,
width_ext, height_ext, dgd_stride, r, 0, B, buf_stride);
boxsum(dgd - dgd_stride * SGRPROJ_BORDER_VERT - SGRPROJ_BORDER_HORZ,
width_ext, height_ext, dgd_stride, r, 1, A, buf_stride);
calculate_intermediate_result(dgd, width, height, dgd_stride, bit_depth,
sgr_params_idx, radius_idx, 0, A, B);
A += SGRPROJ_BORDER_VERT * buf_stride + SGRPROJ_BORDER_HORZ;
B += SGRPROJ_BORDER_VERT * buf_stride + SGRPROJ_BORDER_HORZ;
// Calculate the eventual A[] and B[] arrays. Include a 1-pixel border - ie,
// for a 64x64 processing unit, we calculate 66x66 pixels of A[] and B[].
for (i = -1; i < height + 1; ++i) {
for (j = -1; j < width + 1; ++j) {
const int k = i * buf_stride + j;
const int n = (2 * r + 1) * (2 * r + 1);
// a < 2^16 * n < 2^22 regardless of bit depth
uint32_t a = ROUND_POWER_OF_TWO(A[k], 2 * (bit_depth - 8));
// b < 2^8 * n < 2^14 regardless of bit depth
uint32_t b = ROUND_POWER_OF_TWO(B[k], bit_depth - 8);
// Each term in calculating p = a * n - b * b is < 2^16 * n^2 < 2^28,
// and p itself satisfies p < 2^14 * n^2 < 2^26.
// This bound on p is due to:
// https://en.wikipedia.org/wiki/Popoviciu's_inequality_on_variances
//
// Note: Sometimes, in high bit depth, we can end up with a*n < b*b.
// This is an artefact of rounding, and can only happen if all pixels
// are (almost) identical, so in this case we saturate to p=0.
uint32_t p = (a * n < b * b) ? 0 : a * n - b * b;
const uint32_t s = params->s[radius_idx];
// p * s < (2^14 * n^2) * round(2^20 / n^2 eps) < 2^34 / eps < 2^32
// as long as eps >= 4. So p * s fits into a uint32_t, and z < 2^12
// (this holds even after accounting for the rounding in s)
const uint32_t z = ROUND_POWER_OF_TWO(p * s, SGRPROJ_MTABLE_BITS);
// Note: We have to be quite careful about the value of A[k].
// This is used as a blend factor between individual pixel values and the
// local mean. So it logically has a range of [0, 256], including both
// endpoints.
//
// This is a pain for hardware, as we'd like something which can be stored
// in exactly 8 bits.
// Further, in the calculation of B[k] below, if z == 0 and r == 2,
// then A[k] "should be" 0. But then we can end up setting B[k] to a value
// slightly above 2^(8 + bit depth), due to rounding in the value of
// one_by_x[25-1].
//
// Thus we saturate so that, when z == 0, A[k] is set to 1 instead of 0.
// This fixes the above issues (256 - A[k] fits in a uint8, and we can't
// overflow), without significantly affecting the final result: z == 0
// implies that the image is essentially "flat", so the local mean and
// individual pixel values are very similar.
//
// Note that saturating on the other side, ie. requring A[k] <= 255,
// would be a bad idea, as that corresponds to the case where the image
// is very variable, when we want to preserve the local pixel value as
// much as possible.
A[k] = x_by_xplus1[AOMMIN(z, 255)]; // in range [1, 256]
// SGRPROJ_SGR - A[k] < 2^8 (from above), B[k] < 2^(bit_depth) * n,
// one_by_x[n - 1] = round(2^12 / n)
// => the product here is < 2^(20 + bit_depth) <= 2^32,
// and B[k] is set to a value < 2^(8 + bit depth)
// This holds even with the rounding in one_by_x and in the overall
// result, as long as SGRPROJ_SGR - A[k] is strictly less than 2^8.
B[k] = (int32_t)ROUND_POWER_OF_TWO((uint32_t)(SGRPROJ_SGR - A[k]) *
(uint32_t)B[k] *
(uint32_t)one_by_x[n - 1],
SGRPROJ_RECIP_BITS);
}
}
// Use the A[] and B[] arrays to calculate the filtered image
for (i = 0; i < height; ++i) {
for (j = 0; j < width; ++j) {
@ -911,10 +859,10 @@ static void selfguided_restoration_internal(int32_t *dgd, int width, int height,
}
}
void av1_selfguided_restoration_c(const uint8_t *dgd8, int width, int height,
int dgd_stride, int32_t *flt0, int32_t *flt1,
int flt_stride, int sgr_params_idx,
int bit_depth, int highbd) {
int av1_selfguided_restoration_c(const uint8_t *dgd8, int width, int height,
int dgd_stride, int32_t *flt0, int32_t *flt1,
int flt_stride, int sgr_params_idx,
int bit_depth, int highbd) {
int32_t dgd32_[RESTORATION_PROC_UNIT_PELS];
const int dgd32_stride = width + 2 * SGRPROJ_BORDER_HORZ;
int32_t *dgd32 =
@ -948,6 +896,7 @@ void av1_selfguided_restoration_c(const uint8_t *dgd8, int width, int height,
if (params->r[1] > 0)
selfguided_restoration_internal(dgd32, width, height, dgd32_stride, flt1,
flt_stride, bit_depth, sgr_params_idx, 1);
return 0;
}
void apply_selfguided_restoration_c(const uint8_t *dat8, int width, int height,
@ -959,8 +908,10 @@ void apply_selfguided_restoration_c(const uint8_t *dat8, int width, int height,
int32_t *flt1 = flt0 + RESTORATION_UNITPELS_MAX;
assert(width * height <= RESTORATION_UNITPELS_MAX);
av1_selfguided_restoration_c(dat8, width, height, stride, flt0, flt1, width,
eps, bit_depth, highbd);
const int ret = av1_selfguided_restoration_c(
dat8, width, height, stride, flt0, flt1, width, eps, bit_depth, highbd);
(void)ret;
assert(!ret);
const sgr_params_type *const params = &sgr_params[eps];
int xq[2];
decode_xq(xqd, xq, params);

View file

@ -9,8 +9,8 @@
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AV1_COMMON_RESTORATION_H_
#define AV1_COMMON_RESTORATION_H_
#ifndef AOM_AV1_COMMON_RESTORATION_H_
#define AOM_AV1_COMMON_RESTORATION_H_
#include "aom_ports/mem.h"
#include "config/aom_config.h"
@ -120,6 +120,7 @@ extern "C" {
// If WIENER_WIN_CHROMA == WIENER_WIN - 2, that implies 5x5 filters are used for
// chroma. To use 7x7 for chroma set WIENER_WIN_CHROMA to WIENER_WIN.
#define WIENER_WIN_CHROMA (WIENER_WIN - 2)
#define WIENER_WIN2_CHROMA ((WIENER_WIN_CHROMA) * (WIENER_WIN_CHROMA))
#define WIENER_FILT_PREC_BITS 7
#define WIENER_FILT_STEP (1 << WIENER_FILT_PREC_BITS)
@ -373,4 +374,4 @@ void av1_lr_sync_write_dummy(void *const lr_sync, int r, int c,
} // extern "C"
#endif
#endif // AV1_COMMON_RESTORATION_H_
#endif // AOM_AV1_COMMON_RESTORATION_H_

View file

@ -9,12 +9,11 @@
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AV1_COMMON_SCALE_H_
#define AV1_COMMON_SCALE_H_
#ifndef AOM_AV1_COMMON_SCALE_H_
#define AOM_AV1_COMMON_SCALE_H_
#include "av1/common/convolve.h"
#include "av1/common/mv.h"
#include "aom_dsp/aom_convolve.h"
#ifdef __cplusplus
extern "C" {
@ -65,4 +64,4 @@ static INLINE int valid_ref_frame_size(int ref_width, int ref_height,
} // extern "C"
#endif
#endif // AV1_COMMON_SCALE_H_
#endif // AOM_AV1_COMMON_SCALE_H_

View file

@ -9,8 +9,8 @@
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AV1_COMMON_SCAN_H_
#define AV1_COMMON_SCAN_H_
#ifndef AOM_AV1_COMMON_SCAN_H_
#define AOM_AV1_COMMON_SCAN_H_
#include "aom/aom_integer.h"
#include "aom_ports/mem.h"
@ -52,4 +52,4 @@ static INLINE const SCAN_ORDER *get_scan(TX_SIZE tx_size, TX_TYPE tx_type) {
} // extern "C"
#endif
#endif // AV1_COMMON_SCAN_H_
#endif // AOM_AV1_COMMON_SCAN_H_

View file

@ -9,8 +9,8 @@
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AV1_COMMON_SEG_COMMON_H_
#define AV1_COMMON_SEG_COMMON_H_
#ifndef AOM_AV1_COMMON_SEG_COMMON_H_
#define AOM_AV1_COMMON_SEG_COMMON_H_
#include "aom_dsp/prob.h"
@ -101,4 +101,4 @@ static INLINE int get_segdata(const struct segmentation *seg, int segment_id,
} // extern "C"
#endif
#endif // AV1_COMMON_SEG_COMMON_H_
#endif // AOM_AV1_COMMON_SEG_COMMON_H_

View file

@ -304,8 +304,9 @@ static INLINE void thread_loop_filter_rows(
}
// Row-based multi-threaded loopfilter hook
static int loop_filter_row_worker(AV1LfSync *const lf_sync,
LFWorkerData *const lf_data) {
static int loop_filter_row_worker(void *arg1, void *arg2) {
AV1LfSync *const lf_sync = (AV1LfSync *)arg1;
LFWorkerData *const lf_data = (LFWorkerData *)arg2;
thread_loop_filter_rows(lf_data->frame_buffer, lf_data->cm, lf_data->planes,
lf_data->xd, lf_sync);
return 1;
@ -342,7 +343,7 @@ static void loop_filter_rows_mt(YV12_BUFFER_CONFIG *frame, AV1_COMMON *cm,
AVxWorker *const worker = &workers[i];
LFWorkerData *const lf_data = &lf_sync->lfdata[i];
worker->hook = (AVxWorkerHook)loop_filter_row_worker;
worker->hook = loop_filter_row_worker;
worker->data1 = lf_sync;
worker->data2 = lf_data;
@ -649,8 +650,9 @@ AV1LrMTInfo *get_lr_job_info(AV1LrSync *lr_sync) {
}
// Implement row loop restoration for each thread.
static int loop_restoration_row_worker(AV1LrSync *const lr_sync,
LRWorkerData *lrworkerdata) {
static int loop_restoration_row_worker(void *arg1, void *arg2) {
AV1LrSync *const lr_sync = (AV1LrSync *)arg1;
LRWorkerData *lrworkerdata = (LRWorkerData *)arg2;
AV1LrStruct *lr_ctxt = (AV1LrStruct *)lrworkerdata->lr_ctxt;
FilterFrameCtxt *ctxt = lr_ctxt->ctxt;
int lr_unit_row;
@ -714,10 +716,12 @@ static void foreach_rest_unit_in_planes_mt(AV1LrStruct *lr_ctxt,
int num_rows_lr = 0;
for (int plane = 0; plane < num_planes; plane++) {
if (cm->rst_info[plane].frame_restoration_type == RESTORE_NONE) continue;
const AV1PixelRect tile_rect = ctxt[plane].tile_rect;
const int max_tile_h = tile_rect.bottom - tile_rect.top;
const int unit_size = cm->seq_params.sb_size == BLOCK_128X128 ? 128 : 64;
const int unit_size = cm->rst_info[plane].restoration_unit_size;
num_rows_lr =
AOMMAX(num_rows_lr, av1_lr_count_units_in_tile(unit_size, max_tile_h));
@ -746,7 +750,7 @@ static void foreach_rest_unit_in_planes_mt(AV1LrStruct *lr_ctxt,
for (i = 0; i < num_workers; ++i) {
AVxWorker *const worker = &workers[i];
lr_sync->lrworkerdata[i].lr_ctxt = (void *)lr_ctxt;
worker->hook = (AVxWorkerHook)loop_restoration_row_worker;
worker->hook = loop_restoration_row_worker;
worker->data1 = lr_sync;
worker->data2 = &lr_sync->lrworkerdata[i];

View file

@ -9,8 +9,8 @@
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AV1_COMMON_LOOPFILTER_THREAD_H_
#define AV1_COMMON_LOOPFILTER_THREAD_H_
#ifndef AOM_AV1_COMMON_THREAD_COMMON_H_
#define AOM_AV1_COMMON_THREAD_COMMON_H_
#include "config/aom_config.h"
@ -116,4 +116,4 @@ void av1_loop_restoration_dealloc(AV1LrSync *lr_sync, int num_workers);
} // extern "C"
#endif
#endif // AV1_COMMON_LOOPFILTER_THREAD_H_
#endif // AOM_AV1_COMMON_THREAD_COMMON_H_

View file

@ -127,6 +127,22 @@ void av1_tile_set_col(TileInfo *tile, const AV1_COMMON *cm, int col) {
assert(tile->mi_col_end > tile->mi_col_start);
}
int av1_get_sb_rows_in_tile(AV1_COMMON *cm, TileInfo tile) {
int mi_rows_aligned_to_sb = ALIGN_POWER_OF_TWO(
tile.mi_row_end - tile.mi_row_start, cm->seq_params.mib_size_log2);
int sb_rows = mi_rows_aligned_to_sb >> cm->seq_params.mib_size_log2;
return sb_rows;
}
int av1_get_sb_cols_in_tile(AV1_COMMON *cm, TileInfo tile) {
int mi_cols_aligned_to_sb = ALIGN_POWER_OF_TWO(
tile.mi_col_end - tile.mi_col_start, cm->seq_params.mib_size_log2);
int sb_cols = mi_cols_aligned_to_sb >> cm->seq_params.mib_size_log2;
return sb_cols;
}
int get_tile_size(int mi_frame_size, int log2_tile_num, int *ntiles) {
// Round the frame up to a whole number of max superblocks
mi_frame_size = ALIGN_POWER_OF_TWO(mi_frame_size, MAX_MIB_SIZE_LOG2);

View file

@ -9,8 +9,8 @@
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AV1_COMMON_TILE_COMMON_H_
#define AV1_COMMON_TILE_COMMON_H_
#ifndef AOM_AV1_COMMON_TILE_COMMON_H_
#define AOM_AV1_COMMON_TILE_COMMON_H_
#ifdef __cplusplus
extern "C" {
@ -44,6 +44,9 @@ void av1_get_tile_n_bits(int mi_cols, int *min_log2_tile_cols,
// tiles horizontally or vertically in the frame.
int get_tile_size(int mi_frame_size, int log2_tile_num, int *ntiles);
int av1_get_sb_rows_in_tile(struct AV1Common *cm, TileInfo tile);
int av1_get_sb_cols_in_tile(struct AV1Common *cm, TileInfo tile);
typedef struct {
int left, top, right, bottom;
} AV1PixelRect;
@ -66,4 +69,4 @@ void av1_calculate_tile_rows(struct AV1Common *const cm);
} // extern "C"
#endif
#endif // AV1_COMMON_TILE_COMMON_H_
#endif // AOM_AV1_COMMON_TILE_COMMON_H_

View file

@ -9,8 +9,8 @@
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AOM_TIMING_H_
#define AOM_TIMING_H_
#ifndef AOM_AV1_COMMON_TIMING_H_
#define AOM_AV1_COMMON_TIMING_H_
#include "aom/aom_integer.h"
#include "av1/common/enums.h"
@ -56,4 +56,4 @@ void set_resource_availability_parameters(
int64_t max_level_bitrate(BITSTREAM_PROFILE seq_profile, int seq_level_idx,
int seq_tier);
#endif // AOM_TIMING_H_
#endif // AOM_AV1_COMMON_TIMING_H_

View file

@ -9,6 +9,9 @@
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AOM_AV1_COMMON_TOKEN_CDFS_H_
#define AOM_AV1_COMMON_TOKEN_CDFS_H_
#include "config/aom_config.h"
#include "av1/common/entropy.h"
@ -3548,3 +3551,5 @@ static const aom_cdf_prob av1_default_coeff_base_eob_multi_cdfs
{ AOM_CDF3(10923, 21845) },
{ AOM_CDF3(10923, 21845) },
{ AOM_CDF3(10923, 21845) } } } } };
#endif // AOM_AV1_COMMON_TOKEN_CDFS_H_

View file

@ -9,8 +9,8 @@
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AV1_COMMON_TXB_COMMON_H_
#define AV1_COMMON_TXB_COMMON_H_
#ifndef AOM_AV1_COMMON_TXB_COMMON_H_
#define AOM_AV1_COMMON_TXB_COMMON_H_
extern const int16_t k_eob_group_start[12];
extern const int16_t k_eob_offset_bits[12];
@ -34,24 +34,6 @@ static const int base_level_count_to_index[13] = {
0, 0, 1, 1, 1, 2, 2, 2, 2, 3, 3, 3, 3,
};
// Note: TX_PAD_2D is dependent to this offset table.
static const int base_ref_offset[BASE_CONTEXT_POSITION_NUM][2] = {
/* clang-format off*/
{ -2, 0 }, { -1, -1 }, { -1, 0 }, { -1, 1 }, { 0, -2 }, { 0, -1 }, { 0, 1 },
{ 0, 2 }, { 1, -1 }, { 1, 0 }, { 1, 1 }, { 2, 0 }
/* clang-format on*/
};
#define CONTEXT_MAG_POSITION_NUM 3
static const int mag_ref_offset_with_txclass[3][CONTEXT_MAG_POSITION_NUM][2] = {
{ { 0, 1 }, { 1, 0 }, { 1, 1 } },
{ { 0, 1 }, { 1, 0 }, { 0, 2 } },
{ { 0, 1 }, { 1, 0 }, { 2, 0 } }
};
static const int mag_ref_offset[CONTEXT_MAG_POSITION_NUM][2] = {
{ 0, 1 }, { 1, 0 }, { 1, 1 }
};
static const TX_CLASS tx_type_to_class[TX_TYPES] = {
TX_CLASS_2D, // DCT_DCT
TX_CLASS_2D, // ADST_DCT
@ -71,61 +53,6 @@ static const TX_CLASS tx_type_to_class[TX_TYPES] = {
TX_CLASS_HORIZ, // H_FLIPADST
};
static const int8_t eob_to_pos_small[33] = {
0, 1, 2, // 0-2
3, 3, // 3-4
4, 4, 4, 4, // 5-8
5, 5, 5, 5, 5, 5, 5, 5, // 9-16
6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6 // 17-32
};
static const int8_t eob_to_pos_large[17] = {
6, // place holder
7, // 33-64
8, 8, // 65-128
9, 9, 9, 9, // 129-256
10, 10, 10, 10, 10, 10, 10, 10, // 257-512
11 // 513-
};
static INLINE int get_eob_pos_token(const int eob, int *const extra) {
int t;
if (eob < 33) {
t = eob_to_pos_small[eob];
} else {
const int e = AOMMIN((eob - 1) >> 5, 16);
t = eob_to_pos_large[e];
}
*extra = eob - k_eob_group_start[t];
return t;
}
static INLINE int av1_get_eob_pos_ctx(const TX_TYPE tx_type,
const int eob_token) {
static const int8_t tx_type_to_offset[TX_TYPES] = {
-1, // DCT_DCT
-1, // ADST_DCT
-1, // DCT_ADST
-1, // ADST_ADST
-1, // FLIPADST_DCT
-1, // DCT_FLIPADST
-1, // FLIPADST_FLIPADST
-1, // ADST_FLIPADST
-1, // FLIPADST_ADST
-1, // IDTX
10, // V_DCT
10, // H_DCT
10, // V_ADST
10, // H_ADST
10, // V_FLIPADST
10, // H_FLIPADST
};
return eob_token + tx_type_to_offset[tx_type];
}
static INLINE int get_txb_bwl(TX_SIZE tx_size) {
tx_size = av1_get_adjusted_tx_size(tx_size);
return tx_size_wide_log2[tx_size];
@ -141,36 +68,6 @@ static INLINE int get_txb_high(TX_SIZE tx_size) {
return tx_size_high[tx_size];
}
static INLINE void get_base_count_mag(int *mag, int *count,
const tran_low_t *tcoeffs, int bwl,
int height, int row, int col) {
mag[0] = 0;
mag[1] = 0;
for (int i = 0; i < NUM_BASE_LEVELS; ++i) count[i] = 0;
for (int idx = 0; idx < BASE_CONTEXT_POSITION_NUM; ++idx) {
const int ref_row = row + base_ref_offset[idx][0];
const int ref_col = col + base_ref_offset[idx][1];
if (ref_row < 0 || ref_col < 0 || ref_row >= height ||
ref_col >= (1 << bwl))
continue;
const int pos = (ref_row << bwl) + ref_col;
tran_low_t abs_coeff = abs(tcoeffs[pos]);
// count
for (int i = 0; i < NUM_BASE_LEVELS; ++i) {
count[i] += abs_coeff > i;
}
// mag
if (base_ref_offset[idx][0] >= 0 && base_ref_offset[idx][1] >= 0) {
if (abs_coeff > mag[0]) {
mag[0] = abs_coeff;
mag[1] = 1;
} else if (abs_coeff == mag[0]) {
++mag[1];
}
}
}
}
static INLINE uint8_t *set_levels(uint8_t *const levels_buf, const int width) {
return levels_buf + TX_PAD_TOP * (width + TX_PAD_HOR);
}
@ -179,30 +76,6 @@ static INLINE int get_padded_idx(const int idx, const int bwl) {
return idx + ((idx >> bwl) << TX_PAD_HOR_LOG2);
}
static INLINE int get_level_count(const uint8_t *const levels, const int stride,
const int row, const int col, const int level,
const int (*nb_offset)[2], const int nb_num) {
int count = 0;
for (int idx = 0; idx < nb_num; ++idx) {
const int ref_row = row + nb_offset[idx][0];
const int ref_col = col + nb_offset[idx][1];
const int pos = ref_row * stride + ref_col;
count += levels[pos] > level;
}
return count;
}
static INLINE void get_level_mag(const uint8_t *const levels, const int stride,
const int row, const int col, int *const mag) {
for (int idx = 0; idx < CONTEXT_MAG_POSITION_NUM; ++idx) {
const int ref_row = row + mag_ref_offset[idx][0];
const int ref_col = col + mag_ref_offset[idx][1];
const int pos = ref_row * stride + ref_col;
mag[idx] = levels[pos];
}
}
static INLINE int get_base_ctx_from_count_mag(int row, int col, int count,
int sig_mag) {
const int ctx = base_level_count_to_index[count];
@ -267,84 +140,6 @@ static INLINE int get_base_ctx_from_count_mag(int row, int col, int count,
return ctx_idx;
}
static INLINE int get_base_ctx(const uint8_t *const levels,
const int c, // raster order
const int bwl, const int level_minus_1,
const int count) {
const int row = c >> bwl;
const int col = c - (row << bwl);
const int stride = (1 << bwl) + TX_PAD_HOR;
int mag_count = 0;
int nb_mag[3] = { 0 };
get_level_mag(levels, stride, row, col, nb_mag);
for (int idx = 0; idx < 3; ++idx)
mag_count += nb_mag[idx] > (level_minus_1 + 1);
const int ctx_idx =
get_base_ctx_from_count_mag(row, col, count, AOMMIN(2, mag_count));
return ctx_idx;
}
#define BR_CONTEXT_POSITION_NUM 8 // Base range coefficient context
// Note: TX_PAD_2D is dependent to this offset table.
static const int br_ref_offset[BR_CONTEXT_POSITION_NUM][2] = {
/* clang-format off*/
{ -1, -1 }, { -1, 0 }, { -1, 1 }, { 0, -1 },
{ 0, 1 }, { 1, -1 }, { 1, 0 }, { 1, 1 },
/* clang-format on*/
};
static const int br_level_map[9] = {
0, 0, 1, 1, 2, 2, 3, 3, 3,
};
// Note: If BR_MAG_OFFSET changes, the calculation of offset in
// get_br_ctx_from_count_mag() must be updated.
#define BR_MAG_OFFSET 1
// TODO(angiebird): optimize this function by using a table to map from
// count/mag to ctx
static INLINE int get_br_count_mag(int *mag, const tran_low_t *tcoeffs, int bwl,
int height, int row, int col, int level) {
mag[0] = 0;
mag[1] = 0;
int count = 0;
for (int idx = 0; idx < BR_CONTEXT_POSITION_NUM; ++idx) {
const int ref_row = row + br_ref_offset[idx][0];
const int ref_col = col + br_ref_offset[idx][1];
if (ref_row < 0 || ref_col < 0 || ref_row >= height ||
ref_col >= (1 << bwl))
continue;
const int pos = (ref_row << bwl) + ref_col;
tran_low_t abs_coeff = abs(tcoeffs[pos]);
count += abs_coeff > level;
if (br_ref_offset[idx][0] >= 0 && br_ref_offset[idx][1] >= 0) {
if (abs_coeff > mag[0]) {
mag[0] = abs_coeff;
mag[1] = 1;
} else if (abs_coeff == mag[0]) {
++mag[1];
}
}
}
return count;
}
static INLINE int get_br_ctx_from_count_mag(const int row, const int col,
const int count, const int mag) {
// DC: 0 - 1
// Top row: 2 - 4
// Left column: 5 - 7
// others: 8 - 11
static const int offset_pos[2][2] = { { 8, 5 }, { 2, 0 } };
const int mag_clamp = AOMMIN(mag, 6);
const int offset = mag_clamp >> 1;
const int ctx =
br_level_map[count] + offset * BR_TMP_OFFSET + offset_pos[!row][!col];
return ctx;
}
static INLINE int get_br_ctx_2d(const uint8_t *const levels,
const int c, // raster order
const int bwl) {
@ -396,38 +191,6 @@ static AOM_FORCE_INLINE int get_br_ctx(const uint8_t *const levels,
return mag + 14;
}
#define SIG_REF_OFFSET_NUM 5
// Note: TX_PAD_2D is dependent to these offset tables.
static const int sig_ref_offset[SIG_REF_OFFSET_NUM][2] = {
{ 0, 1 }, { 1, 0 }, { 1, 1 }, { 0, 2 }, { 2, 0 }
// , { 1, 2 }, { 2, 1 },
};
static const int sig_ref_offset_vert[SIG_REF_OFFSET_NUM][2] = {
{ 1, 0 }, { 2, 0 }, { 0, 1 }, { 3, 0 }, { 4, 0 }
// , { 1, 1 }, { 2, 1 },
};
static const int sig_ref_offset_horiz[SIG_REF_OFFSET_NUM][2] = {
{ 0, 1 }, { 0, 2 }, { 1, 0 }, { 0, 3 }, { 0, 4 }
// , { 1, 1 }, { 1, 2 },
};
#define SIG_REF_DIFF_OFFSET_NUM 3
static const int sig_ref_diff_offset[SIG_REF_DIFF_OFFSET_NUM][2] = {
{ 1, 1 }, { 0, 2 }, { 2, 0 }
};
static const int sig_ref_diff_offset_vert[SIG_REF_DIFF_OFFSET_NUM][2] = {
{ 2, 0 }, { 3, 0 }, { 4, 0 }
};
static const int sig_ref_diff_offset_horiz[SIG_REF_DIFF_OFFSET_NUM][2] = {
{ 0, 2 }, { 0, 3 }, { 0, 4 }
};
static const uint8_t clip_max3[256] = {
0, 1, 2, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3,
3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3,
@ -658,4 +421,4 @@ static INLINE void get_txb_ctx(const BLOCK_SIZE plane_bsize,
void av1_init_lv_map(AV1_COMMON *cm);
#endif // AV1_COMMON_TXB_COMMON_H_
#endif // AOM_AV1_COMMON_TXB_COMMON_H_

View file

@ -562,7 +562,7 @@ static int64_t highbd_warp_error(
const int error_bsize_h = AOMMIN(p_height, WARP_ERROR_BLOCK);
uint16_t tmp[WARP_ERROR_BLOCK * WARP_ERROR_BLOCK];
ConvolveParams conv_params = get_conv_params(0, 0, 0, bd);
ConvolveParams conv_params = get_conv_params(0, 0, bd);
conv_params.use_jnt_comp_avg = 0;
for (int i = p_row; i < p_row + p_height; i += WARP_ERROR_BLOCK) {
for (int j = p_col; j < p_col + p_width; j += WARP_ERROR_BLOCK) {
@ -845,7 +845,7 @@ static int64_t warp_error(WarpedMotionParams *wm, const uint8_t *const ref,
int error_bsize_w = AOMMIN(p_width, WARP_ERROR_BLOCK);
int error_bsize_h = AOMMIN(p_height, WARP_ERROR_BLOCK);
uint8_t tmp[WARP_ERROR_BLOCK * WARP_ERROR_BLOCK];
ConvolveParams conv_params = get_conv_params(0, 0, 0, 8);
ConvolveParams conv_params = get_conv_params(0, 0, 8);
conv_params.use_jnt_comp_avg = 0;
for (int i = p_row; i < p_row + p_height; i += WARP_ERROR_BLOCK) {

View file

@ -9,8 +9,8 @@
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AV1_COMMON_WARPED_MOTION_H_
#define AV1_COMMON_WARPED_MOTION_H_
#ifndef AOM_AV1_COMMON_WARPED_MOTION_H_
#define AOM_AV1_COMMON_WARPED_MOTION_H_
#include <stdio.h>
#include <stdlib.h>
@ -92,4 +92,4 @@ int find_projection(int np, int *pts1, int *pts2, BLOCK_SIZE bsize, int mvy,
int mi_col);
int get_shear_params(WarpedMotionParams *wm);
#endif // AV1_COMMON_WARPED_MOTION_H_
#endif // AOM_AV1_COMMON_WARPED_MOTION_H_

View file

@ -14,7 +14,6 @@
#include "config/aom_dsp_rtcd.h"
#include "aom_dsp/aom_convolve.h"
#include "aom_dsp/aom_dsp_common.h"
#include "aom_dsp/aom_filter.h"
#include "av1/common/convolve.h"

View file

@ -18,6 +18,12 @@
#include "av1/common/x86/av1_inv_txfm_avx2.h"
#include "av1/common/x86/av1_inv_txfm_ssse3.h"
// TODO(venkatsanampudi@ittiam.com): move this to header file
// Sqrt2, Sqrt2^2, Sqrt2^3, Sqrt2^4, Sqrt2^5
static int32_t NewSqrt2list[TX_SIZES] = { 5793, 2 * 4096, 2 * 5793, 4 * 4096,
4 * 5793 };
static INLINE void idct16_stage5_avx2(__m256i *x1, const int32_t *cospi,
const __m256i _r, int8_t cos_bit) {
const __m256i cospi_m32_p32 = pair_set_w16_epi16(-cospi[32], cospi[32]);

View file

@ -8,8 +8,8 @@
* Media Patent License 1.0 was not distributed with this source code in the
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AV1_COMMON_X86_AV1_INV_TXFM_AVX2_H_
#define AV1_COMMON_X86_AV1_INV_TXFM_AVX2_H_
#ifndef AOM_AV1_COMMON_X86_AV1_INV_TXFM_AVX2_H_
#define AOM_AV1_COMMON_X86_AV1_INV_TXFM_AVX2_H_
#include <immintrin.h>
@ -68,4 +68,4 @@ void av1_lowbd_inv_txfm2d_add_avx2(const int32_t *input, uint8_t *output,
}
#endif
#endif // AV1_COMMON_X86_AV1_INV_TXFM_AVX2_H_
#endif // AOM_AV1_COMMON_X86_AV1_INV_TXFM_AVX2_H_

View file

@ -16,6 +16,12 @@
#include "av1/common/x86/av1_inv_txfm_ssse3.h"
#include "av1/common/x86/av1_txfm_sse2.h"
// TODO(venkatsanampudi@ittiam.com): move this to header file
// Sqrt2, Sqrt2^2, Sqrt2^3, Sqrt2^4, Sqrt2^5
static int32_t NewSqrt2list[TX_SIZES] = { 5793, 2 * 4096, 2 * 5793, 4 * 4096,
4 * 5793 };
// TODO(binpengsmail@gmail.com): replace some for loop with do {} while
static void idct4_new_sse2(const __m128i *input, __m128i *output,

View file

@ -8,8 +8,8 @@
* Media Patent License 1.0 was not distributed with this source code in the
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AV1_COMMON_X86_AV1_INV_TXFM_SSSE3_H_
#define AV1_COMMON_X86_AV1_INV_TXFM_SSSE3_H_
#ifndef AOM_AV1_COMMON_X86_AV1_INV_TXFM_SSSE3_H_
#define AOM_AV1_COMMON_X86_AV1_INV_TXFM_SSSE3_H_
#include <emmintrin.h> // SSE2
#include <tmmintrin.h> // SSSE3
@ -94,10 +94,6 @@ static const ITX_TYPE_1D hitx_1d_tab[TX_TYPES] = {
IIDENTITY_1D, IADST_1D, IIDENTITY_1D, IFLIPADST_1D,
};
// Sqrt2, Sqrt2^2, Sqrt2^3, Sqrt2^4, Sqrt2^5
static int32_t NewSqrt2list[TX_SIZES] = { 5793, 2 * 4096, 2 * 5793, 4 * 4096,
4 * 5793 };
DECLARE_ALIGNED(16, static const int16_t, av1_eob_to_eobxy_8x8_default[8]) = {
0x0707, 0x0707, 0x0707, 0x0707, 0x0707, 0x0707, 0x0707, 0x0707,
};
@ -233,4 +229,4 @@ void av1_lowbd_inv_txfm2d_add_ssse3(const int32_t *input, uint8_t *output,
} // extern "C"
#endif
#endif // AV1_COMMON_X86_AV1_INV_TXFM_SSSE3_H_
#endif // AOM_AV1_COMMON_X86_AV1_INV_TXFM_SSSE3_H_

View file

@ -8,8 +8,8 @@
* Media Patent License 1.0 was not distributed with this source code in the
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AV1_COMMON_X86_AV1_TXFM_SSE2_H_
#define AV1_COMMON_X86_AV1_TXFM_SSE2_H_
#ifndef AOM_AV1_COMMON_X86_AV1_TXFM_SSE2_H_
#define AOM_AV1_COMMON_X86_AV1_TXFM_SSE2_H_
#include <emmintrin.h> // SSE2
@ -314,4 +314,4 @@ typedef struct {
#ifdef __cplusplus
}
#endif // __cplusplus
#endif // AV1_COMMON_X86_AV1_TXFM_SSE2_H_
#endif // AOM_AV1_COMMON_X86_AV1_TXFM_SSE2_H_

View file

@ -9,8 +9,8 @@
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AV1_TXFM_SSE4_H_
#define AV1_TXFM_SSE4_H_
#ifndef AOM_AV1_COMMON_X86_AV1_TXFM_SSE4_H_
#define AOM_AV1_COMMON_X86_AV1_TXFM_SSE4_H_
#include <smmintrin.h>
@ -45,8 +45,9 @@ static INLINE void av1_round_shift_array_32_sse4_1(__m128i *input,
static INLINE void av1_round_shift_rect_array_32_sse4_1(__m128i *input,
__m128i *output,
const int size,
const int bit) {
const __m128i sqrt2 = _mm_set1_epi32(NewSqrt2);
const int bit,
const int val) {
const __m128i sqrt2 = _mm_set1_epi32(val);
if (bit > 0) {
int i;
for (i = 0; i < size; i++) {
@ -68,4 +69,4 @@ static INLINE void av1_round_shift_rect_array_32_sse4_1(__m128i *input,
}
#endif
#endif // AV1_TXFM_SSE4_H_
#endif // AOM_AV1_COMMON_X86_AV1_TXFM_SSE4_H_

View file

@ -9,6 +9,9 @@
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef AOM_AV1_COMMON_X86_CFL_SIMD_H_
#define AOM_AV1_COMMON_X86_CFL_SIMD_H_
#include "av1/common/blockd.h"
// SSSE3 version is optimal for with == 4, we reuse them in AVX2
@ -236,3 +239,5 @@ void predict_hbd_16x16_ssse3(const int16_t *pred_buf_q3, uint16_t *dst,
int dst_stride, int alpha_q3, int bd);
void predict_hbd_16x32_ssse3(const int16_t *pred_buf_q3, uint16_t *dst,
int dst_stride, int alpha_q3, int bd);
#endif // AOM_AV1_COMMON_X86_CFL_SIMD_H_

View file

@ -11,10 +11,8 @@
#include <immintrin.h>
#include "config/aom_dsp_rtcd.h"
#include "config/av1_rtcd.h"
#include "aom_dsp/aom_convolve.h"
#include "aom_dsp/x86/convolve_avx2.h"
#include "aom_dsp/x86/convolve_common_intrin.h"
#include "aom_dsp/aom_dsp_common.h"

View file

@ -11,9 +11,8 @@
#include <emmintrin.h>
#include "config/aom_dsp_rtcd.h"
#include "config/av1_rtcd.h"
#include "aom_dsp/aom_convolve.h"
#include "aom_dsp/aom_dsp_common.h"
#include "aom_dsp/aom_filter.h"
#include "aom_dsp/x86/convolve_sse2.h"

View file

@ -11,9 +11,8 @@
#include <emmintrin.h>
#include "config/aom_dsp_rtcd.h"
#include "config/av1_rtcd.h"
#include "aom_dsp/aom_convolve.h"
#include "aom_dsp/aom_dsp_common.h"
#include "aom_dsp/aom_filter.h"
#include "aom_dsp/x86/convolve_common_intrin.h"
@ -76,8 +75,8 @@ static INLINE __m128i convolve_hi_y(const __m128i *const s,
return convolve(ss, coeffs);
}
void av1_convolve_y_sr_sse2(const uint8_t *src, int src_stride,
const uint8_t *dst, int dst_stride, int w, int h,
void av1_convolve_y_sr_sse2(const uint8_t *src, int src_stride, uint8_t *dst,
int dst_stride, int w, int h,
const InterpFilterParams *filter_params_x,
const InterpFilterParams *filter_params_y,
const int subpel_x_q4, const int subpel_y_q4,
@ -237,8 +236,8 @@ void av1_convolve_y_sr_sse2(const uint8_t *src, int src_stride,
}
}
void av1_convolve_x_sr_sse2(const uint8_t *src, int src_stride,
const uint8_t *dst, int dst_stride, int w, int h,
void av1_convolve_x_sr_sse2(const uint8_t *src, int src_stride, uint8_t *dst,
int dst_stride, int w, int h,
const InterpFilterParams *filter_params_x,
const InterpFilterParams *filter_params_y,
const int subpel_x_q4, const int subpel_y_q4,

View file

@ -14,7 +14,6 @@
#include "config/aom_dsp_rtcd.h"
#include "aom_dsp/aom_convolve.h"
#include "aom_dsp/x86/convolve_avx2.h"
#include "aom_dsp/x86/synonyms.h"
#include "aom_dsp/aom_dsp_common.h"

View file

@ -15,7 +15,6 @@
#include "config/aom_dsp_rtcd.h"
#include "aom_dsp/aom_convolve.h"
#include "aom_dsp/aom_dsp_common.h"
#include "aom_dsp/aom_filter.h"
#include "aom_dsp/x86/convolve_sse2.h"

View file

@ -14,7 +14,6 @@
#include "config/aom_dsp_rtcd.h"
#include "aom_dsp/aom_convolve.h"
#include "aom_dsp/aom_dsp_common.h"
#include "aom_dsp/aom_filter.h"
#include "aom_dsp/x86/convolve_sse2.h"

File diff suppressed because it is too large Load diff

File diff suppressed because it is too large Load diff

View file

@ -14,7 +14,6 @@
#include "config/aom_dsp_rtcd.h"
#include "aom_dsp/aom_convolve.h"
#include "aom_dsp/x86/convolve_avx2.h"
#include "aom_dsp/x86/convolve_common_intrin.h"
#include "aom_dsp/x86/convolve_sse4_1.h"

View file

@ -9,8 +9,8 @@
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
*/
#ifndef _HIGHBD_TXFM_UTILITY_SSE4_H
#define _HIGHBD_TXFM_UTILITY_SSE4_H
#ifndef AOM_AV1_COMMON_X86_HIGHBD_TXFM_UTILITY_SSE4_H_
#define AOM_AV1_COMMON_X86_HIGHBD_TXFM_UTILITY_SSE4_H_
#include <smmintrin.h> /* SSE4.1 */
@ -75,6 +75,17 @@ static INLINE void transpose_16x16(const __m128i *in, __m128i *out) {
out[63]);
}
static INLINE void transpose_32x32(const __m128i *input, __m128i *output) {
for (int j = 0; j < 8; j++) {
for (int i = 0; i < 8; i++) {
TRANSPOSE_4X4(input[i * 32 + j + 0], input[i * 32 + j + 8],
input[i * 32 + j + 16], input[i * 32 + j + 24],
output[j * 32 + i + 0], output[j * 32 + i + 8],
output[j * 32 + i + 16], output[j * 32 + i + 24]);
}
}
}
// Note:
// rounding = 1 << (bit - 1)
static INLINE __m128i half_btf_sse4_1(const __m128i *w0, const __m128i *n0,
@ -100,4 +111,15 @@ static INLINE __m128i half_btf_0_sse4_1(const __m128i *w0, const __m128i *n0,
return x;
}
#endif // _HIGHBD_TXFM_UTILITY_SSE4_H
typedef void (*transform_1d_sse4_1)(__m128i *in, __m128i *out, int bit,
int do_cols, int bd, int out_shift);
typedef void (*fwd_transform_1d_sse4_1)(__m128i *in, __m128i *out, int bit,
const int num_cols);
void av1_highbd_inv_txfm2d_add_universe_sse4_1(const int32_t *input,
uint8_t *output, int stride,
TX_TYPE tx_type, TX_SIZE tx_size,
int eob, const int bd);
#endif // AOM_AV1_COMMON_X86_HIGHBD_TXFM_UTILITY_SSE4_H_

View file

@ -19,10 +19,21 @@ static const uint8_t warp_highbd_arrange_bytes[16] = {
0, 2, 4, 6, 8, 10, 12, 14, 1, 3, 5, 7, 9, 11, 13, 15
};
static INLINE void horizontal_filter(__m128i src, __m128i src2, __m128i *tmp,
int sx, int alpha, int k,
const int offset_bits_horiz,
const int reduce_bits_horiz) {
static const uint8_t highbd_shuffle_alpha0_mask0[16] = {
0, 1, 2, 3, 0, 1, 2, 3, 0, 1, 2, 3, 0, 1, 2, 3
};
static const uint8_t highbd_shuffle_alpha0_mask1[16] = {
4, 5, 6, 7, 4, 5, 6, 7, 4, 5, 6, 7, 4, 5, 6, 7
};
static const uint8_t highbd_shuffle_alpha0_mask2[16] = {
8, 9, 10, 11, 8, 9, 10, 11, 8, 9, 10, 11, 8, 9, 10, 11
};
static const uint8_t highbd_shuffle_alpha0_mask3[16] = {
12, 13, 14, 15, 12, 13, 14, 15, 12, 13, 14, 15, 12, 13, 14, 15
};
static INLINE void highbd_prepare_horizontal_filter_coeff(int alpha, int sx,
__m128i *coeff) {
// Filter even-index pixels
const __m128i tmp_0 = _mm_loadu_si128(
(__m128i *)(warped_filter + ((sx + 0 * alpha) >> WARPEDDIFF_PREC_BITS)));
@ -43,27 +54,13 @@ static INLINE void horizontal_filter(__m128i src, __m128i src2, __m128i *tmp,
const __m128i tmp_14 = _mm_unpackhi_epi32(tmp_4, tmp_6);
// coeffs 0 1 0 1 0 1 0 1 for pixels 0, 2, 4, 6
const __m128i coeff_0 = _mm_unpacklo_epi64(tmp_8, tmp_10);
coeff[0] = _mm_unpacklo_epi64(tmp_8, tmp_10);
// coeffs 2 3 2 3 2 3 2 3 for pixels 0, 2, 4, 6
const __m128i coeff_2 = _mm_unpackhi_epi64(tmp_8, tmp_10);
coeff[2] = _mm_unpackhi_epi64(tmp_8, tmp_10);
// coeffs 4 5 4 5 4 5 4 5 for pixels 0, 2, 4, 6
const __m128i coeff_4 = _mm_unpacklo_epi64(tmp_12, tmp_14);
coeff[4] = _mm_unpacklo_epi64(tmp_12, tmp_14);
// coeffs 6 7 6 7 6 7 6 7 for pixels 0, 2, 4, 6
const __m128i coeff_6 = _mm_unpackhi_epi64(tmp_12, tmp_14);
const __m128i round_const = _mm_set1_epi32((1 << offset_bits_horiz) +
((1 << reduce_bits_horiz) >> 1));
// Calculate filtered results
const __m128i res_0 = _mm_madd_epi16(src, coeff_0);
const __m128i res_2 = _mm_madd_epi16(_mm_alignr_epi8(src2, src, 4), coeff_2);
const __m128i res_4 = _mm_madd_epi16(_mm_alignr_epi8(src2, src, 8), coeff_4);
const __m128i res_6 = _mm_madd_epi16(_mm_alignr_epi8(src2, src, 12), coeff_6);
__m128i res_even =
_mm_add_epi32(_mm_add_epi32(res_0, res_4), _mm_add_epi32(res_2, res_6));
res_even = _mm_sra_epi32(_mm_add_epi32(res_even, round_const),
_mm_cvtsi32_si128(reduce_bits_horiz));
coeff[6] = _mm_unpackhi_epi64(tmp_12, tmp_14);
// Filter odd-index pixels
const __m128i tmp_1 = _mm_loadu_si128(
@ -80,15 +77,63 @@ static INLINE void horizontal_filter(__m128i src, __m128i src2, __m128i *tmp,
const __m128i tmp_13 = _mm_unpackhi_epi32(tmp_1, tmp_3);
const __m128i tmp_15 = _mm_unpackhi_epi32(tmp_5, tmp_7);
const __m128i coeff_1 = _mm_unpacklo_epi64(tmp_9, tmp_11);
const __m128i coeff_3 = _mm_unpackhi_epi64(tmp_9, tmp_11);
const __m128i coeff_5 = _mm_unpacklo_epi64(tmp_13, tmp_15);
const __m128i coeff_7 = _mm_unpackhi_epi64(tmp_13, tmp_15);
coeff[1] = _mm_unpacklo_epi64(tmp_9, tmp_11);
coeff[3] = _mm_unpackhi_epi64(tmp_9, tmp_11);
coeff[5] = _mm_unpacklo_epi64(tmp_13, tmp_15);
coeff[7] = _mm_unpackhi_epi64(tmp_13, tmp_15);
}
const __m128i res_1 = _mm_madd_epi16(_mm_alignr_epi8(src2, src, 2), coeff_1);
const __m128i res_3 = _mm_madd_epi16(_mm_alignr_epi8(src2, src, 6), coeff_3);
const __m128i res_5 = _mm_madd_epi16(_mm_alignr_epi8(src2, src, 10), coeff_5);
const __m128i res_7 = _mm_madd_epi16(_mm_alignr_epi8(src2, src, 14), coeff_7);
static INLINE void highbd_prepare_horizontal_filter_coeff_alpha0(
int sx, __m128i *coeff) {
// Filter coeff
const __m128i tmp_0 = _mm_loadu_si128(
(__m128i *)(warped_filter + (sx >> WARPEDDIFF_PREC_BITS)));
coeff[0] = _mm_shuffle_epi8(
tmp_0, _mm_loadu_si128((__m128i *)highbd_shuffle_alpha0_mask0));
coeff[2] = _mm_shuffle_epi8(
tmp_0, _mm_loadu_si128((__m128i *)highbd_shuffle_alpha0_mask1));
coeff[4] = _mm_shuffle_epi8(
tmp_0, _mm_loadu_si128((__m128i *)highbd_shuffle_alpha0_mask2));
coeff[6] = _mm_shuffle_epi8(
tmp_0, _mm_loadu_si128((__m128i *)highbd_shuffle_alpha0_mask3));
coeff[1] = coeff[0];
coeff[3] = coeff[2];
coeff[5] = coeff[4];
coeff[7] = coeff[6];
}
static INLINE void highbd_filter_src_pixels(
const __m128i *src, const __m128i *src2, __m128i *tmp, __m128i *coeff,
const int offset_bits_horiz, const int reduce_bits_horiz, int k) {
const __m128i src_1 = *src;
const __m128i src2_1 = *src2;
const __m128i round_const = _mm_set1_epi32((1 << offset_bits_horiz) +
((1 << reduce_bits_horiz) >> 1));
const __m128i res_0 = _mm_madd_epi16(src_1, coeff[0]);
const __m128i res_2 =
_mm_madd_epi16(_mm_alignr_epi8(src2_1, src_1, 4), coeff[2]);
const __m128i res_4 =
_mm_madd_epi16(_mm_alignr_epi8(src2_1, src_1, 8), coeff[4]);
const __m128i res_6 =
_mm_madd_epi16(_mm_alignr_epi8(src2_1, src_1, 12), coeff[6]);
__m128i res_even =
_mm_add_epi32(_mm_add_epi32(res_0, res_4), _mm_add_epi32(res_2, res_6));
res_even = _mm_sra_epi32(_mm_add_epi32(res_even, round_const),
_mm_cvtsi32_si128(reduce_bits_horiz));
const __m128i res_1 =
_mm_madd_epi16(_mm_alignr_epi8(src2_1, src_1, 2), coeff[1]);
const __m128i res_3 =
_mm_madd_epi16(_mm_alignr_epi8(src2_1, src_1, 6), coeff[3]);
const __m128i res_5 =
_mm_madd_epi16(_mm_alignr_epi8(src2_1, src_1, 10), coeff[5]);
const __m128i res_7 =
_mm_madd_epi16(_mm_alignr_epi8(src2_1, src_1, 14), coeff[7]);
__m128i res_odd =
_mm_add_epi32(_mm_add_epi32(res_1, res_5), _mm_add_epi32(res_3, res_7));
@ -101,6 +146,145 @@ static INLINE void horizontal_filter(__m128i src, __m128i src2, __m128i *tmp,
tmp[k + 7] = _mm_packs_epi32(res_even, res_odd);
}
static INLINE void highbd_horiz_filter(const __m128i *src, const __m128i *src2,
__m128i *tmp, int sx, int alpha, int k,
const int offset_bits_horiz,
const int reduce_bits_horiz) {
__m128i coeff[8];
highbd_prepare_horizontal_filter_coeff(alpha, sx, coeff);
highbd_filter_src_pixels(src, src2, tmp, coeff, offset_bits_horiz,
reduce_bits_horiz, k);
}
static INLINE void highbd_warp_horizontal_filter_alpha0_beta0(
const uint16_t *ref, __m128i *tmp, int stride, int32_t ix4, int32_t iy4,
int32_t sx4, int alpha, int beta, int p_height, int height, int i,
const int offset_bits_horiz, const int reduce_bits_horiz) {
(void)beta;
(void)alpha;
int k;
__m128i coeff[8];
highbd_prepare_horizontal_filter_coeff_alpha0(sx4, coeff);
for (k = -7; k < AOMMIN(8, p_height - i); ++k) {
int iy = iy4 + k;
if (iy < 0)
iy = 0;
else if (iy > height - 1)
iy = height - 1;
// Load source pixels
const __m128i src =
_mm_loadu_si128((__m128i *)(ref + iy * stride + ix4 - 7));
const __m128i src2 =
_mm_loadu_si128((__m128i *)(ref + iy * stride + ix4 + 1));
highbd_filter_src_pixels(&src, &src2, tmp, coeff, offset_bits_horiz,
reduce_bits_horiz, k);
}
}
static INLINE void highbd_warp_horizontal_filter_alpha0(
const uint16_t *ref, __m128i *tmp, int stride, int32_t ix4, int32_t iy4,
int32_t sx4, int alpha, int beta, int p_height, int height, int i,
const int offset_bits_horiz, const int reduce_bits_horiz) {
(void)alpha;
int k;
for (k = -7; k < AOMMIN(8, p_height - i); ++k) {
int iy = iy4 + k;
if (iy < 0)
iy = 0;
else if (iy > height - 1)
iy = height - 1;
int sx = sx4 + beta * (k + 4);
// Load source pixels
const __m128i src =
_mm_loadu_si128((__m128i *)(ref + iy * stride + ix4 - 7));
const __m128i src2 =
_mm_loadu_si128((__m128i *)(ref + iy * stride + ix4 + 1));
__m128i coeff[8];
highbd_prepare_horizontal_filter_coeff_alpha0(sx, coeff);
highbd_filter_src_pixels(&src, &src2, tmp, coeff, offset_bits_horiz,
reduce_bits_horiz, k);
}
}
static INLINE void highbd_warp_horizontal_filter_beta0(
const uint16_t *ref, __m128i *tmp, int stride, int32_t ix4, int32_t iy4,
int32_t sx4, int alpha, int beta, int p_height, int height, int i,
const int offset_bits_horiz, const int reduce_bits_horiz) {
(void)beta;
int k;
__m128i coeff[8];
highbd_prepare_horizontal_filter_coeff(alpha, sx4, coeff);
for (k = -7; k < AOMMIN(8, p_height - i); ++k) {
int iy = iy4 + k;
if (iy < 0)
iy = 0;
else if (iy > height - 1)
iy = height - 1;
// Load source pixels
const __m128i src =
_mm_loadu_si128((__m128i *)(ref + iy * stride + ix4 - 7));
const __m128i src2 =
_mm_loadu_si128((__m128i *)(ref + iy * stride + ix4 + 1));
highbd_filter_src_pixels(&src, &src2, tmp, coeff, offset_bits_horiz,
reduce_bits_horiz, k);
}
}
static INLINE void highbd_warp_horizontal_filter(
const uint16_t *ref, __m128i *tmp, int stride, int32_t ix4, int32_t iy4,
int32_t sx4, int alpha, int beta, int p_height, int height, int i,
const int offset_bits_horiz, const int reduce_bits_horiz) {
int k;
for (k = -7; k < AOMMIN(8, p_height - i); ++k) {
int iy = iy4 + k;
if (iy < 0)
iy = 0;
else if (iy > height - 1)
iy = height - 1;
int sx = sx4 + beta * (k + 4);
// Load source pixels
const __m128i src =
_mm_loadu_si128((__m128i *)(ref + iy * stride + ix4 - 7));
const __m128i src2 =
_mm_loadu_si128((__m128i *)(ref + iy * stride + ix4 + 1));
highbd_horiz_filter(&src, &src2, tmp, sx, alpha, k, offset_bits_horiz,
reduce_bits_horiz);
}
}
static INLINE void highbd_prepare_warp_horizontal_filter(
const uint16_t *ref, __m128i *tmp, int stride, int32_t ix4, int32_t iy4,
int32_t sx4, int alpha, int beta, int p_height, int height, int i,
const int offset_bits_horiz, const int reduce_bits_horiz) {
if (alpha == 0 && beta == 0)
highbd_warp_horizontal_filter_alpha0_beta0(
ref, tmp, stride, ix4, iy4, sx4, alpha, beta, p_height, height, i,
offset_bits_horiz, reduce_bits_horiz);
else if (alpha == 0 && beta != 0)
highbd_warp_horizontal_filter_alpha0(ref, tmp, stride, ix4, iy4, sx4, alpha,
beta, p_height, height, i,
offset_bits_horiz, reduce_bits_horiz);
else if (alpha != 0 && beta == 0)
highbd_warp_horizontal_filter_beta0(ref, tmp, stride, ix4, iy4, sx4, alpha,
beta, p_height, height, i,
offset_bits_horiz, reduce_bits_horiz);
else
highbd_warp_horizontal_filter(ref, tmp, stride, ix4, iy4, sx4, alpha, beta,
p_height, height, i, offset_bits_horiz,
reduce_bits_horiz);
}
void av1_highbd_warp_affine_sse4_1(const int32_t *mat, const uint16_t *ref,
int width, int height, int stride,
uint16_t *pred, int p_col, int p_row,
@ -247,27 +431,13 @@ void av1_highbd_warp_affine_sse4_1(const int32_t *mat, const uint16_t *ref,
const __m128i src_padded = _mm_unpacklo_epi8(src_lo, src_hi);
const __m128i src2_padded = _mm_unpackhi_epi8(src_lo, src_hi);
horizontal_filter(src_padded, src2_padded, tmp, sx, alpha, k,
offset_bits_horiz, reduce_bits_horiz);
highbd_horiz_filter(&src_padded, &src2_padded, tmp, sx, alpha, k,
offset_bits_horiz, reduce_bits_horiz);
}
} else {
for (k = -7; k < AOMMIN(8, p_height - i); ++k) {
int iy = iy4 + k;
if (iy < 0)
iy = 0;
else if (iy > height - 1)
iy = height - 1;
int sx = sx4 + beta * (k + 4);
// Load source pixels
const __m128i src =
_mm_loadu_si128((__m128i *)(ref + iy * stride + ix4 - 7));
const __m128i src2 =
_mm_loadu_si128((__m128i *)(ref + iy * stride + ix4 + 1));
horizontal_filter(src, src2, tmp, sx, alpha, k, offset_bits_horiz,
reduce_bits_horiz);
}
highbd_prepare_warp_horizontal_filter(
ref, tmp, stride, ix4, iy4, sx4, alpha, beta, p_height, height, i,
offset_bits_horiz, reduce_bits_horiz);
}
// Vertical filter

View file

@ -13,7 +13,6 @@
#include "config/aom_dsp_rtcd.h"
#include "aom_dsp/aom_convolve.h"
#include "aom_dsp/x86/convolve_avx2.h"
#include "aom_dsp/x86/convolve_common_intrin.h"
#include "aom_dsp/x86/convolve_sse4_1.h"
@ -21,6 +20,21 @@
#include "aom_dsp/aom_filter.h"
#include "av1/common/convolve.h"
static INLINE __m256i unpack_weights_avx2(ConvolveParams *conv_params) {
const int w0 = conv_params->fwd_offset;
const int w1 = conv_params->bck_offset;
const __m256i wt0 = _mm256_set1_epi16(w0);
const __m256i wt1 = _mm256_set1_epi16(w1);
const __m256i wt = _mm256_unpacklo_epi16(wt0, wt1);
return wt;
}
static INLINE __m256i load_line2_avx2(const void *a, const void *b) {
return _mm256_permute2x128_si256(
_mm256_castsi128_si256(_mm_loadu_si128((__m128i *)a)),
_mm256_castsi128_si256(_mm_loadu_si128((__m128i *)b)), 0x20);
}
void av1_jnt_convolve_x_avx2(const uint8_t *src, int src_stride, uint8_t *dst0,
int dst_stride0, int w, int h,
const InterpFilterParams *filter_params_x,
@ -34,11 +48,7 @@ void av1_jnt_convolve_x_avx2(const uint8_t *src, int src_stride, uint8_t *dst0,
const int fo_horiz = filter_params_x->taps / 2 - 1;
const uint8_t *const src_ptr = src - fo_horiz;
const int bits = FILTER_BITS - conv_params->round_1;
const int w0 = conv_params->fwd_offset;
const int w1 = conv_params->bck_offset;
const __m256i wt0 = _mm256_set1_epi16(w0);
const __m256i wt1 = _mm256_set1_epi16(w1);
const __m256i wt = _mm256_unpacklo_epi16(wt0, wt1);
const __m256i wt = unpack_weights_avx2(conv_params);
const int do_average = conv_params->do_average;
const int use_jnt_comp_avg = conv_params->use_jnt_comp_avg;
const int offset_0 =
@ -68,13 +78,11 @@ void av1_jnt_convolve_x_avx2(const uint8_t *src, int src_stride, uint8_t *dst0,
(void)subpel_y_q4;
for (i = 0; i < h; i += 2) {
const uint8_t *src_data = src_ptr + i * src_stride;
CONV_BUF_TYPE *dst_data = dst + i * dst_stride;
for (j = 0; j < w; j += 8) {
const __m256i data = _mm256_permute2x128_si256(
_mm256_castsi128_si256(
_mm_loadu_si128((__m128i *)(&src_ptr[i * src_stride + j]))),
_mm256_castsi128_si256(_mm_loadu_si128(
(__m128i *)(&src_ptr[i * src_stride + j + src_stride]))),
0x20);
const __m256i data =
load_line2_avx2(&src_data[j], &src_data[j + src_stride]);
__m256i res = convolve_lowbd_x(data, coeffs, filt);
@ -86,13 +94,8 @@ void av1_jnt_convolve_x_avx2(const uint8_t *src, int src_stride, uint8_t *dst0,
// Accumulate values into the destination buffer
if (do_average) {
const __m256i data_ref_0 = _mm256_permute2x128_si256(
_mm256_castsi128_si256(
_mm_loadu_si128((__m128i *)(&dst[i * dst_stride + j]))),
_mm256_castsi128_si256(_mm_loadu_si128(
(__m128i *)(&dst[i * dst_stride + j + dst_stride]))),
0x20);
const __m256i data_ref_0 =
load_line2_avx2(&dst_data[j], &dst_data[j + dst_stride]);
const __m256i comp_avg_res =
comp_avg(&data_ref_0, &res_unsigned, &wt, use_jnt_comp_avg);
@ -141,11 +144,7 @@ void av1_jnt_convolve_y_avx2(const uint8_t *src, int src_stride, uint8_t *dst0,
const __m256i round_const =
_mm256_set1_epi32((1 << conv_params->round_1) >> 1);
const __m128i round_shift = _mm_cvtsi32_si128(conv_params->round_1);
const int w0 = conv_params->fwd_offset;
const int w1 = conv_params->bck_offset;
const __m256i wt0 = _mm256_set1_epi16(w0);
const __m256i wt1 = _mm256_set1_epi16(w1);
const __m256i wt = _mm256_unpacklo_epi16(wt0, wt1);
const __m256i wt = unpack_weights_avx2(conv_params);
const int do_average = conv_params->do_average;
const int use_jnt_comp_avg = conv_params->use_jnt_comp_avg;
const int offset_0 =
@ -172,72 +171,35 @@ void av1_jnt_convolve_y_avx2(const uint8_t *src, int src_stride, uint8_t *dst0,
for (j = 0; j < w; j += 16) {
const uint8_t *data = &src_ptr[j];
__m256i src6;
// Load lines a and b. Line a to lower 128, line b to upper 128
const __m256i src_01a = _mm256_permute2x128_si256(
_mm256_castsi128_si256(
_mm_loadu_si128((__m128i *)(data + 0 * src_stride))),
_mm256_castsi128_si256(
_mm_loadu_si128((__m128i *)(data + 1 * src_stride))),
0x20);
const __m256i src_12a = _mm256_permute2x128_si256(
_mm256_castsi128_si256(
_mm_loadu_si128((__m128i *)(data + 1 * src_stride))),
_mm256_castsi128_si256(
_mm_loadu_si128((__m128i *)(data + 2 * src_stride))),
0x20);
const __m256i src_23a = _mm256_permute2x128_si256(
_mm256_castsi128_si256(
_mm_loadu_si128((__m128i *)(data + 2 * src_stride))),
_mm256_castsi128_si256(
_mm_loadu_si128((__m128i *)(data + 3 * src_stride))),
0x20);
const __m256i src_34a = _mm256_permute2x128_si256(
_mm256_castsi128_si256(
_mm_loadu_si128((__m128i *)(data + 3 * src_stride))),
_mm256_castsi128_si256(
_mm_loadu_si128((__m128i *)(data + 4 * src_stride))),
0x20);
const __m256i src_45a = _mm256_permute2x128_si256(
_mm256_castsi128_si256(
_mm_loadu_si128((__m128i *)(data + 4 * src_stride))),
_mm256_castsi128_si256(
_mm_loadu_si128((__m128i *)(data + 5 * src_stride))),
0x20);
src6 = _mm256_castsi128_si256(
_mm_loadu_si128((__m128i *)(data + 6 * src_stride)));
const __m256i src_56a = _mm256_permute2x128_si256(
_mm256_castsi128_si256(
_mm_loadu_si128((__m128i *)(data + 5 * src_stride))),
src6, 0x20);
s[0] = _mm256_unpacklo_epi8(src_01a, src_12a);
s[1] = _mm256_unpacklo_epi8(src_23a, src_34a);
s[2] = _mm256_unpacklo_epi8(src_45a, src_56a);
s[4] = _mm256_unpackhi_epi8(src_01a, src_12a);
s[5] = _mm256_unpackhi_epi8(src_23a, src_34a);
s[6] = _mm256_unpackhi_epi8(src_45a, src_56a);
{
__m256i src_ab[7];
__m256i src_a[7];
src_a[0] = _mm256_castsi128_si256(_mm_loadu_si128((__m128i *)data));
for (int kk = 0; kk < 6; ++kk) {
data += src_stride;
src_a[kk + 1] =
_mm256_castsi128_si256(_mm_loadu_si128((__m128i *)data));
src_ab[kk] = _mm256_permute2x128_si256(src_a[kk], src_a[kk + 1], 0x20);
}
src6 = src_a[6];
s[0] = _mm256_unpacklo_epi8(src_ab[0], src_ab[1]);
s[1] = _mm256_unpacklo_epi8(src_ab[2], src_ab[3]);
s[2] = _mm256_unpacklo_epi8(src_ab[4], src_ab[5]);
s[4] = _mm256_unpackhi_epi8(src_ab[0], src_ab[1]);
s[5] = _mm256_unpackhi_epi8(src_ab[2], src_ab[3]);
s[6] = _mm256_unpackhi_epi8(src_ab[4], src_ab[5]);
}
for (i = 0; i < h; i += 2) {
data = &src_ptr[i * src_stride + j];
const __m256i src_67a = _mm256_permute2x128_si256(
src6,
_mm256_castsi128_si256(
_mm_loadu_si128((__m128i *)(data + 7 * src_stride))),
0x20);
data = &src_ptr[(i + 7) * src_stride + j];
const __m256i src7 =
_mm256_castsi128_si256(_mm_loadu_si128((__m128i *)data));
const __m256i src_67a = _mm256_permute2x128_si256(src6, src7, 0x20);
src6 = _mm256_castsi128_si256(
_mm_loadu_si128((__m128i *)(data + 8 * src_stride)));
const __m256i src_78a = _mm256_permute2x128_si256(
_mm256_castsi128_si256(
_mm_loadu_si128((__m128i *)(data + 7 * src_stride))),
src6, 0x20);
_mm_loadu_si128((__m128i *)(data + src_stride)));
const __m256i src_78a = _mm256_permute2x128_si256(src7, src6, 0x20);
s[3] = _mm256_unpacklo_epi8(src_67a, src_78a);
s[7] = _mm256_unpackhi_epi8(src_67a, src_78a);
@ -266,13 +228,8 @@ void av1_jnt_convolve_y_avx2(const uint8_t *src, int src_stride, uint8_t *dst0,
if (w - j < 16) {
if (do_average) {
const __m256i data_ref_0 = _mm256_permute2x128_si256(
_mm256_castsi128_si256(
_mm_loadu_si128((__m128i *)(&dst[i * dst_stride + j]))),
_mm256_castsi128_si256(_mm_loadu_si128(
(__m128i *)(&dst[i * dst_stride + j + dst_stride]))),
0x20);
const __m256i data_ref_0 = load_line2_avx2(
&dst[i * dst_stride + j], &dst[i * dst_stride + j + dst_stride]);
const __m256i comp_avg_res =
comp_avg(&data_ref_0, &res_lo_unsigned, &wt, use_jnt_comp_avg);
@ -325,19 +282,12 @@ void av1_jnt_convolve_y_avx2(const uint8_t *src, int src_stride, uint8_t *dst0,
_mm256_add_epi16(res_hi_round, offset_const_2);
if (do_average) {
const __m256i data_ref_0_lo = _mm256_permute2x128_si256(
_mm256_castsi128_si256(
_mm_loadu_si128((__m128i *)(&dst[i * dst_stride + j]))),
_mm256_castsi128_si256(_mm_loadu_si128(
(__m128i *)(&dst[i * dst_stride + j + dst_stride]))),
0x20);
const __m256i data_ref_0_lo = load_line2_avx2(
&dst[i * dst_stride + j], &dst[i * dst_stride + j + dst_stride]);
const __m256i data_ref_0_hi = _mm256_permute2x128_si256(
_mm256_castsi128_si256(
_mm_loadu_si128((__m128i *)(&dst[i * dst_stride + j + 8]))),
_mm256_castsi128_si256(_mm_loadu_si128(
(__m128i *)(&dst[i * dst_stride + j + 8 + dst_stride]))),
0x20);
const __m256i data_ref_0_hi =
load_line2_avx2(&dst[i * dst_stride + j + 8],
&dst[i * dst_stride + j + 8 + dst_stride]);
const __m256i comp_avg_res_lo =
comp_avg(&data_ref_0_lo, &res_lo_unsigned, &wt, use_jnt_comp_avg);
@ -404,11 +354,7 @@ void av1_jnt_convolve_2d_avx2(const uint8_t *src, int src_stride, uint8_t *dst0,
const int fo_vert = filter_params_y->taps / 2 - 1;
const int fo_horiz = filter_params_x->taps / 2 - 1;
const uint8_t *const src_ptr = src - fo_vert * src_stride - fo_horiz;
const int w0 = conv_params->fwd_offset;
const int w1 = conv_params->bck_offset;
const __m256i wt0 = _mm256_set1_epi16(w0);
const __m256i wt1 = _mm256_set1_epi16(w1);
const __m256i wt = _mm256_unpacklo_epi16(wt0, wt1);
const __m256i wt = unpack_weights_avx2(conv_params);
const int do_average = conv_params->do_average;
const int use_jnt_comp_avg = conv_params->use_jnt_comp_avg;
const int offset_0 =
@ -442,15 +388,14 @@ void av1_jnt_convolve_2d_avx2(const uint8_t *src, int src_stride, uint8_t *dst0,
for (j = 0; j < w; j += 8) {
/* Horizontal filter */
{
const uint8_t *src_h = src_ptr + j;
for (i = 0; i < im_h; i += 2) {
__m256i data = _mm256_castsi128_si256(
_mm_loadu_si128((__m128i *)&src_ptr[(i * src_stride) + j]));
__m256i data =
_mm256_castsi128_si256(_mm_loadu_si128((__m128i *)src_h));
if (i + 1 < im_h)
data = _mm256_inserti128_si256(
data,
_mm_loadu_si128(
(__m128i *)&src_ptr[(i * src_stride) + j + src_stride]),
1);
data, _mm_loadu_si128((__m128i *)(src_h + src_stride)), 1);
src_h += (src_stride << 1);
__m256i res = convolve_lowbd_x(data, coeffs_x, filt);
res = _mm256_sra_epi16(_mm256_add_epi16(res, round_const_h),
@ -500,13 +445,9 @@ void av1_jnt_convolve_2d_avx2(const uint8_t *src, int src_stride, uint8_t *dst0,
const __m256i res_unsigned = _mm256_add_epi16(res_16b, offset_const);
if (do_average) {
const __m256i data_ref_0 = _mm256_permute2x128_si256(
_mm256_castsi128_si256(
_mm_loadu_si128((__m128i *)(&dst[i * dst_stride + j]))),
_mm256_castsi128_si256(_mm_loadu_si128(
(__m128i *)(&dst[i * dst_stride + j + dst_stride]))),
0x20);
const __m256i data_ref_0 =
load_line2_avx2(&dst[i * dst_stride + j],
&dst[i * dst_stride + j + dst_stride]);
const __m256i comp_avg_res =
comp_avg(&data_ref_0, &res_unsigned, &wt, use_jnt_comp_avg);
@ -534,12 +475,9 @@ void av1_jnt_convolve_2d_avx2(const uint8_t *src, int src_stride, uint8_t *dst0,
const __m256i res_unsigned = _mm256_add_epi16(res_16b, offset_const);
if (do_average) {
const __m256i data_ref_0 = _mm256_permute2x128_si256(
_mm256_castsi128_si256(
_mm_loadu_si128((__m128i *)(&dst[i * dst_stride + j]))),
_mm256_castsi128_si256(_mm_loadu_si128(
(__m128i *)(&dst[i * dst_stride + j + dst_stride]))),
0x20);
const __m256i data_ref_0 =
load_line2_avx2(&dst[i * dst_stride + j],
&dst[i * dst_stride + j + dst_stride]);
const __m256i comp_avg_res =
comp_avg(&data_ref_0, &res_unsigned, &wt, use_jnt_comp_avg);
@ -598,11 +536,7 @@ void av1_jnt_convolve_2d_copy_avx2(const uint8_t *src, int src_stride,
const __m128i left_shift = _mm_cvtsi32_si128(bits);
const int do_average = conv_params->do_average;
const int use_jnt_comp_avg = conv_params->use_jnt_comp_avg;
const int w0 = conv_params->fwd_offset;
const int w1 = conv_params->bck_offset;
const __m256i wt0 = _mm256_set1_epi16(w0);
const __m256i wt1 = _mm256_set1_epi16(w1);
const __m256i wt = _mm256_unpacklo_epi16(wt0, wt1);
const __m256i wt = unpack_weights_avx2(conv_params);
const __m256i zero = _mm256_setzero_si256();
const int offset_0 =
@ -663,13 +597,8 @@ void av1_jnt_convolve_2d_copy_avx2(const uint8_t *src, int src_stride,
// Accumulate values into the destination buffer
if (do_average) {
const __m256i data_ref_0 = _mm256_permute2x128_si256(
_mm256_castsi128_si256(
_mm_loadu_si128((__m128i *)(&dst[i * dst_stride + j]))),
_mm256_castsi128_si256(_mm_loadu_si128(
(__m128i *)(&dst[i * dst_stride + j + dst_stride]))),
0x20);
const __m256i data_ref_0 = load_line2_avx2(
&dst[i * dst_stride + j], &dst[i * dst_stride + j + dst_stride]);
const __m256i comp_avg_res =
comp_avg(&data_ref_0, &res_unsigned, &wt, use_jnt_comp_avg);

View file

@ -16,8 +16,504 @@
#include "aom/aom_integer.h"
#include "aom_dsp/blend.h"
#include "aom_dsp/x86/synonyms.h"
#include "aom_dsp/x86/synonyms_avx2.h"
#include "av1/common/blockd.h"
static INLINE __m256i calc_mask_avx2(const __m256i mask_base, const __m256i s0,
const __m256i s1) {
const __m256i diff = _mm256_abs_epi16(_mm256_sub_epi16(s0, s1));
return _mm256_abs_epi16(
_mm256_add_epi16(mask_base, _mm256_srli_epi16(diff, 4)));
// clamp(diff, 0, 64) can be skiped for diff is always in the range ( 38, 54)
}
void av1_build_compound_diffwtd_mask_avx2(uint8_t *mask,
DIFFWTD_MASK_TYPE mask_type,
const uint8_t *src0, int stride0,
const uint8_t *src1, int stride1,
int h, int w) {
const int mb = (mask_type == DIFFWTD_38_INV) ? AOM_BLEND_A64_MAX_ALPHA : 0;
const __m256i y_mask_base = _mm256_set1_epi16(38 - mb);
int i = 0;
if (4 == w) {
do {
const __m128i s0A = xx_loadl_32(src0);
const __m128i s0B = xx_loadl_32(src0 + stride0);
const __m128i s0C = xx_loadl_32(src0 + stride0 * 2);
const __m128i s0D = xx_loadl_32(src0 + stride0 * 3);
const __m128i s0AB = _mm_unpacklo_epi32(s0A, s0B);
const __m128i s0CD = _mm_unpacklo_epi32(s0C, s0D);
const __m128i s0ABCD = _mm_unpacklo_epi64(s0AB, s0CD);
const __m256i s0ABCD_w = _mm256_cvtepu8_epi16(s0ABCD);
const __m128i s1A = xx_loadl_32(src1);
const __m128i s1B = xx_loadl_32(src1 + stride1);
const __m128i s1C = xx_loadl_32(src1 + stride1 * 2);
const __m128i s1D = xx_loadl_32(src1 + stride1 * 3);
const __m128i s1AB = _mm_unpacklo_epi32(s1A, s1B);
const __m128i s1CD = _mm_unpacklo_epi32(s1C, s1D);
const __m128i s1ABCD = _mm_unpacklo_epi64(s1AB, s1CD);
const __m256i s1ABCD_w = _mm256_cvtepu8_epi16(s1ABCD);
const __m256i m16 = calc_mask_avx2(y_mask_base, s0ABCD_w, s1ABCD_w);
const __m256i m8 = _mm256_packus_epi16(m16, _mm256_setzero_si256());
const __m128i x_m8 =
_mm256_castsi256_si128(_mm256_permute4x64_epi64(m8, 0xd8));
xx_storeu_128(mask, x_m8);
src0 += (stride0 << 2);
src1 += (stride1 << 2);
mask += 16;
i += 4;
} while (i < h);
} else if (8 == w) {
do {
const __m128i s0A = xx_loadl_64(src0);
const __m128i s0B = xx_loadl_64(src0 + stride0);
const __m128i s0C = xx_loadl_64(src0 + stride0 * 2);
const __m128i s0D = xx_loadl_64(src0 + stride0 * 3);
const __m256i s0AC_w = _mm256_cvtepu8_epi16(_mm_unpacklo_epi64(s0A, s0C));
const __m256i s0BD_w = _mm256_cvtepu8_epi16(_mm_unpacklo_epi64(s0B, s0D));
const __m128i s1A = xx_loadl_64(src1);
const __m128i s1B = xx_loadl_64(src1 + stride1);
const __m128i s1C = xx_loadl_64(src1 + stride1 * 2);
const __m128i s1D = xx_loadl_64(src1 + stride1 * 3);
const __m256i s1AB_w = _mm256_cvtepu8_epi16(_mm_unpacklo_epi64(s1A, s1C));
const __m256i s1CD_w = _mm256_cvtepu8_epi16(_mm_unpacklo_epi64(s1B, s1D));
const __m256i m16AC = calc_mask_avx2(y_mask_base, s0AC_w, s1AB_w);
const __m256i m16BD = calc_mask_avx2(y_mask_base, s0BD_w, s1CD_w);
const __m256i m8 = _mm256_packus_epi16(m16AC, m16BD);
yy_storeu_256(mask, m8);
src0 += stride0 << 2;
src1 += stride1 << 2;
mask += 32;
i += 4;
} while (i < h);
} else if (16 == w) {
do {
const __m128i s0A = xx_load_128(src0);
const __m128i s0B = xx_load_128(src0 + stride0);
const __m128i s1A = xx_load_128(src1);
const __m128i s1B = xx_load_128(src1 + stride1);
const __m256i s0AL = _mm256_cvtepu8_epi16(s0A);
const __m256i s0BL = _mm256_cvtepu8_epi16(s0B);
const __m256i s1AL = _mm256_cvtepu8_epi16(s1A);
const __m256i s1BL = _mm256_cvtepu8_epi16(s1B);
const __m256i m16AL = calc_mask_avx2(y_mask_base, s0AL, s1AL);
const __m256i m16BL = calc_mask_avx2(y_mask_base, s0BL, s1BL);
const __m256i m8 =
_mm256_permute4x64_epi64(_mm256_packus_epi16(m16AL, m16BL), 0xd8);
yy_storeu_256(mask, m8);
src0 += stride0 << 1;
src1 += stride1 << 1;
mask += 32;
i += 2;
} while (i < h);
} else {
do {
int j = 0;
do {
const __m256i s0 = yy_loadu_256(src0 + j);
const __m256i s1 = yy_loadu_256(src1 + j);
const __m256i s0L = _mm256_cvtepu8_epi16(_mm256_castsi256_si128(s0));
const __m256i s1L = _mm256_cvtepu8_epi16(_mm256_castsi256_si128(s1));
const __m256i s0H =
_mm256_cvtepu8_epi16(_mm256_extracti128_si256(s0, 1));
const __m256i s1H =
_mm256_cvtepu8_epi16(_mm256_extracti128_si256(s1, 1));
const __m256i m16L = calc_mask_avx2(y_mask_base, s0L, s1L);
const __m256i m16H = calc_mask_avx2(y_mask_base, s0H, s1H);
const __m256i m8 =
_mm256_permute4x64_epi64(_mm256_packus_epi16(m16L, m16H), 0xd8);
yy_storeu_256(mask + j, m8);
j += 32;
} while (j < w);
src0 += stride0;
src1 += stride1;
mask += w;
i += 1;
} while (i < h);
}
}
static INLINE __m256i calc_mask_d16_avx2(const __m256i *data_src0,
const __m256i *data_src1,
const __m256i *round_const,
const __m256i *mask_base_16,
const __m256i *clip_diff, int round) {
const __m256i diffa = _mm256_subs_epu16(*data_src0, *data_src1);
const __m256i diffb = _mm256_subs_epu16(*data_src1, *data_src0);
const __m256i diff = _mm256_max_epu16(diffa, diffb);
const __m256i diff_round =
_mm256_srli_epi16(_mm256_adds_epu16(diff, *round_const), round);
const __m256i diff_factor = _mm256_srli_epi16(diff_round, DIFF_FACTOR_LOG2);
const __m256i diff_mask = _mm256_adds_epi16(diff_factor, *mask_base_16);
const __m256i diff_clamp = _mm256_min_epi16(diff_mask, *clip_diff);
return diff_clamp;
}
static INLINE __m256i calc_mask_d16_inv_avx2(const __m256i *data_src0,
const __m256i *data_src1,
const __m256i *round_const,
const __m256i *mask_base_16,
const __m256i *clip_diff,
int round) {
const __m256i diffa = _mm256_subs_epu16(*data_src0, *data_src1);
const __m256i diffb = _mm256_subs_epu16(*data_src1, *data_src0);
const __m256i diff = _mm256_max_epu16(diffa, diffb);
const __m256i diff_round =
_mm256_srli_epi16(_mm256_adds_epu16(diff, *round_const), round);
const __m256i diff_factor = _mm256_srli_epi16(diff_round, DIFF_FACTOR_LOG2);
const __m256i diff_mask = _mm256_adds_epi16(diff_factor, *mask_base_16);
const __m256i diff_clamp = _mm256_min_epi16(diff_mask, *clip_diff);
const __m256i diff_const_16 = _mm256_sub_epi16(*clip_diff, diff_clamp);
return diff_const_16;
}
static INLINE void build_compound_diffwtd_mask_d16_avx2(
uint8_t *mask, const CONV_BUF_TYPE *src0, int src0_stride,
const CONV_BUF_TYPE *src1, int src1_stride, int h, int w, int shift) {
const int mask_base = 38;
const __m256i _r = _mm256_set1_epi16((1 << shift) >> 1);
const __m256i y38 = _mm256_set1_epi16(mask_base);
const __m256i y64 = _mm256_set1_epi16(AOM_BLEND_A64_MAX_ALPHA);
int i = 0;
if (w == 4) {
do {
const __m128i s0A = xx_loadl_64(src0);
const __m128i s0B = xx_loadl_64(src0 + src0_stride);
const __m128i s0C = xx_loadl_64(src0 + src0_stride * 2);
const __m128i s0D = xx_loadl_64(src0 + src0_stride * 3);
const __m128i s1A = xx_loadl_64(src1);
const __m128i s1B = xx_loadl_64(src1 + src1_stride);
const __m128i s1C = xx_loadl_64(src1 + src1_stride * 2);
const __m128i s1D = xx_loadl_64(src1 + src1_stride * 3);
const __m256i s0 = yy_set_m128i(_mm_unpacklo_epi64(s0C, s0D),
_mm_unpacklo_epi64(s0A, s0B));
const __m256i s1 = yy_set_m128i(_mm_unpacklo_epi64(s1C, s1D),
_mm_unpacklo_epi64(s1A, s1B));
const __m256i m16 = calc_mask_d16_avx2(&s0, &s1, &_r, &y38, &y64, shift);
const __m256i m8 = _mm256_packus_epi16(m16, _mm256_setzero_si256());
xx_storeu_128(mask,
_mm256_castsi256_si128(_mm256_permute4x64_epi64(m8, 0xd8)));
src0 += src0_stride << 2;
src1 += src1_stride << 2;
mask += 16;
i += 4;
} while (i < h);
} else if (w == 8) {
do {
const __m256i s0AB = yy_loadu2_128(src0 + src0_stride, src0);
const __m256i s0CD =
yy_loadu2_128(src0 + src0_stride * 3, src0 + src0_stride * 2);
const __m256i s1AB = yy_loadu2_128(src1 + src1_stride, src1);
const __m256i s1CD =
yy_loadu2_128(src1 + src1_stride * 3, src1 + src1_stride * 2);
const __m256i m16AB =
calc_mask_d16_avx2(&s0AB, &s1AB, &_r, &y38, &y64, shift);
const __m256i m16CD =
calc_mask_d16_avx2(&s0CD, &s1CD, &_r, &y38, &y64, shift);
const __m256i m8 = _mm256_packus_epi16(m16AB, m16CD);
yy_storeu_256(mask, _mm256_permute4x64_epi64(m8, 0xd8));
src0 += src0_stride << 2;
src1 += src1_stride << 2;
mask += 32;
i += 4;
} while (i < h);
} else if (w == 16) {
do {
const __m256i s0A = yy_loadu_256(src0);
const __m256i s0B = yy_loadu_256(src0 + src0_stride);
const __m256i s1A = yy_loadu_256(src1);
const __m256i s1B = yy_loadu_256(src1 + src1_stride);
const __m256i m16A =
calc_mask_d16_avx2(&s0A, &s1A, &_r, &y38, &y64, shift);
const __m256i m16B =
calc_mask_d16_avx2(&s0B, &s1B, &_r, &y38, &y64, shift);
const __m256i m8 = _mm256_packus_epi16(m16A, m16B);
yy_storeu_256(mask, _mm256_permute4x64_epi64(m8, 0xd8));
src0 += src0_stride << 1;
src1 += src1_stride << 1;
mask += 32;
i += 2;
} while (i < h);
} else if (w == 32) {
do {
const __m256i s0A = yy_loadu_256(src0);
const __m256i s0B = yy_loadu_256(src0 + 16);
const __m256i s1A = yy_loadu_256(src1);
const __m256i s1B = yy_loadu_256(src1 + 16);
const __m256i m16A =
calc_mask_d16_avx2(&s0A, &s1A, &_r, &y38, &y64, shift);
const __m256i m16B =
calc_mask_d16_avx2(&s0B, &s1B, &_r, &y38, &y64, shift);
const __m256i m8 = _mm256_packus_epi16(m16A, m16B);
yy_storeu_256(mask, _mm256_permute4x64_epi64(m8, 0xd8));
src0 += src0_stride;
src1 += src1_stride;
mask += 32;
i += 1;
} while (i < h);
} else if (w == 64) {
do {
const __m256i s0A = yy_loadu_256(src0);
const __m256i s0B = yy_loadu_256(src0 + 16);
const __m256i s0C = yy_loadu_256(src0 + 32);
const __m256i s0D = yy_loadu_256(src0 + 48);
const __m256i s1A = yy_loadu_256(src1);
const __m256i s1B = yy_loadu_256(src1 + 16);
const __m256i s1C = yy_loadu_256(src1 + 32);
const __m256i s1D = yy_loadu_256(src1 + 48);
const __m256i m16A =
calc_mask_d16_avx2(&s0A, &s1A, &_r, &y38, &y64, shift);
const __m256i m16B =
calc_mask_d16_avx2(&s0B, &s1B, &_r, &y38, &y64, shift);
const __m256i m16C =
calc_mask_d16_avx2(&s0C, &s1C, &_r, &y38, &y64, shift);
const __m256i m16D =
calc_mask_d16_avx2(&s0D, &s1D, &_r, &y38, &y64, shift);
const __m256i m8AB = _mm256_packus_epi16(m16A, m16B);
const __m256i m8CD = _mm256_packus_epi16(m16C, m16D);
yy_storeu_256(mask, _mm256_permute4x64_epi64(m8AB, 0xd8));
yy_storeu_256(mask + 32, _mm256_permute4x64_epi64(m8CD, 0xd8));
src0 += src0_stride;
src1 += src1_stride;
mask += 64;
i += 1;
} while (i < h);
} else {
do {
const __m256i s0A = yy_loadu_256(src0);
const __m256i s0B = yy_loadu_256(src0 + 16);
const __m256i s0C = yy_loadu_256(src0 + 32);
const __m256i s0D = yy_loadu_256(src0 + 48);
const __m256i s0E = yy_loadu_256(src0 + 64);
const __m256i s0F = yy_loadu_256(src0 + 80);
const __m256i s0G = yy_loadu_256(src0 + 96);
const __m256i s0H = yy_loadu_256(src0 + 112);
const __m256i s1A = yy_loadu_256(src1);
const __m256i s1B = yy_loadu_256(src1 + 16);
const __m256i s1C = yy_loadu_256(src1 + 32);
const __m256i s1D = yy_loadu_256(src1 + 48);
const __m256i s1E = yy_loadu_256(src1 + 64);
const __m256i s1F = yy_loadu_256(src1 + 80);
const __m256i s1G = yy_loadu_256(src1 + 96);
const __m256i s1H = yy_loadu_256(src1 + 112);
const __m256i m16A =
calc_mask_d16_avx2(&s0A, &s1A, &_r, &y38, &y64, shift);
const __m256i m16B =
calc_mask_d16_avx2(&s0B, &s1B, &_r, &y38, &y64, shift);
const __m256i m16C =
calc_mask_d16_avx2(&s0C, &s1C, &_r, &y38, &y64, shift);
const __m256i m16D =
calc_mask_d16_avx2(&s0D, &s1D, &_r, &y38, &y64, shift);
const __m256i m16E =
calc_mask_d16_avx2(&s0E, &s1E, &_r, &y38, &y64, shift);
const __m256i m16F =
calc_mask_d16_avx2(&s0F, &s1F, &_r, &y38, &y64, shift);
const __m256i m16G =
calc_mask_d16_avx2(&s0G, &s1G, &_r, &y38, &y64, shift);
const __m256i m16H =
calc_mask_d16_avx2(&s0H, &s1H, &_r, &y38, &y64, shift);
const __m256i m8AB = _mm256_packus_epi16(m16A, m16B);
const __m256i m8CD = _mm256_packus_epi16(m16C, m16D);
const __m256i m8EF = _mm256_packus_epi16(m16E, m16F);
const __m256i m8GH = _mm256_packus_epi16(m16G, m16H);
yy_storeu_256(mask, _mm256_permute4x64_epi64(m8AB, 0xd8));
yy_storeu_256(mask + 32, _mm256_permute4x64_epi64(m8CD, 0xd8));
yy_storeu_256(mask + 64, _mm256_permute4x64_epi64(m8EF, 0xd8));
yy_storeu_256(mask + 96, _mm256_permute4x64_epi64(m8GH, 0xd8));
src0 += src0_stride;
src1 += src1_stride;
mask += 128;
i += 1;
} while (i < h);
}
}
static INLINE void build_compound_diffwtd_mask_d16_inv_avx2(
uint8_t *mask, const CONV_BUF_TYPE *src0, int src0_stride,
const CONV_BUF_TYPE *src1, int src1_stride, int h, int w, int shift) {
const int mask_base = 38;
const __m256i _r = _mm256_set1_epi16((1 << shift) >> 1);
const __m256i y38 = _mm256_set1_epi16(mask_base);
const __m256i y64 = _mm256_set1_epi16(AOM_BLEND_A64_MAX_ALPHA);
int i = 0;
if (w == 4) {
do {
const __m128i s0A = xx_loadl_64(src0);
const __m128i s0B = xx_loadl_64(src0 + src0_stride);
const __m128i s0C = xx_loadl_64(src0 + src0_stride * 2);
const __m128i s0D = xx_loadl_64(src0 + src0_stride * 3);
const __m128i s1A = xx_loadl_64(src1);
const __m128i s1B = xx_loadl_64(src1 + src1_stride);
const __m128i s1C = xx_loadl_64(src1 + src1_stride * 2);
const __m128i s1D = xx_loadl_64(src1 + src1_stride * 3);
const __m256i s0 = yy_set_m128i(_mm_unpacklo_epi64(s0C, s0D),
_mm_unpacklo_epi64(s0A, s0B));
const __m256i s1 = yy_set_m128i(_mm_unpacklo_epi64(s1C, s1D),
_mm_unpacklo_epi64(s1A, s1B));
const __m256i m16 =
calc_mask_d16_inv_avx2(&s0, &s1, &_r, &y38, &y64, shift);
const __m256i m8 = _mm256_packus_epi16(m16, _mm256_setzero_si256());
xx_storeu_128(mask,
_mm256_castsi256_si128(_mm256_permute4x64_epi64(m8, 0xd8)));
src0 += src0_stride << 2;
src1 += src1_stride << 2;
mask += 16;
i += 4;
} while (i < h);
} else if (w == 8) {
do {
const __m256i s0AB = yy_loadu2_128(src0 + src0_stride, src0);
const __m256i s0CD =
yy_loadu2_128(src0 + src0_stride * 3, src0 + src0_stride * 2);
const __m256i s1AB = yy_loadu2_128(src1 + src1_stride, src1);
const __m256i s1CD =
yy_loadu2_128(src1 + src1_stride * 3, src1 + src1_stride * 2);
const __m256i m16AB =
calc_mask_d16_inv_avx2(&s0AB, &s1AB, &_r, &y38, &y64, shift);
const __m256i m16CD =
calc_mask_d16_inv_avx2(&s0CD, &s1CD, &_r, &y38, &y64, shift);
const __m256i m8 = _mm256_packus_epi16(m16AB, m16CD);
yy_storeu_256(mask, _mm256_permute4x64_epi64(m8, 0xd8));
src0 += src0_stride << 2;
src1 += src1_stride << 2;
mask += 32;
i += 4;
} while (i < h);
} else if (w == 16) {
do {
const __m256i s0A = yy_loadu_256(src0);
const __m256i s0B = yy_loadu_256(src0 + src0_stride);
const __m256i s1A = yy_loadu_256(src1);
const __m256i s1B = yy_loadu_256(src1 + src1_stride);
const __m256i m16A =
calc_mask_d16_inv_avx2(&s0A, &s1A, &_r, &y38, &y64, shift);
const __m256i m16B =
calc_mask_d16_inv_avx2(&s0B, &s1B, &_r, &y38, &y64, shift);
const __m256i m8 = _mm256_packus_epi16(m16A, m16B);
yy_storeu_256(mask, _mm256_permute4x64_epi64(m8, 0xd8));
src0 += src0_stride << 1;
src1 += src1_stride << 1;
mask += 32;
i += 2;
} while (i < h);
} else if (w == 32) {
do {
const __m256i s0A = yy_loadu_256(src0);
const __m256i s0B = yy_loadu_256(src0 + 16);
const __m256i s1A = yy_loadu_256(src1);
const __m256i s1B = yy_loadu_256(src1 + 16);
const __m256i m16A =
calc_mask_d16_inv_avx2(&s0A, &s1A, &_r, &y38, &y64, shift);
const __m256i m16B =
calc_mask_d16_inv_avx2(&s0B, &s1B, &_r, &y38, &y64, shift);
const __m256i m8 = _mm256_packus_epi16(m16A, m16B);
yy_storeu_256(mask, _mm256_permute4x64_epi64(m8, 0xd8));
src0 += src0_stride;
src1 += src1_stride;
mask += 32;
i += 1;
} while (i < h);
} else if (w == 64) {
do {
const __m256i s0A = yy_loadu_256(src0);
const __m256i s0B = yy_loadu_256(src0 + 16);
const __m256i s0C = yy_loadu_256(src0 + 32);
const __m256i s0D = yy_loadu_256(src0 + 48);
const __m256i s1A = yy_loadu_256(src1);
const __m256i s1B = yy_loadu_256(src1 + 16);
const __m256i s1C = yy_loadu_256(src1 + 32);
const __m256i s1D = yy_loadu_256(src1 + 48);
const __m256i m16A =
calc_mask_d16_inv_avx2(&s0A, &s1A, &_r, &y38, &y64, shift);
const __m256i m16B =
calc_mask_d16_inv_avx2(&s0B, &s1B, &_r, &y38, &y64, shift);
const __m256i m16C =
calc_mask_d16_inv_avx2(&s0C, &s1C, &_r, &y38, &y64, shift);
const __m256i m16D =
calc_mask_d16_inv_avx2(&s0D, &s1D, &_r, &y38, &y64, shift);
const __m256i m8AB = _mm256_packus_epi16(m16A, m16B);
const __m256i m8CD = _mm256_packus_epi16(m16C, m16D);
yy_storeu_256(mask, _mm256_permute4x64_epi64(m8AB, 0xd8));
yy_storeu_256(mask + 32, _mm256_permute4x64_epi64(m8CD, 0xd8));
src0 += src0_stride;
src1 += src1_stride;
mask += 64;
i += 1;
} while (i < h);
} else {
do {
const __m256i s0A = yy_loadu_256(src0);
const __m256i s0B = yy_loadu_256(src0 + 16);
const __m256i s0C = yy_loadu_256(src0 + 32);
const __m256i s0D = yy_loadu_256(src0 + 48);
const __m256i s0E = yy_loadu_256(src0 + 64);
const __m256i s0F = yy_loadu_256(src0 + 80);
const __m256i s0G = yy_loadu_256(src0 + 96);
const __m256i s0H = yy_loadu_256(src0 + 112);
const __m256i s1A = yy_loadu_256(src1);
const __m256i s1B = yy_loadu_256(src1 + 16);
const __m256i s1C = yy_loadu_256(src1 + 32);
const __m256i s1D = yy_loadu_256(src1 + 48);
const __m256i s1E = yy_loadu_256(src1 + 64);
const __m256i s1F = yy_loadu_256(src1 + 80);
const __m256i s1G = yy_loadu_256(src1 + 96);
const __m256i s1H = yy_loadu_256(src1 + 112);
const __m256i m16A =
calc_mask_d16_inv_avx2(&s0A, &s1A, &_r, &y38, &y64, shift);
const __m256i m16B =
calc_mask_d16_inv_avx2(&s0B, &s1B, &_r, &y38, &y64, shift);
const __m256i m16C =
calc_mask_d16_inv_avx2(&s0C, &s1C, &_r, &y38, &y64, shift);
const __m256i m16D =
calc_mask_d16_inv_avx2(&s0D, &s1D, &_r, &y38, &y64, shift);
const __m256i m16E =
calc_mask_d16_inv_avx2(&s0E, &s1E, &_r, &y38, &y64, shift);
const __m256i m16F =
calc_mask_d16_inv_avx2(&s0F, &s1F, &_r, &y38, &y64, shift);
const __m256i m16G =
calc_mask_d16_inv_avx2(&s0G, &s1G, &_r, &y38, &y64, shift);
const __m256i m16H =
calc_mask_d16_inv_avx2(&s0H, &s1H, &_r, &y38, &y64, shift);
const __m256i m8AB = _mm256_packus_epi16(m16A, m16B);
const __m256i m8CD = _mm256_packus_epi16(m16C, m16D);
const __m256i m8EF = _mm256_packus_epi16(m16E, m16F);
const __m256i m8GH = _mm256_packus_epi16(m16G, m16H);
yy_storeu_256(mask, _mm256_permute4x64_epi64(m8AB, 0xd8));
yy_storeu_256(mask + 32, _mm256_permute4x64_epi64(m8CD, 0xd8));
yy_storeu_256(mask + 64, _mm256_permute4x64_epi64(m8EF, 0xd8));
yy_storeu_256(mask + 96, _mm256_permute4x64_epi64(m8GH, 0xd8));
src0 += src0_stride;
src1 += src1_stride;
mask += 128;
i += 1;
} while (i < h);
}
}
void av1_build_compound_diffwtd_mask_d16_avx2(
uint8_t *mask, DIFFWTD_MASK_TYPE mask_type, const CONV_BUF_TYPE *src0,
int src0_stride, const CONV_BUF_TYPE *src1, int src1_stride, int h, int w,
ConvolveParams *conv_params, int bd) {
const int shift =
2 * FILTER_BITS - conv_params->round_0 - conv_params->round_1 + (bd - 8);
// When rounding constant is added, there is a possibility of overflow.
// However that much precision is not required. Code should very well work for
// other values of DIFF_FACTOR_LOG2 and AOM_BLEND_A64_MAX_ALPHA as well. But
// there is a possibility of corner case bugs.
assert(DIFF_FACTOR_LOG2 == 4);
assert(AOM_BLEND_A64_MAX_ALPHA == 64);
if (mask_type == DIFFWTD_38) {
build_compound_diffwtd_mask_d16_avx2(mask, src0, src0_stride, src1,
src1_stride, h, w, shift);
} else {
build_compound_diffwtd_mask_d16_inv_avx2(mask, src0, src0_stride, src1,
src1_stride, h, w, shift);
}
}
void av1_build_compound_diffwtd_mask_highbd_avx2(
uint8_t *mask, DIFFWTD_MASK_TYPE mask_type, const uint8_t *src0,
int src0_stride, const uint8_t *src1, int src1_stride, int h, int w,

View file

@ -546,17 +546,18 @@ static void final_filter_fast(int32_t *dst, int dst_stride, const int32_t *A,
}
}
void av1_selfguided_restoration_avx2(const uint8_t *dgd8, int width, int height,
int dgd_stride, int32_t *flt0,
int32_t *flt1, int flt_stride,
int sgr_params_idx, int bit_depth,
int highbd) {
int av1_selfguided_restoration_avx2(const uint8_t *dgd8, int width, int height,
int dgd_stride, int32_t *flt0,
int32_t *flt1, int flt_stride,
int sgr_params_idx, int bit_depth,
int highbd) {
// The ALIGN_POWER_OF_TWO macro here ensures that column 1 of Atl, Btl,
// Ctl and Dtl is 32-byte aligned.
const int buf_elts = ALIGN_POWER_OF_TWO(RESTORATION_PROC_UNIT_PELS, 3);
DECLARE_ALIGNED(32, int32_t,
buf[4 * ALIGN_POWER_OF_TWO(RESTORATION_PROC_UNIT_PELS, 3)]);
int32_t *buf = aom_memalign(
32, 4 * sizeof(*buf) * ALIGN_POWER_OF_TWO(RESTORATION_PROC_UNIT_PELS, 3));
if (!buf) return -1;
const int width_ext = width + 2 * SGRPROJ_BORDER_HORZ;
const int height_ext = height + 2 * SGRPROJ_BORDER_VERT;
@ -625,6 +626,8 @@ void av1_selfguided_restoration_avx2(const uint8_t *dgd8, int width, int height,
final_filter(flt1, flt_stride, A, B, buf_stride, dgd8, dgd_stride, width,
height, highbd);
}
aom_free(buf);
return 0;
}
void apply_selfguided_restoration_avx2(const uint8_t *dat8, int width,
@ -635,8 +638,10 @@ void apply_selfguided_restoration_avx2(const uint8_t *dat8, int width,
int32_t *flt0 = tmpbuf;
int32_t *flt1 = flt0 + RESTORATION_UNITPELS_MAX;
assert(width * height <= RESTORATION_UNITPELS_MAX);
av1_selfguided_restoration_avx2(dat8, width, height, stride, flt0, flt1,
width, eps, bit_depth, highbd);
const int ret = av1_selfguided_restoration_avx2(
dat8, width, height, stride, flt0, flt1, width, eps, bit_depth, highbd);
(void)ret;
assert(!ret);
const sgr_params_type *const params = &sgr_params[eps];
int xq[2];
decode_xq(xqd, xq, params);

View file

@ -499,13 +499,15 @@ static void final_filter_fast(int32_t *dst, int dst_stride, const int32_t *A,
}
}
void av1_selfguided_restoration_sse4_1(const uint8_t *dgd8, int width,
int height, int dgd_stride,
int32_t *flt0, int32_t *flt1,
int flt_stride, int sgr_params_idx,
int bit_depth, int highbd) {
DECLARE_ALIGNED(16, int32_t, buf[4 * RESTORATION_PROC_UNIT_PELS]);
memset(buf, 0, sizeof(buf));
int av1_selfguided_restoration_sse4_1(const uint8_t *dgd8, int width,
int height, int dgd_stride, int32_t *flt0,
int32_t *flt1, int flt_stride,
int sgr_params_idx, int bit_depth,
int highbd) {
int32_t *buf = (int32_t *)aom_memalign(
16, 4 * sizeof(*buf) * RESTORATION_PROC_UNIT_PELS);
if (!buf) return -1;
memset(buf, 0, 4 * sizeof(*buf) * RESTORATION_PROC_UNIT_PELS);
const int width_ext = width + 2 * SGRPROJ_BORDER_HORZ;
const int height_ext = height + 2 * SGRPROJ_BORDER_VERT;
@ -574,6 +576,8 @@ void av1_selfguided_restoration_sse4_1(const uint8_t *dgd8, int width,
final_filter(flt1, flt_stride, A, B, buf_stride, dgd8, dgd_stride, width,
height, highbd);
}
aom_free(buf);
return 0;
}
void apply_selfguided_restoration_sse4_1(const uint8_t *dat8, int width,
@ -584,8 +588,10 @@ void apply_selfguided_restoration_sse4_1(const uint8_t *dat8, int width,
int32_t *flt0 = tmpbuf;
int32_t *flt1 = flt0 + RESTORATION_UNITPELS_MAX;
assert(width * height <= RESTORATION_UNITPELS_MAX);
av1_selfguided_restoration_sse4_1(dat8, width, height, stride, flt0, flt1,
width, eps, bit_depth, highbd);
const int ret = av1_selfguided_restoration_sse4_1(
dat8, width, height, stride, flt0, flt1, width, eps, bit_depth, highbd);
(void)ret;
assert(!ret);
const sgr_params_type *const params = &sgr_params[eps];
int xq[2];
decode_xq(xqd, xq, params);

Some files were not shown because too many files have changed in this diff Show more