mirror of
https://repo.dactyloidae.xyz/Dactyloidae/UXP.git
synced 2026-10-09 00:37:32 +09:00
Update aom to commit id e87fb2378f01103d5d6e477a4ef6892dc714e614
This commit is contained in:
parent
debbee1e2a
commit
992c6637e3
429 changed files with 76047 additions and 40937 deletions
|
|
@ -39,7 +39,7 @@ struct CYCLIC_REFRESH {
|
|||
// RD mult. parameters for segment 1.
|
||||
int rdmult;
|
||||
// Cyclic refresh map.
|
||||
signed char *map;
|
||||
int8_t *map;
|
||||
// Map of the last q a block was coded at.
|
||||
uint8_t *last_coded_q_map;
|
||||
// Thresholds applied to the projected rate/distortion of the coding block,
|
||||
|
|
@ -397,6 +397,7 @@ static void cyclic_refresh_update_map(AV1_COMP *const cpi) {
|
|||
// Set the segmentation map: cycle through the superblocks, starting at
|
||||
// cr->mb_index, and stopping when either block_count blocks have been found
|
||||
// to be refreshed, or we have passed through whole frame.
|
||||
if (cr->sb_index >= sbs_in_frame) cr->sb_index = 0;
|
||||
assert(cr->sb_index < sbs_in_frame);
|
||||
i = cr->sb_index;
|
||||
cr->target_num_seg_blocks = 0;
|
||||
|
|
|
|||
4
third_party/aom/av1/encoder/aq_variance.c
vendored
4
third_party/aom/av1/encoder/aq_variance.c
vendored
|
|
@ -151,8 +151,8 @@ static unsigned int block_variance(const AV1_COMP *const cpi, MACROBLOCK *x,
|
|||
(xd->mb_to_bottom_edge < 0) ? ((-xd->mb_to_bottom_edge) >> 3) : 0;
|
||||
|
||||
if (right_overflow || bottom_overflow) {
|
||||
const int bw = 8 * mi_size_wide[bs] - right_overflow;
|
||||
const int bh = 8 * mi_size_high[bs] - bottom_overflow;
|
||||
const int bw = MI_SIZE * mi_size_wide[bs] - right_overflow;
|
||||
const int bh = MI_SIZE * mi_size_high[bs] - bottom_overflow;
|
||||
int avg;
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
if (xd->cur_buf->flags & YV12_FLAG_HIGHBITDEPTH) {
|
||||
|
|
|
|||
36
third_party/aom/av1/encoder/arm/neon/dct_neon.c
vendored
36
third_party/aom/av1/encoder/arm/neon/dct_neon.c
vendored
|
|
@ -1,36 +0,0 @@
|
|||
/*
|
||||
* Copyright (c) 2016, Alliance for Open Media. All rights reserved
|
||||
*
|
||||
* This source code is subject to the terms of the BSD 2 Clause License and
|
||||
* the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
|
||||
* was not distributed with this source code in the LICENSE file, you can
|
||||
* obtain it at www.aomedia.org/license/software. If the Alliance for Open
|
||||
* Media Patent License 1.0 was not distributed with this source code in the
|
||||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#include <arm_neon.h>
|
||||
|
||||
#include "./av1_rtcd.h"
|
||||
#include "./aom_config.h"
|
||||
#include "./aom_dsp_rtcd.h"
|
||||
|
||||
#include "av1/common/blockd.h"
|
||||
#include "aom_dsp/txfm_common.h"
|
||||
|
||||
void av1_fdct8x8_quant_neon(const int16_t *input, int stride,
|
||||
int16_t *coeff_ptr, intptr_t n_coeffs,
|
||||
int skip_block, const int16_t *zbin_ptr,
|
||||
const int16_t *round_ptr, const int16_t *quant_ptr,
|
||||
const int16_t *quant_shift_ptr, int16_t *qcoeff_ptr,
|
||||
int16_t *dqcoeff_ptr, const int16_t *dequant_ptr,
|
||||
uint16_t *eob_ptr, const int16_t *scan_ptr,
|
||||
const int16_t *iscan_ptr) {
|
||||
int16_t temp_buffer[64];
|
||||
(void)coeff_ptr;
|
||||
|
||||
aom_fdct8x8_neon(input, temp_buffer, stride);
|
||||
av1_quantize_fp_neon(temp_buffer, n_coeffs, skip_block, zbin_ptr, round_ptr,
|
||||
quant_ptr, quant_shift_ptr, qcoeff_ptr, dqcoeff_ptr,
|
||||
dequant_ptr, eob_ptr, scan_ptr, iscan_ptr);
|
||||
}
|
||||
564
third_party/aom/av1/encoder/av1_quantize.c
vendored
564
third_party/aom/av1/encoder/av1_quantize.c
vendored
|
|
@ -443,11 +443,8 @@ static void quantize_fp_helper_c(
|
|||
const int16_t *zbin_ptr, const int16_t *round_ptr, const int16_t *quant_ptr,
|
||||
const int16_t *quant_shift_ptr, tran_low_t *qcoeff_ptr,
|
||||
tran_low_t *dqcoeff_ptr, const int16_t *dequant_ptr, uint16_t *eob_ptr,
|
||||
const int16_t *scan, const int16_t *iscan,
|
||||
#if CONFIG_AOM_QM
|
||||
const qm_val_t *qm_ptr, const qm_val_t *iqm_ptr,
|
||||
#endif
|
||||
int log_scale) {
|
||||
const int16_t *scan, const int16_t *iscan, const qm_val_t *qm_ptr,
|
||||
const qm_val_t *iqm_ptr, int log_scale) {
|
||||
int i, eob = -1;
|
||||
// TODO(jingning) Decide the need of these arguments after the
|
||||
// quantization process is completed.
|
||||
|
|
@ -464,35 +461,22 @@ static void quantize_fp_helper_c(
|
|||
for (i = 0; i < n_coeffs; i++) {
|
||||
const int rc = scan[i];
|
||||
const int coeff = coeff_ptr[rc];
|
||||
#if CONFIG_AOM_QM
|
||||
const qm_val_t wt = qm_ptr[rc];
|
||||
const qm_val_t iwt = iqm_ptr[rc];
|
||||
const qm_val_t wt = qm_ptr ? qm_ptr[rc] : (1 << AOM_QM_BITS);
|
||||
const qm_val_t iwt = iqm_ptr ? iqm_ptr[rc] : (1 << AOM_QM_BITS);
|
||||
const int dequant =
|
||||
(dequant_ptr[rc != 0] * iwt + (1 << (AOM_QM_BITS - 1))) >>
|
||||
AOM_QM_BITS;
|
||||
#endif
|
||||
const int coeff_sign = (coeff >> 31);
|
||||
int abs_coeff = (coeff ^ coeff_sign) - coeff_sign;
|
||||
int64_t abs_coeff = (coeff ^ coeff_sign) - coeff_sign;
|
||||
int tmp32 = 0;
|
||||
#if CONFIG_AOM_QM
|
||||
if (abs_coeff * wt >=
|
||||
(dequant_ptr[rc != 0] << (AOM_QM_BITS - (1 + log_scale)))) {
|
||||
#else
|
||||
if (abs_coeff >= (dequant_ptr[rc != 0] >> (1 + log_scale))) {
|
||||
#endif
|
||||
abs_coeff += ROUND_POWER_OF_TWO(round_ptr[rc != 0], log_scale);
|
||||
abs_coeff = clamp(abs_coeff, INT16_MIN, INT16_MAX);
|
||||
#if CONFIG_AOM_QM
|
||||
abs_coeff = clamp64(abs_coeff, INT16_MIN, INT16_MAX);
|
||||
tmp32 = (int)((abs_coeff * wt * quant_ptr[rc != 0]) >>
|
||||
((16 - log_scale) + AOM_QM_BITS));
|
||||
(16 - log_scale + AOM_QM_BITS));
|
||||
qcoeff_ptr[rc] = (tmp32 ^ coeff_sign) - coeff_sign;
|
||||
dqcoeff_ptr[rc] = qcoeff_ptr[rc] * dequant / (1 << log_scale);
|
||||
#else
|
||||
tmp32 = (int)((abs_coeff * quant_ptr[rc != 0]) >> (16 - log_scale));
|
||||
qcoeff_ptr[rc] = (tmp32 ^ coeff_sign) - coeff_sign;
|
||||
dqcoeff_ptr[rc] =
|
||||
qcoeff_ptr[rc] * dequant_ptr[rc != 0] / (1 << log_scale);
|
||||
#endif
|
||||
}
|
||||
|
||||
if (tmp32) eob = i;
|
||||
|
|
@ -501,25 +485,60 @@ static void quantize_fp_helper_c(
|
|||
*eob_ptr = eob + 1;
|
||||
}
|
||||
|
||||
static void highbd_quantize_fp_helper_c(
|
||||
const tran_low_t *coeff_ptr, intptr_t count, int skip_block,
|
||||
const int16_t *zbin_ptr, const int16_t *round_ptr, const int16_t *quant_ptr,
|
||||
const int16_t *quant_shift_ptr, tran_low_t *qcoeff_ptr,
|
||||
tran_low_t *dqcoeff_ptr, const int16_t *dequant_ptr, uint16_t *eob_ptr,
|
||||
const int16_t *scan, const int16_t *iscan, const qm_val_t *qm_ptr,
|
||||
const qm_val_t *iqm_ptr, int log_scale) {
|
||||
int i;
|
||||
int eob = -1;
|
||||
const int scale = 1 << log_scale;
|
||||
const int shift = 16 - log_scale;
|
||||
// TODO(jingning) Decide the need of these arguments after the
|
||||
// quantization process is completed.
|
||||
(void)zbin_ptr;
|
||||
(void)quant_shift_ptr;
|
||||
(void)iscan;
|
||||
|
||||
memset(qcoeff_ptr, 0, count * sizeof(*qcoeff_ptr));
|
||||
memset(dqcoeff_ptr, 0, count * sizeof(*dqcoeff_ptr));
|
||||
|
||||
if (!skip_block) {
|
||||
// Quantization pass: All coefficients with index >= zero_flag are
|
||||
// skippable. Note: zero_flag can be zero.
|
||||
for (i = 0; i < count; i++) {
|
||||
const int rc = scan[i];
|
||||
const int coeff = coeff_ptr[rc];
|
||||
const qm_val_t wt = qm_ptr != NULL ? qm_ptr[rc] : (1 << AOM_QM_BITS);
|
||||
const qm_val_t iwt = iqm_ptr != NULL ? iqm_ptr[rc] : (1 << AOM_QM_BITS);
|
||||
const int dequant =
|
||||
(dequant_ptr[rc != 0] * iwt + (1 << (AOM_QM_BITS - 1))) >>
|
||||
AOM_QM_BITS;
|
||||
const int coeff_sign = (coeff >> 31);
|
||||
const int abs_coeff = (coeff ^ coeff_sign) - coeff_sign;
|
||||
const int64_t tmp = abs_coeff + (round_ptr[rc != 0] >> log_scale);
|
||||
const int abs_qcoeff =
|
||||
(int)((tmp * quant_ptr[rc != 0] * wt) >> (shift + AOM_QM_BITS));
|
||||
qcoeff_ptr[rc] = (tran_low_t)((abs_qcoeff ^ coeff_sign) - coeff_sign);
|
||||
dqcoeff_ptr[rc] = qcoeff_ptr[rc] * dequant / scale;
|
||||
if (abs_qcoeff) eob = i;
|
||||
}
|
||||
}
|
||||
*eob_ptr = eob + 1;
|
||||
}
|
||||
|
||||
void av1_quantize_fp_c(const tran_low_t *coeff_ptr, intptr_t n_coeffs,
|
||||
int skip_block, const int16_t *zbin_ptr,
|
||||
const int16_t *round_ptr, const int16_t *quant_ptr,
|
||||
const int16_t *quant_shift_ptr, tran_low_t *qcoeff_ptr,
|
||||
tran_low_t *dqcoeff_ptr, const int16_t *dequant_ptr,
|
||||
uint16_t *eob_ptr, const int16_t *scan,
|
||||
const int16_t *iscan
|
||||
#if CONFIG_AOM_QM
|
||||
,
|
||||
const qm_val_t *qm_ptr, const qm_val_t *iqm_ptr
|
||||
#endif
|
||||
) {
|
||||
const int16_t *iscan) {
|
||||
quantize_fp_helper_c(coeff_ptr, n_coeffs, skip_block, zbin_ptr, round_ptr,
|
||||
quant_ptr, quant_shift_ptr, qcoeff_ptr, dqcoeff_ptr,
|
||||
dequant_ptr, eob_ptr, scan, iscan,
|
||||
#if CONFIG_AOM_QM
|
||||
qm_ptr, iqm_ptr,
|
||||
#endif
|
||||
0);
|
||||
dequant_ptr, eob_ptr, scan, iscan, NULL, NULL, 0);
|
||||
}
|
||||
|
||||
void av1_quantize_fp_32x32_c(const tran_low_t *coeff_ptr, intptr_t n_coeffs,
|
||||
|
|
@ -528,19 +547,10 @@ void av1_quantize_fp_32x32_c(const tran_low_t *coeff_ptr, intptr_t n_coeffs,
|
|||
const int16_t *quant_shift_ptr,
|
||||
tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr,
|
||||
const int16_t *dequant_ptr, uint16_t *eob_ptr,
|
||||
const int16_t *scan, const int16_t *iscan
|
||||
#if CONFIG_AOM_QM
|
||||
,
|
||||
const qm_val_t *qm_ptr, const qm_val_t *iqm_ptr
|
||||
#endif
|
||||
) {
|
||||
const int16_t *scan, const int16_t *iscan) {
|
||||
quantize_fp_helper_c(coeff_ptr, n_coeffs, skip_block, zbin_ptr, round_ptr,
|
||||
quant_ptr, quant_shift_ptr, qcoeff_ptr, dqcoeff_ptr,
|
||||
dequant_ptr, eob_ptr, scan, iscan,
|
||||
#if CONFIG_AOM_QM
|
||||
qm_ptr, iqm_ptr,
|
||||
#endif
|
||||
1);
|
||||
dequant_ptr, eob_ptr, scan, iscan, NULL, NULL, 1);
|
||||
}
|
||||
|
||||
#if CONFIG_TX64X64
|
||||
|
|
@ -550,19 +560,10 @@ void av1_quantize_fp_64x64_c(const tran_low_t *coeff_ptr, intptr_t n_coeffs,
|
|||
const int16_t *quant_shift_ptr,
|
||||
tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr,
|
||||
const int16_t *dequant_ptr, uint16_t *eob_ptr,
|
||||
const int16_t *scan, const int16_t *iscan
|
||||
#if CONFIG_AOM_QM
|
||||
,
|
||||
const qm_val_t *qm_ptr, const qm_val_t *iqm_ptr
|
||||
#endif
|
||||
) {
|
||||
const int16_t *scan, const int16_t *iscan) {
|
||||
quantize_fp_helper_c(coeff_ptr, n_coeffs, skip_block, zbin_ptr, round_ptr,
|
||||
quant_ptr, quant_shift_ptr, qcoeff_ptr, dqcoeff_ptr,
|
||||
dequant_ptr, eob_ptr, scan, iscan,
|
||||
#if CONFIG_AOM_QM
|
||||
qm_ptr, iqm_ptr,
|
||||
#endif
|
||||
2);
|
||||
dequant_ptr, eob_ptr, scan, iscan, NULL, NULL, 2);
|
||||
}
|
||||
#endif // CONFIG_TX64X64
|
||||
|
||||
|
|
@ -576,58 +577,47 @@ void av1_quantize_fp_facade(const tran_low_t *coeff_ptr, intptr_t n_coeffs,
|
|||
#if CONFIG_AOM_QM
|
||||
const qm_val_t *qm_ptr = qparam->qmatrix;
|
||||
const qm_val_t *iqm_ptr = qparam->iqmatrix;
|
||||
#endif // CONFIG_AOM_QM
|
||||
|
||||
switch (qparam->log_scale) {
|
||||
case 0:
|
||||
if (n_coeffs < 16) {
|
||||
// TODO(jingning): Need SIMD implementation for smaller block size
|
||||
// quantization.
|
||||
quantize_fp_helper_c(coeff_ptr, n_coeffs, skip_block, p->zbin,
|
||||
p->round_fp, p->quant_fp, p->quant_shift,
|
||||
qcoeff_ptr, dqcoeff_ptr, pd->dequant, eob_ptr,
|
||||
sc->scan, sc->iscan,
|
||||
#if CONFIG_AOM_QM
|
||||
qm_ptr, iqm_ptr,
|
||||
if (qm_ptr != NULL && iqm_ptr != NULL) {
|
||||
quantize_fp_helper_c(coeff_ptr, n_coeffs, skip_block, p->zbin, p->round_fp,
|
||||
p->quant_fp, p->quant_shift, qcoeff_ptr, dqcoeff_ptr,
|
||||
pd->dequant, eob_ptr, sc->scan, sc->iscan, qm_ptr,
|
||||
iqm_ptr, qparam->log_scale);
|
||||
} else {
|
||||
#endif
|
||||
qparam->log_scale);
|
||||
} else {
|
||||
av1_quantize_fp(coeff_ptr, n_coeffs, skip_block, p->zbin, p->round_fp,
|
||||
p->quant_fp, p->quant_shift, qcoeff_ptr, dqcoeff_ptr,
|
||||
pd->dequant, eob_ptr, sc->scan, sc->iscan
|
||||
#if CONFIG_AOM_QM
|
||||
,
|
||||
qm_ptr, iqm_ptr
|
||||
#endif
|
||||
);
|
||||
}
|
||||
break;
|
||||
case 1:
|
||||
av1_quantize_fp_32x32(coeff_ptr, n_coeffs, skip_block, p->zbin,
|
||||
p->round_fp, p->quant_fp, p->quant_shift,
|
||||
qcoeff_ptr, dqcoeff_ptr, pd->dequant, eob_ptr,
|
||||
sc->scan, sc->iscan
|
||||
#if CONFIG_AOM_QM
|
||||
,
|
||||
qm_ptr, iqm_ptr
|
||||
#endif
|
||||
);
|
||||
break;
|
||||
switch (qparam->log_scale) {
|
||||
case 0:
|
||||
if (n_coeffs < 16) {
|
||||
// TODO(jingning): Need SIMD implementation for smaller block size
|
||||
// quantization.
|
||||
quantize_fp_helper_c(
|
||||
coeff_ptr, n_coeffs, skip_block, p->zbin, p->round_fp,
|
||||
p->quant_fp, p->quant_shift, qcoeff_ptr, dqcoeff_ptr, pd->dequant,
|
||||
eob_ptr, sc->scan, sc->iscan, NULL, NULL, qparam->log_scale);
|
||||
} else {
|
||||
av1_quantize_fp(coeff_ptr, n_coeffs, skip_block, p->zbin, p->round_fp,
|
||||
p->quant_fp, p->quant_shift, qcoeff_ptr, dqcoeff_ptr,
|
||||
pd->dequant, eob_ptr, sc->scan, sc->iscan);
|
||||
}
|
||||
break;
|
||||
case 1:
|
||||
av1_quantize_fp_32x32(coeff_ptr, n_coeffs, skip_block, p->zbin,
|
||||
p->round_fp, p->quant_fp, p->quant_shift,
|
||||
qcoeff_ptr, dqcoeff_ptr, pd->dequant, eob_ptr,
|
||||
sc->scan, sc->iscan);
|
||||
break;
|
||||
#if CONFIG_TX64X64
|
||||
case 2:
|
||||
av1_quantize_fp_64x64(coeff_ptr, n_coeffs, skip_block, p->zbin,
|
||||
p->round_fp, p->quant_fp, p->quant_shift,
|
||||
qcoeff_ptr, dqcoeff_ptr, pd->dequant, eob_ptr,
|
||||
sc->scan, sc->iscan
|
||||
#if CONFIG_AOM_QM
|
||||
,
|
||||
qm_ptr, iqm_ptr
|
||||
#endif
|
||||
);
|
||||
break;
|
||||
case 2:
|
||||
av1_quantize_fp_64x64(coeff_ptr, n_coeffs, skip_block, p->zbin,
|
||||
p->round_fp, p->quant_fp, p->quant_shift,
|
||||
qcoeff_ptr, dqcoeff_ptr, pd->dequant, eob_ptr,
|
||||
sc->scan, sc->iscan);
|
||||
break;
|
||||
#endif // CONFIG_TX64X64
|
||||
default: assert(0);
|
||||
default: assert(0);
|
||||
}
|
||||
#if CONFIG_AOM_QM
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
void av1_quantize_b_facade(const tran_low_t *coeff_ptr, intptr_t n_coeffs,
|
||||
|
|
@ -640,43 +630,69 @@ void av1_quantize_b_facade(const tran_low_t *coeff_ptr, intptr_t n_coeffs,
|
|||
#if CONFIG_AOM_QM
|
||||
const qm_val_t *qm_ptr = qparam->qmatrix;
|
||||
const qm_val_t *iqm_ptr = qparam->iqmatrix;
|
||||
if (qm_ptr != NULL && iqm_ptr != NULL) {
|
||||
quantize_b_helper_c(coeff_ptr, n_coeffs, skip_block, p->zbin, p->round,
|
||||
p->quant, p->quant_shift, qcoeff_ptr, dqcoeff_ptr,
|
||||
pd->dequant, eob_ptr, sc->scan, sc->iscan, qm_ptr,
|
||||
iqm_ptr, qparam->log_scale);
|
||||
} else {
|
||||
#endif // CONFIG_AOM_QM
|
||||
|
||||
switch (qparam->log_scale) {
|
||||
case 0:
|
||||
aom_quantize_b(coeff_ptr, n_coeffs, skip_block, p->zbin, p->round,
|
||||
p->quant, p->quant_shift, qcoeff_ptr, dqcoeff_ptr,
|
||||
pd->dequant, eob_ptr, sc->scan, sc->iscan
|
||||
#if CONFIG_AOM_QM
|
||||
,
|
||||
qm_ptr, iqm_ptr
|
||||
#endif
|
||||
);
|
||||
break;
|
||||
case 1:
|
||||
aom_quantize_b_32x32(coeff_ptr, n_coeffs, skip_block, p->zbin, p->round,
|
||||
p->quant, p->quant_shift, qcoeff_ptr, dqcoeff_ptr,
|
||||
pd->dequant, eob_ptr, sc->scan, sc->iscan
|
||||
#if CONFIG_AOM_QM
|
||||
,
|
||||
qm_ptr, iqm_ptr
|
||||
#endif
|
||||
);
|
||||
break;
|
||||
switch (qparam->log_scale) {
|
||||
case 0:
|
||||
aom_quantize_b(coeff_ptr, n_coeffs, skip_block, p->zbin, p->round,
|
||||
p->quant, p->quant_shift, qcoeff_ptr, dqcoeff_ptr,
|
||||
pd->dequant, eob_ptr, sc->scan, sc->iscan);
|
||||
break;
|
||||
case 1:
|
||||
aom_quantize_b_32x32(coeff_ptr, n_coeffs, skip_block, p->zbin, p->round,
|
||||
p->quant, p->quant_shift, qcoeff_ptr, dqcoeff_ptr,
|
||||
pd->dequant, eob_ptr, sc->scan, sc->iscan);
|
||||
break;
|
||||
#if CONFIG_TX64X64
|
||||
case 2:
|
||||
aom_quantize_b_64x64(coeff_ptr, n_coeffs, skip_block, p->zbin, p->round,
|
||||
p->quant, p->quant_shift, qcoeff_ptr, dqcoeff_ptr,
|
||||
pd->dequant, eob_ptr, sc->scan, sc->iscan
|
||||
#if CONFIG_AOM_QM
|
||||
,
|
||||
qm_ptr, iqm_ptr
|
||||
#endif
|
||||
);
|
||||
break;
|
||||
case 2:
|
||||
aom_quantize_b_64x64(coeff_ptr, n_coeffs, skip_block, p->zbin, p->round,
|
||||
p->quant, p->quant_shift, qcoeff_ptr, dqcoeff_ptr,
|
||||
pd->dequant, eob_ptr, sc->scan, sc->iscan);
|
||||
break;
|
||||
#endif // CONFIG_TX64X64
|
||||
default: assert(0);
|
||||
default: assert(0);
|
||||
}
|
||||
#if CONFIG_AOM_QM
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
static void quantize_dc(const tran_low_t *coeff_ptr, int n_coeffs,
|
||||
int skip_block, const int16_t *round_ptr,
|
||||
const int16_t quant, tran_low_t *qcoeff_ptr,
|
||||
tran_low_t *dqcoeff_ptr, const int16_t dequant_ptr,
|
||||
uint16_t *eob_ptr, const qm_val_t *qm_ptr,
|
||||
const qm_val_t *iqm_ptr, const int log_scale) {
|
||||
const int rc = 0;
|
||||
const int coeff = coeff_ptr[rc];
|
||||
const int coeff_sign = (coeff >> 31);
|
||||
const int abs_coeff = (coeff ^ coeff_sign) - coeff_sign;
|
||||
int64_t tmp;
|
||||
int eob = -1;
|
||||
int32_t tmp32;
|
||||
int dequant;
|
||||
|
||||
memset(qcoeff_ptr, 0, n_coeffs * sizeof(*qcoeff_ptr));
|
||||
memset(dqcoeff_ptr, 0, n_coeffs * sizeof(*dqcoeff_ptr));
|
||||
|
||||
if (!skip_block) {
|
||||
const int wt = qm_ptr != NULL ? qm_ptr[rc] : (1 << AOM_QM_BITS);
|
||||
const int iwt = iqm_ptr != NULL ? iqm_ptr[rc] : (1 << AOM_QM_BITS);
|
||||
tmp = clamp(abs_coeff + ROUND_POWER_OF_TWO(round_ptr[rc != 0], log_scale),
|
||||
INT16_MIN, INT16_MAX);
|
||||
tmp32 = (int32_t)((tmp * wt * quant) >> (16 - log_scale + AOM_QM_BITS));
|
||||
qcoeff_ptr[rc] = (tmp32 ^ coeff_sign) - coeff_sign;
|
||||
dequant = (dequant_ptr * iwt + (1 << (AOM_QM_BITS - 1))) >> AOM_QM_BITS;
|
||||
dqcoeff_ptr[rc] = (qcoeff_ptr[rc] * dequant) / (1 << log_scale);
|
||||
if (tmp32) eob = 0;
|
||||
}
|
||||
*eob_ptr = eob + 1;
|
||||
}
|
||||
|
||||
void av1_quantize_dc_facade(const tran_low_t *coeff_ptr, intptr_t n_coeffs,
|
||||
|
|
@ -686,45 +702,18 @@ void av1_quantize_dc_facade(const tran_low_t *coeff_ptr, intptr_t n_coeffs,
|
|||
const SCAN_ORDER *sc, const QUANT_PARAM *qparam) {
|
||||
// obsolete skip_block
|
||||
const int skip_block = 0;
|
||||
(void)sc;
|
||||
assert(qparam->log_scale >= 0 && qparam->log_scale < (2 + CONFIG_TX64X64));
|
||||
#if CONFIG_AOM_QM
|
||||
const qm_val_t *qm_ptr = qparam->qmatrix;
|
||||
const qm_val_t *iqm_ptr = qparam->iqmatrix;
|
||||
#endif // CONFIG_AOM_QM
|
||||
|
||||
(void)sc;
|
||||
|
||||
switch (qparam->log_scale) {
|
||||
case 0:
|
||||
aom_quantize_dc(coeff_ptr, (int)n_coeffs, skip_block, p->round,
|
||||
p->quant_fp[0], qcoeff_ptr, dqcoeff_ptr, pd->dequant[0],
|
||||
eob_ptr
|
||||
#if CONFIG_AOM_QM
|
||||
,
|
||||
qm_ptr, iqm_ptr
|
||||
#else
|
||||
const qm_val_t *qm_ptr = NULL;
|
||||
const qm_val_t *iqm_ptr = NULL;
|
||||
#endif
|
||||
);
|
||||
break;
|
||||
case 1:
|
||||
aom_quantize_dc_32x32(coeff_ptr, skip_block, p->round, p->quant_fp[0],
|
||||
qcoeff_ptr, dqcoeff_ptr, pd->dequant[0], eob_ptr
|
||||
#if CONFIG_AOM_QM
|
||||
,
|
||||
qm_ptr, iqm_ptr
|
||||
#endif
|
||||
);
|
||||
break;
|
||||
#if CONFIG_TX64X64
|
||||
aom_quantize_dc_64x64(coeff_ptr, skip_block, p->round, p->quant_fp[0],
|
||||
qcoeff_ptr, dqcoeff_ptr, pd->dequant[0], eob_ptr
|
||||
#if CONFIG_AOM_QM
|
||||
,
|
||||
qm_ptr, iqm_ptr
|
||||
#endif
|
||||
);
|
||||
case 2: break;
|
||||
#endif // CONFIG_TX64X64
|
||||
default: assert(0);
|
||||
}
|
||||
quantize_dc(coeff_ptr, (int)n_coeffs, skip_block, p->round, p->quant_fp[0],
|
||||
qcoeff_ptr, dqcoeff_ptr, pd->dequant[0], eob_ptr, qm_ptr, iqm_ptr,
|
||||
qparam->log_scale);
|
||||
}
|
||||
|
||||
#if CONFIG_NEW_QUANT
|
||||
|
|
@ -857,29 +846,31 @@ void av1_highbd_quantize_fp_facade(const tran_low_t *coeff_ptr,
|
|||
#if CONFIG_AOM_QM
|
||||
const qm_val_t *qm_ptr = qparam->qmatrix;
|
||||
const qm_val_t *iqm_ptr = qparam->iqmatrix;
|
||||
if (qm_ptr != NULL && iqm_ptr != NULL) {
|
||||
highbd_quantize_fp_helper_c(
|
||||
coeff_ptr, n_coeffs, skip_block, p->zbin, p->round_fp, p->quant_fp,
|
||||
p->quant_shift, qcoeff_ptr, dqcoeff_ptr, pd->dequant, eob_ptr, sc->scan,
|
||||
sc->iscan, qm_ptr, iqm_ptr, qparam->log_scale);
|
||||
} else {
|
||||
#endif // CONFIG_AOM_QM
|
||||
|
||||
if (n_coeffs < 16) {
|
||||
// TODO(jingning): Need SIMD implementation for smaller block size
|
||||
// quantization.
|
||||
av1_highbd_quantize_fp_c(coeff_ptr, n_coeffs, skip_block, p->zbin,
|
||||
p->round_fp, p->quant_fp, p->quant_shift,
|
||||
qcoeff_ptr, dqcoeff_ptr, pd->dequant, eob_ptr,
|
||||
sc->scan, sc->iscan,
|
||||
#if CONFIG_AOM_QM
|
||||
qm_ptr, iqm_ptr,
|
||||
#endif
|
||||
qparam->log_scale);
|
||||
return;
|
||||
}
|
||||
if (n_coeffs < 16) {
|
||||
// TODO(jingning): Need SIMD implementation for smaller block size
|
||||
// quantization.
|
||||
av1_highbd_quantize_fp_c(coeff_ptr, n_coeffs, skip_block, p->zbin,
|
||||
p->round_fp, p->quant_fp, p->quant_shift,
|
||||
qcoeff_ptr, dqcoeff_ptr, pd->dequant, eob_ptr,
|
||||
sc->scan, sc->iscan, qparam->log_scale);
|
||||
return;
|
||||
}
|
||||
|
||||
av1_highbd_quantize_fp(coeff_ptr, n_coeffs, skip_block, p->zbin, p->round_fp,
|
||||
p->quant_fp, p->quant_shift, qcoeff_ptr, dqcoeff_ptr,
|
||||
pd->dequant, eob_ptr, sc->scan, sc->iscan,
|
||||
av1_highbd_quantize_fp(coeff_ptr, n_coeffs, skip_block, p->zbin,
|
||||
p->round_fp, p->quant_fp, p->quant_shift, qcoeff_ptr,
|
||||
dqcoeff_ptr, pd->dequant, eob_ptr, sc->scan,
|
||||
sc->iscan, qparam->log_scale);
|
||||
#if CONFIG_AOM_QM
|
||||
qm_ptr, iqm_ptr,
|
||||
}
|
||||
#endif
|
||||
qparam->log_scale);
|
||||
}
|
||||
|
||||
void av1_highbd_quantize_b_facade(const tran_low_t *coeff_ptr,
|
||||
|
|
@ -894,86 +885,76 @@ void av1_highbd_quantize_b_facade(const tran_low_t *coeff_ptr,
|
|||
#if CONFIG_AOM_QM
|
||||
const qm_val_t *qm_ptr = qparam->qmatrix;
|
||||
const qm_val_t *iqm_ptr = qparam->iqmatrix;
|
||||
if (qm_ptr != NULL && iqm_ptr != NULL) {
|
||||
highbd_quantize_b_helper_c(coeff_ptr, n_coeffs, skip_block, p->zbin,
|
||||
p->round, p->quant, p->quant_shift, qcoeff_ptr,
|
||||
dqcoeff_ptr, pd->dequant, eob_ptr, sc->scan,
|
||||
sc->iscan, qm_ptr, iqm_ptr, qparam->log_scale);
|
||||
} else {
|
||||
#endif // CONFIG_AOM_QM
|
||||
|
||||
switch (qparam->log_scale) {
|
||||
case 0:
|
||||
if (LIKELY(n_coeffs >= 8)) {
|
||||
aom_highbd_quantize_b(coeff_ptr, n_coeffs, skip_block, p->zbin,
|
||||
p->round, p->quant, p->quant_shift, qcoeff_ptr,
|
||||
dqcoeff_ptr, pd->dequant, eob_ptr, sc->scan,
|
||||
sc->iscan
|
||||
#if CONFIG_AOM_QM
|
||||
,
|
||||
qm_ptr, iqm_ptr
|
||||
#endif
|
||||
);
|
||||
} else {
|
||||
// TODO(luoyi): Need SIMD (e.g. sse2) for smaller block size
|
||||
// quantization
|
||||
aom_highbd_quantize_b_c(coeff_ptr, n_coeffs, skip_block, p->zbin,
|
||||
switch (qparam->log_scale) {
|
||||
case 0:
|
||||
if (LIKELY(n_coeffs >= 8)) {
|
||||
aom_highbd_quantize_b(coeff_ptr, n_coeffs, skip_block, p->zbin,
|
||||
p->round, p->quant, p->quant_shift, qcoeff_ptr,
|
||||
dqcoeff_ptr, pd->dequant, eob_ptr, sc->scan,
|
||||
sc->iscan
|
||||
#if CONFIG_AOM_QM
|
||||
,
|
||||
qm_ptr, iqm_ptr
|
||||
#endif
|
||||
);
|
||||
}
|
||||
break;
|
||||
case 1:
|
||||
aom_highbd_quantize_b_32x32(coeff_ptr, n_coeffs, skip_block, p->zbin,
|
||||
sc->iscan);
|
||||
} else {
|
||||
// TODO(luoyi): Need SIMD (e.g. sse2) for smaller block size
|
||||
// quantization
|
||||
aom_highbd_quantize_b_c(coeff_ptr, n_coeffs, skip_block, p->zbin,
|
||||
p->round, p->quant, p->quant_shift,
|
||||
qcoeff_ptr, dqcoeff_ptr, pd->dequant, eob_ptr,
|
||||
sc->scan, sc->iscan
|
||||
#if CONFIG_AOM_QM
|
||||
,
|
||||
qm_ptr, iqm_ptr
|
||||
#endif
|
||||
);
|
||||
break;
|
||||
sc->scan, sc->iscan);
|
||||
}
|
||||
break;
|
||||
case 1:
|
||||
aom_highbd_quantize_b_32x32(coeff_ptr, n_coeffs, skip_block, p->zbin,
|
||||
p->round, p->quant, p->quant_shift,
|
||||
qcoeff_ptr, dqcoeff_ptr, pd->dequant,
|
||||
eob_ptr, sc->scan, sc->iscan);
|
||||
break;
|
||||
#if CONFIG_TX64X64
|
||||
case 2:
|
||||
aom_highbd_quantize_b_64x64(coeff_ptr, n_coeffs, skip_block, p->zbin,
|
||||
p->round, p->quant, p->quant_shift,
|
||||
qcoeff_ptr, dqcoeff_ptr, pd->dequant, eob_ptr,
|
||||
sc->scan, sc->iscan
|
||||
#if CONFIG_AOM_QM
|
||||
,
|
||||
qm_ptr, iqm_ptr
|
||||
#endif
|
||||
);
|
||||
break;
|
||||
case 2:
|
||||
aom_highbd_quantize_b_64x64(coeff_ptr, n_coeffs, skip_block, p->zbin,
|
||||
p->round, p->quant, p->quant_shift,
|
||||
qcoeff_ptr, dqcoeff_ptr, pd->dequant,
|
||||
eob_ptr, sc->scan, sc->iscan);
|
||||
break;
|
||||
#endif // CONFIG_TX64X64
|
||||
default: assert(0);
|
||||
default: assert(0);
|
||||
}
|
||||
#if CONFIG_AOM_QM
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
static INLINE void highbd_quantize_dc(
|
||||
const tran_low_t *coeff_ptr, int n_coeffs, int skip_block,
|
||||
const int16_t *round_ptr, const int16_t quant, tran_low_t *qcoeff_ptr,
|
||||
tran_low_t *dqcoeff_ptr, const int16_t dequant_ptr, uint16_t *eob_ptr,
|
||||
#if CONFIG_AOM_QM
|
||||
const qm_val_t *qm_ptr, const qm_val_t *iqm_ptr,
|
||||
#endif
|
||||
const int log_scale) {
|
||||
const qm_val_t *qm_ptr, const qm_val_t *iqm_ptr, const int log_scale) {
|
||||
int eob = -1;
|
||||
|
||||
memset(qcoeff_ptr, 0, n_coeffs * sizeof(*qcoeff_ptr));
|
||||
memset(dqcoeff_ptr, 0, n_coeffs * sizeof(*dqcoeff_ptr));
|
||||
#if CONFIG_AOM_QM
|
||||
(void)qm_ptr;
|
||||
(void)iqm_ptr;
|
||||
#endif
|
||||
|
||||
if (!skip_block) {
|
||||
const qm_val_t wt = qm_ptr != NULL ? qm_ptr[0] : (1 << AOM_QM_BITS);
|
||||
const qm_val_t iwt = iqm_ptr != NULL ? iqm_ptr[0] : (1 << AOM_QM_BITS);
|
||||
const int coeff = coeff_ptr[0];
|
||||
const int coeff_sign = (coeff >> 31);
|
||||
const int abs_coeff = (coeff ^ coeff_sign) - coeff_sign;
|
||||
const int64_t tmp = abs_coeff + round_ptr[0];
|
||||
const int abs_qcoeff = (int)((tmp * quant) >> (16 - log_scale));
|
||||
const int64_t tmp = abs_coeff + ROUND_POWER_OF_TWO(round_ptr[0], log_scale);
|
||||
const int64_t tmpw = tmp * wt;
|
||||
const int abs_qcoeff =
|
||||
(int)((tmpw * quant) >> (16 - log_scale + AOM_QM_BITS));
|
||||
qcoeff_ptr[0] = (tran_low_t)((abs_qcoeff ^ coeff_sign) - coeff_sign);
|
||||
dqcoeff_ptr[0] = qcoeff_ptr[0] * dequant_ptr / (1 << log_scale);
|
||||
const int dequant =
|
||||
(dequant_ptr * iwt + (1 << (AOM_QM_BITS - 1))) >> AOM_QM_BITS;
|
||||
|
||||
dqcoeff_ptr[0] = (qcoeff_ptr[0] * dequant) / (1 << log_scale);
|
||||
if (abs_qcoeff) eob = 0;
|
||||
}
|
||||
*eob_ptr = eob + 1;
|
||||
|
|
@ -991,17 +972,16 @@ void av1_highbd_quantize_dc_facade(const tran_low_t *coeff_ptr,
|
|||
#if CONFIG_AOM_QM
|
||||
const qm_val_t *qm_ptr = qparam->qmatrix;
|
||||
const qm_val_t *iqm_ptr = qparam->iqmatrix;
|
||||
#else
|
||||
const qm_val_t *qm_ptr = NULL;
|
||||
const qm_val_t *iqm_ptr = NULL;
|
||||
#endif // CONFIG_AOM_QM
|
||||
|
||||
(void)sc;
|
||||
|
||||
highbd_quantize_dc(coeff_ptr, (int)n_coeffs, skip_block, p->round,
|
||||
p->quant_fp[0], qcoeff_ptr, dqcoeff_ptr, pd->dequant[0],
|
||||
eob_ptr,
|
||||
#if CONFIG_AOM_QM
|
||||
qm_ptr, iqm_ptr,
|
||||
#endif
|
||||
qparam->log_scale);
|
||||
eob_ptr, qm_ptr, iqm_ptr, qparam->log_scale);
|
||||
}
|
||||
|
||||
#if CONFIG_NEW_QUANT
|
||||
|
|
@ -1517,61 +1497,16 @@ void av1_highbd_quantize_dc_nuq_facade(
|
|||
}
|
||||
#endif // CONFIG_NEW_QUANT
|
||||
|
||||
void av1_highbd_quantize_fp_c(const tran_low_t *coeff_ptr, intptr_t count,
|
||||
int skip_block, const int16_t *zbin_ptr,
|
||||
const int16_t *round_ptr,
|
||||
const int16_t *quant_ptr,
|
||||
const int16_t *quant_shift_ptr,
|
||||
tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr,
|
||||
const int16_t *dequant_ptr, uint16_t *eob_ptr,
|
||||
const int16_t *scan, const int16_t *iscan,
|
||||
#if CONFIG_AOM_QM
|
||||
const qm_val_t *qm_ptr, const qm_val_t *iqm_ptr,
|
||||
#endif
|
||||
int log_scale) {
|
||||
int i;
|
||||
int eob = -1;
|
||||
const int scale = 1 << log_scale;
|
||||
const int shift = 16 - log_scale;
|
||||
// TODO(jingning) Decide the need of these arguments after the
|
||||
// quantization process is completed.
|
||||
(void)zbin_ptr;
|
||||
(void)quant_shift_ptr;
|
||||
(void)iscan;
|
||||
|
||||
memset(qcoeff_ptr, 0, count * sizeof(*qcoeff_ptr));
|
||||
memset(dqcoeff_ptr, 0, count * sizeof(*dqcoeff_ptr));
|
||||
|
||||
if (!skip_block) {
|
||||
// Quantization pass: All coefficients with index >= zero_flag are
|
||||
// skippable. Note: zero_flag can be zero.
|
||||
for (i = 0; i < count; i++) {
|
||||
const int rc = scan[i];
|
||||
const int coeff = coeff_ptr[rc];
|
||||
#if CONFIG_AOM_QM
|
||||
const qm_val_t wt = qm_ptr[rc];
|
||||
const qm_val_t iwt = iqm_ptr[rc];
|
||||
const int dequant =
|
||||
(dequant_ptr[rc != 0] * iwt + (1 << (AOM_QM_BITS - 1))) >>
|
||||
AOM_QM_BITS;
|
||||
#endif
|
||||
const int coeff_sign = (coeff >> 31);
|
||||
const int abs_coeff = (coeff ^ coeff_sign) - coeff_sign;
|
||||
const int64_t tmp = abs_coeff + (round_ptr[rc != 0] >> log_scale);
|
||||
#if CONFIG_AOM_QM
|
||||
const int abs_qcoeff =
|
||||
(int)((tmp * quant_ptr[rc != 0] * wt) >> (shift + AOM_QM_BITS));
|
||||
qcoeff_ptr[rc] = (tran_low_t)((abs_qcoeff ^ coeff_sign) - coeff_sign);
|
||||
dqcoeff_ptr[rc] = qcoeff_ptr[rc] * dequant / scale;
|
||||
#else
|
||||
const int abs_qcoeff = (int)((tmp * quant_ptr[rc != 0]) >> shift);
|
||||
qcoeff_ptr[rc] = (tran_low_t)((abs_qcoeff ^ coeff_sign) - coeff_sign);
|
||||
dqcoeff_ptr[rc] = qcoeff_ptr[rc] * dequant_ptr[rc != 0] / scale;
|
||||
#endif
|
||||
if (abs_qcoeff) eob = i;
|
||||
}
|
||||
}
|
||||
*eob_ptr = eob + 1;
|
||||
void av1_highbd_quantize_fp_c(
|
||||
const tran_low_t *coeff_ptr, intptr_t count, int skip_block,
|
||||
const int16_t *zbin_ptr, const int16_t *round_ptr, const int16_t *quant_ptr,
|
||||
const int16_t *quant_shift_ptr, tran_low_t *qcoeff_ptr,
|
||||
tran_low_t *dqcoeff_ptr, const int16_t *dequant_ptr, uint16_t *eob_ptr,
|
||||
const int16_t *scan, const int16_t *iscan, int log_scale) {
|
||||
highbd_quantize_fp_helper_c(coeff_ptr, count, skip_block, zbin_ptr, round_ptr,
|
||||
quant_ptr, quant_shift_ptr, qcoeff_ptr,
|
||||
dqcoeff_ptr, dequant_ptr, eob_ptr, scan, iscan,
|
||||
NULL, NULL, log_scale);
|
||||
}
|
||||
|
||||
static void invert_quant(int16_t *quant, int16_t *shift, int d) {
|
||||
|
|
@ -1682,22 +1617,19 @@ void av1_init_plane_quantizers(const AV1_COMP *cpi, MACROBLOCK *x,
|
|||
MACROBLOCKD *const xd = &x->e_mbd;
|
||||
const QUANTS *const quants = &cpi->quants;
|
||||
|
||||
#if CONFIG_DELTA_Q
|
||||
#if CONFIG_EXT_DELTA_Q
|
||||
int current_q_index = AOMMAX(
|
||||
0, AOMMIN(QINDEX_RANGE - 1, cpi->oxcf.deltaq_mode != NO_DELTA_Q
|
||||
? cm->base_qindex + xd->delta_qindex
|
||||
: cm->base_qindex));
|
||||
int current_q_index =
|
||||
AOMMAX(0, AOMMIN(QINDEX_RANGE - 1,
|
||||
cpi->oxcf.deltaq_mode != NO_DELTA_Q
|
||||
? cm->base_qindex + xd->delta_qindex
|
||||
: cm->base_qindex));
|
||||
#else
|
||||
int current_q_index = AOMMAX(
|
||||
0, AOMMIN(QINDEX_RANGE - 1, cm->delta_q_present_flag
|
||||
? cm->base_qindex + xd->delta_qindex
|
||||
: cm->base_qindex));
|
||||
0, AOMMIN(QINDEX_RANGE - 1,
|
||||
cm->delta_q_present_flag ? cm->base_qindex + xd->delta_qindex
|
||||
: cm->base_qindex));
|
||||
#endif
|
||||
const int qindex = av1_get_qindex(&cm->seg, segment_id, current_q_index);
|
||||
#else
|
||||
const int qindex = av1_get_qindex(&cm->seg, segment_id, cm->base_qindex);
|
||||
#endif
|
||||
const int rdmult = av1_compute_rd_mult(cpi, qindex + cm->y_dc_delta_q);
|
||||
int i;
|
||||
#if CONFIG_AOM_QM
|
||||
|
|
|
|||
903
third_party/aom/av1/encoder/bgsprite.c
vendored
903
third_party/aom/av1/encoder/bgsprite.c
vendored
File diff suppressed because it is too large
Load diff
2454
third_party/aom/av1/encoder/bitstream.c
vendored
2454
third_party/aom/av1/encoder/bitstream.c
vendored
File diff suppressed because it is too large
Load diff
9
third_party/aom/av1/encoder/bitstream.h
vendored
9
third_party/aom/av1/encoder/bitstream.h
vendored
|
|
@ -18,12 +18,11 @@ extern "C" {
|
|||
|
||||
#include "av1/encoder/encoder.h"
|
||||
|
||||
struct aom_write_bit_buffer;
|
||||
|
||||
#if CONFIG_REFERENCE_BUFFER
|
||||
void write_sequence_header(
|
||||
#if CONFIG_EXT_TILE
|
||||
AV1_COMMON *const cm,
|
||||
#endif // CONFIG_EXT_TILE
|
||||
SequenceHeader *seq_params);
|
||||
void write_sequence_header(AV1_COMMON *const cm,
|
||||
struct aom_write_bit_buffer *wb);
|
||||
#endif
|
||||
|
||||
void av1_pack_bitstream(AV1_COMP *const cpi, uint8_t *dest, size_t *size);
|
||||
|
|
|
|||
172
third_party/aom/av1/encoder/block.h
vendored
172
third_party/aom/av1/encoder/block.h
vendored
|
|
@ -18,6 +18,10 @@
|
|||
#include "av1/encoder/encint.h"
|
||||
#endif
|
||||
#include "av1/common/mvref_common.h"
|
||||
#include "av1/encoder/hash.h"
|
||||
#if CONFIG_DIST_8X8
|
||||
#include "aom/aomcx.h"
|
||||
#endif
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
|
|
@ -60,28 +64,52 @@ typedef struct macroblock_plane {
|
|||
#endif // CONFIG_NEW_QUANT
|
||||
} MACROBLOCK_PLANE;
|
||||
|
||||
/* The [2] dimension is for whether we skip the EOB node (i.e. if previous
|
||||
* coefficient in this block was zero) or not. */
|
||||
typedef unsigned int av1_coeff_cost[PLANE_TYPES][REF_TYPES][COEF_BANDS][2]
|
||||
[COEFF_CONTEXTS][ENTROPY_TOKENS];
|
||||
typedef int av1_coeff_cost[PLANE_TYPES][REF_TYPES][COEF_BANDS][COEFF_CONTEXTS]
|
||||
[TAIL_TOKENS];
|
||||
|
||||
#if CONFIG_LV_MAP
|
||||
typedef struct {
|
||||
int txb_skip_cost[TXB_SKIP_CONTEXTS][2];
|
||||
int nz_map_cost[SIG_COEF_CONTEXTS][2];
|
||||
int eob_cost[EOB_COEF_CONTEXTS][2];
|
||||
int dc_sign_cost[DC_SIGN_CONTEXTS][2];
|
||||
int base_cost[NUM_BASE_LEVELS][COEFF_BASE_CONTEXTS][2];
|
||||
#if BR_NODE
|
||||
int lps_cost[LEVEL_CONTEXTS][COEFF_BASE_RANGE + 1];
|
||||
int br_cost[BASE_RANGE_SETS][LEVEL_CONTEXTS][2];
|
||||
#else // BR_NODE
|
||||
int lps_cost[LEVEL_CONTEXTS][2];
|
||||
#endif // BR_NODE
|
||||
#if CONFIG_CTX1D
|
||||
int eob_mode_cost[TX_CLASSES][2];
|
||||
int empty_line_cost[TX_CLASSES][EMPTY_LINE_CONTEXTS][2];
|
||||
int hv_eob_cost[TX_CLASSES][HV_EOB_CONTEXTS][2];
|
||||
#endif
|
||||
} LV_MAP_COEFF_COST;
|
||||
|
||||
typedef struct {
|
||||
int_mv ref_mvs[MODE_CTX_REF_FRAMES][MAX_MV_REF_CANDIDATES];
|
||||
int16_t mode_context[MODE_CTX_REF_FRAMES];
|
||||
#if CONFIG_LV_MAP
|
||||
// TODO(angiebird): Reduce the buffer size according to sb_type
|
||||
tran_low_t tcoeff[MAX_MB_PLANE][MAX_SB_SQUARE];
|
||||
uint16_t eobs[MAX_MB_PLANE][MAX_SB_SQUARE / (TX_SIZE_W_MIN * TX_SIZE_H_MIN)];
|
||||
uint8_t txb_skip_ctx[MAX_MB_PLANE]
|
||||
[MAX_SB_SQUARE / (TX_SIZE_W_MIN * TX_SIZE_H_MIN)];
|
||||
int dc_sign_ctx[MAX_MB_PLANE]
|
||||
[MAX_SB_SQUARE / (TX_SIZE_W_MIN * TX_SIZE_H_MIN)];
|
||||
} CB_COEFF_BUFFER;
|
||||
#endif
|
||||
|
||||
typedef struct {
|
||||
int_mv ref_mvs[MODE_CTX_REF_FRAMES][MAX_MV_REF_CANDIDATES];
|
||||
int16_t mode_context[MODE_CTX_REF_FRAMES];
|
||||
#if CONFIG_LV_MAP
|
||||
// TODO(angiebird): Reduce the buffer size according to sb_type
|
||||
tran_low_t *tcoeff[MAX_MB_PLANE];
|
||||
uint16_t *eobs[MAX_MB_PLANE];
|
||||
uint8_t *txb_skip_ctx[MAX_MB_PLANE];
|
||||
int *dc_sign_ctx[MAX_MB_PLANE];
|
||||
#endif
|
||||
uint8_t ref_mv_count[MODE_CTX_REF_FRAMES];
|
||||
CANDIDATE_MV ref_mv_stack[MODE_CTX_REF_FRAMES][MAX_REF_MV_STACK_SIZE];
|
||||
#if CONFIG_EXT_INTER
|
||||
int16_t compound_mode_context[MODE_CTX_REF_FRAMES];
|
||||
#endif // CONFIG_EXT_INTER
|
||||
} MB_MODE_INFO_EXT;
|
||||
|
||||
typedef struct {
|
||||
|
|
@ -91,17 +119,41 @@ typedef struct {
|
|||
int row_max;
|
||||
} MvLimits;
|
||||
|
||||
#if CONFIG_PALETTE
|
||||
typedef struct {
|
||||
uint8_t best_palette_color_map[MAX_SB_SQUARE];
|
||||
float kmeans_data_buf[2 * MAX_SB_SQUARE];
|
||||
} PALETTE_BUFFER;
|
||||
#endif // CONFIG_PALETTE
|
||||
|
||||
typedef struct {
|
||||
TX_TYPE tx_type;
|
||||
TX_SIZE tx_size;
|
||||
#if CONFIG_VAR_TX
|
||||
TX_SIZE min_tx_size;
|
||||
TX_SIZE inter_tx_size[MAX_MIB_SIZE][MAX_MIB_SIZE];
|
||||
uint8_t blk_skip[MAX_MIB_SIZE * MAX_MIB_SIZE * 8];
|
||||
#endif // CONFIG_VAR_TX
|
||||
#if CONFIG_TXK_SEL
|
||||
TX_TYPE txk_type[MAX_SB_SQUARE / (TX_SIZE_W_MIN * TX_SIZE_H_MIN)];
|
||||
#endif // CONFIG_TXK_SEL
|
||||
RD_STATS rd_stats;
|
||||
uint32_t hash_value;
|
||||
} TX_RD_INFO;
|
||||
|
||||
#define RD_RECORD_BUFFER_LEN 8
|
||||
typedef struct {
|
||||
TX_RD_INFO tx_rd_info[RD_RECORD_BUFFER_LEN]; // Circular buffer.
|
||||
int index_start;
|
||||
int num;
|
||||
CRC_CALCULATOR crc_calculator; // Hash function.
|
||||
} TX_RD_RECORD;
|
||||
|
||||
typedef struct macroblock MACROBLOCK;
|
||||
struct macroblock {
|
||||
struct macroblock_plane plane[MAX_MB_PLANE];
|
||||
|
||||
// Save the transform RD search info.
|
||||
TX_RD_RECORD tx_rd_record;
|
||||
|
||||
MACROBLOCKD e_mbd;
|
||||
MB_MODE_INFO_EXT *mbmi_ext;
|
||||
int skip_block;
|
||||
|
|
@ -150,9 +202,7 @@ struct macroblock {
|
|||
uint8_t *left_pred_buf;
|
||||
#endif // CONFIG_MOTION_VAR
|
||||
|
||||
#if CONFIG_PALETTE
|
||||
PALETTE_BUFFER *palette_buffer;
|
||||
#endif // CONFIG_PALETTE
|
||||
|
||||
// These define limits to motion vector components to prevent them
|
||||
// from extending outside the UMV borders
|
||||
|
|
@ -169,8 +219,92 @@ struct macroblock {
|
|||
int skip_chroma_rd;
|
||||
#endif
|
||||
|
||||
// note that token_costs is the cost when eob node is skipped
|
||||
av1_coeff_cost token_costs[TX_SIZES];
|
||||
#if CONFIG_LV_MAP
|
||||
LV_MAP_COEFF_COST coeff_costs[TX_SIZES][PLANE_TYPES];
|
||||
uint16_t cb_offset;
|
||||
#endif
|
||||
|
||||
av1_coeff_cost token_head_costs[TX_SIZES];
|
||||
av1_coeff_cost token_tail_costs[TX_SIZES];
|
||||
|
||||
// mode costs
|
||||
int mbmode_cost[BLOCK_SIZE_GROUPS][INTRA_MODES];
|
||||
int newmv_mode_cost[NEWMV_MODE_CONTEXTS][2];
|
||||
int zeromv_mode_cost[ZEROMV_MODE_CONTEXTS][2];
|
||||
int refmv_mode_cost[REFMV_MODE_CONTEXTS][2];
|
||||
int drl_mode_cost0[DRL_MODE_CONTEXTS][2];
|
||||
|
||||
int inter_compound_mode_cost[INTER_MODE_CONTEXTS][INTER_COMPOUND_MODES];
|
||||
int compound_type_cost[BLOCK_SIZES_ALL][COMPOUND_TYPES];
|
||||
#if CONFIG_COMPOUND_SINGLEREF
|
||||
int inter_singleref_comp_mode_cost[INTER_MODE_CONTEXTS]
|
||||
[INTER_SINGLEREF_COMP_MODES];
|
||||
#endif // CONFIG_COMPOUND_SINGLEREF
|
||||
#if CONFIG_INTERINTRA
|
||||
int interintra_mode_cost[BLOCK_SIZE_GROUPS][INTERINTRA_MODES];
|
||||
#endif // CONFIG_INTERINTRA
|
||||
#if CONFIG_MOTION_VAR || CONFIG_WARPED_MOTION
|
||||
int motion_mode_cost[BLOCK_SIZES_ALL][MOTION_MODES];
|
||||
#if CONFIG_MOTION_VAR && CONFIG_WARPED_MOTION
|
||||
int motion_mode_cost1[BLOCK_SIZES_ALL][2];
|
||||
#if CONFIG_NCOBMC_ADAPT_WEIGHT
|
||||
int motion_mode_cost2[BLOCK_SIZES_ALL][OBMC_FAMILY_MODES];
|
||||
#endif
|
||||
#endif // CONFIG_MOTION_VAR && CONFIG_WARPED_MOTION
|
||||
#if CONFIG_MOTION_VAR && CONFIG_NCOBMC_ADAPT_WEIGHT
|
||||
int ncobmc_mode_cost[ADAPT_OVERLAP_BLOCKS][MAX_NCOBMC_MODES];
|
||||
#endif // CONFIG_MOTION_VAR && CONFIG_NCOBMC_ADAPT_WEIGHT
|
||||
#endif // CONFIG_MOTION_VAR || CONFIG_WARPED_MOTION
|
||||
int intra_uv_mode_cost[INTRA_MODES][UV_INTRA_MODES];
|
||||
int y_mode_costs[INTRA_MODES][INTRA_MODES][INTRA_MODES];
|
||||
int switchable_interp_costs[SWITCHABLE_FILTER_CONTEXTS][SWITCHABLE_FILTERS];
|
||||
#if CONFIG_EXT_PARTITION_TYPES
|
||||
int partition_cost[PARTITION_CONTEXTS + CONFIG_UNPOISON_PARTITION_CTX]
|
||||
[EXT_PARTITION_TYPES];
|
||||
#else
|
||||
int partition_cost[PARTITION_CONTEXTS + CONFIG_UNPOISON_PARTITION_CTX]
|
||||
[PARTITION_TYPES];
|
||||
#endif // CONFIG_EXT_PARTITION_TYPES
|
||||
#if CONFIG_MRC_TX
|
||||
int mrc_mask_inter_cost[PALETTE_SIZES][PALETTE_COLOR_INDEX_CONTEXTS]
|
||||
[PALETTE_COLORS];
|
||||
int mrc_mask_intra_cost[PALETTE_SIZES][PALETTE_COLOR_INDEX_CONTEXTS]
|
||||
[PALETTE_COLORS];
|
||||
#endif // CONFIG_MRC_TX
|
||||
int palette_y_size_cost[PALETTE_BLOCK_SIZES][PALETTE_SIZES];
|
||||
int palette_uv_size_cost[PALETTE_BLOCK_SIZES][PALETTE_SIZES];
|
||||
int palette_y_color_cost[PALETTE_SIZES][PALETTE_COLOR_INDEX_CONTEXTS]
|
||||
[PALETTE_COLORS];
|
||||
int palette_uv_color_cost[PALETTE_SIZES][PALETTE_COLOR_INDEX_CONTEXTS]
|
||||
[PALETTE_COLORS];
|
||||
#if CONFIG_CFL
|
||||
// The rate associated with each alpha codeword
|
||||
int cfl_cost[CFL_JOINT_SIGNS][CFL_PRED_PLANES][CFL_ALPHABET_SIZE];
|
||||
#endif // CONFIG_CFL
|
||||
int tx_size_cost[TX_SIZES - 1][TX_SIZE_CONTEXTS][TX_SIZES];
|
||||
#if CONFIG_EXT_TX
|
||||
#if CONFIG_LGT_FROM_PRED
|
||||
int intra_lgt_cost[LGT_SIZES][INTRA_MODES][2];
|
||||
int inter_lgt_cost[LGT_SIZES][2];
|
||||
#endif
|
||||
int inter_tx_type_costs[EXT_TX_SETS_INTER][EXT_TX_SIZES][TX_TYPES];
|
||||
int intra_tx_type_costs[EXT_TX_SETS_INTRA][EXT_TX_SIZES][INTRA_MODES]
|
||||
[TX_TYPES];
|
||||
#else
|
||||
int intra_tx_type_costs[EXT_TX_SIZES][TX_TYPES][TX_TYPES];
|
||||
int inter_tx_type_costs[EXT_TX_SIZES][TX_TYPES];
|
||||
#endif // CONFIG_EXT_TX
|
||||
#if CONFIG_EXT_INTRA
|
||||
#if CONFIG_INTRA_INTERP
|
||||
int intra_filter_cost[INTRA_FILTERS + 1][INTRA_FILTERS];
|
||||
#endif // CONFIG_INTRA_INTERP
|
||||
#endif // CONFIG_EXT_INTRA
|
||||
#if CONFIG_LOOP_RESTORATION
|
||||
int switchable_restore_cost[RESTORE_SWITCHABLE_TYPES];
|
||||
#endif // CONFIG_LOOP_RESTORATION
|
||||
#if CONFIG_INTRABC
|
||||
int intrabc_cost[2];
|
||||
#endif // CONFIG_INTRABC
|
||||
|
||||
int optimize;
|
||||
|
||||
|
|
@ -206,6 +340,8 @@ struct macroblock {
|
|||
int pvq_coded; // Indicates whether pvq_info needs be stored to tokenize
|
||||
#endif
|
||||
#if CONFIG_DIST_8X8
|
||||
int using_dist_8x8;
|
||||
aom_tune_metric tune_metric;
|
||||
#if CONFIG_CB4X4
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
DECLARE_ALIGNED(16, uint16_t, decoded_8x8[8 * 8]);
|
||||
|
|
@ -214,10 +350,6 @@ struct macroblock {
|
|||
#endif
|
||||
#endif // CONFIG_CB4X4
|
||||
#endif // CONFIG_DIST_8X8
|
||||
#if CONFIG_CFL
|
||||
// Whether luma needs to be stored during RDO.
|
||||
int cfl_store_y;
|
||||
#endif
|
||||
};
|
||||
|
||||
#ifdef __cplusplus
|
||||
|
|
|
|||
143
third_party/aom/av1/encoder/context_tree.c
vendored
143
third_party/aom/av1/encoder/context_tree.c
vendored
|
|
@ -22,19 +22,14 @@ static const BLOCK_SIZE square[MAX_SB_SIZE_LOG2 - 1] = {
|
|||
#endif // CONFIG_EXT_PARTITION
|
||||
};
|
||||
|
||||
static void alloc_mode_context(AV1_COMMON *cm, int num_4x4_blk,
|
||||
static void alloc_mode_context(AV1_COMMON *cm, int num_pix,
|
||||
#if CONFIG_EXT_PARTITION_TYPES
|
||||
PARTITION_TYPE partition,
|
||||
#endif
|
||||
PICK_MODE_CONTEXT *ctx) {
|
||||
const int num_blk = (num_4x4_blk < 4 ? 4 : num_4x4_blk);
|
||||
const int num_pix = num_blk * tx_size_2d[0];
|
||||
int i;
|
||||
#if CONFIG_CB4X4 && CONFIG_VAR_TX
|
||||
ctx->num_4x4_blk = num_blk / 4;
|
||||
#else
|
||||
const int num_blk = num_pix / 16;
|
||||
ctx->num_4x4_blk = num_blk;
|
||||
#endif
|
||||
|
||||
#if CONFIG_EXT_PARTITION_TYPES
|
||||
ctx->partition = partition;
|
||||
|
|
@ -64,13 +59,15 @@ static void alloc_mode_context(AV1_COMMON *cm, int num_4x4_blk,
|
|||
#endif
|
||||
}
|
||||
|
||||
#if CONFIG_PALETTE
|
||||
for (i = 0; i < 2; ++i) {
|
||||
CHECK_MEM_ERROR(
|
||||
cm, ctx->color_index_map[i],
|
||||
aom_memalign(32, num_pix * sizeof(*ctx->color_index_map[i])));
|
||||
}
|
||||
#endif // CONFIG_PALETTE
|
||||
#if CONFIG_MRC_TX
|
||||
CHECK_MEM_ERROR(cm, ctx->mrc_mask,
|
||||
aom_memalign(32, num_pix * sizeof(*ctx->mrc_mask)));
|
||||
#endif // CONFIG_MRC_TX
|
||||
}
|
||||
|
||||
static void free_mode_context(PICK_MODE_CONTEXT *ctx) {
|
||||
|
|
@ -98,80 +95,63 @@ static void free_mode_context(PICK_MODE_CONTEXT *ctx) {
|
|||
#endif
|
||||
}
|
||||
|
||||
#if CONFIG_PALETTE
|
||||
for (i = 0; i < 2; ++i) {
|
||||
aom_free(ctx->color_index_map[i]);
|
||||
ctx->color_index_map[i] = 0;
|
||||
}
|
||||
#endif // CONFIG_PALETTE
|
||||
#if CONFIG_MRC_TX
|
||||
aom_free(ctx->mrc_mask);
|
||||
ctx->mrc_mask = 0;
|
||||
#endif // CONFIG_MRC_TX
|
||||
}
|
||||
|
||||
static void alloc_tree_contexts(AV1_COMMON *cm, PC_TREE *tree,
|
||||
int num_4x4_blk) {
|
||||
static void alloc_tree_contexts(AV1_COMMON *cm, PC_TREE *tree, int num_pix) {
|
||||
#if CONFIG_EXT_PARTITION_TYPES
|
||||
alloc_mode_context(cm, num_4x4_blk, PARTITION_NONE, &tree->none);
|
||||
alloc_mode_context(cm, num_4x4_blk / 2, PARTITION_HORZ, &tree->horizontal[0]);
|
||||
alloc_mode_context(cm, num_4x4_blk / 2, PARTITION_VERT, &tree->vertical[0]);
|
||||
alloc_mode_context(cm, num_4x4_blk / 2, PARTITION_VERT, &tree->horizontal[1]);
|
||||
alloc_mode_context(cm, num_4x4_blk / 2, PARTITION_VERT, &tree->vertical[1]);
|
||||
alloc_mode_context(cm, num_pix, PARTITION_NONE, &tree->none);
|
||||
alloc_mode_context(cm, num_pix / 2, PARTITION_HORZ, &tree->horizontal[0]);
|
||||
alloc_mode_context(cm, num_pix / 2, PARTITION_VERT, &tree->vertical[0]);
|
||||
alloc_mode_context(cm, num_pix / 2, PARTITION_VERT, &tree->horizontal[1]);
|
||||
alloc_mode_context(cm, num_pix / 2, PARTITION_VERT, &tree->vertical[1]);
|
||||
|
||||
alloc_mode_context(cm, num_4x4_blk / 4, PARTITION_HORZ_A,
|
||||
&tree->horizontala[0]);
|
||||
alloc_mode_context(cm, num_4x4_blk / 4, PARTITION_HORZ_A,
|
||||
&tree->horizontala[1]);
|
||||
alloc_mode_context(cm, num_4x4_blk / 2, PARTITION_HORZ_A,
|
||||
&tree->horizontala[2]);
|
||||
alloc_mode_context(cm, num_4x4_blk / 2, PARTITION_HORZ_B,
|
||||
&tree->horizontalb[0]);
|
||||
alloc_mode_context(cm, num_4x4_blk / 4, PARTITION_HORZ_B,
|
||||
&tree->horizontalb[1]);
|
||||
alloc_mode_context(cm, num_4x4_blk / 4, PARTITION_HORZ_B,
|
||||
&tree->horizontalb[2]);
|
||||
alloc_mode_context(cm, num_4x4_blk / 4, PARTITION_VERT_A,
|
||||
&tree->verticala[0]);
|
||||
alloc_mode_context(cm, num_4x4_blk / 4, PARTITION_VERT_A,
|
||||
&tree->verticala[1]);
|
||||
alloc_mode_context(cm, num_4x4_blk / 2, PARTITION_VERT_A,
|
||||
&tree->verticala[2]);
|
||||
alloc_mode_context(cm, num_4x4_blk / 2, PARTITION_VERT_B,
|
||||
&tree->verticalb[0]);
|
||||
alloc_mode_context(cm, num_4x4_blk / 4, PARTITION_VERT_B,
|
||||
&tree->verticalb[1]);
|
||||
alloc_mode_context(cm, num_4x4_blk / 4, PARTITION_VERT_B,
|
||||
&tree->verticalb[2]);
|
||||
alloc_mode_context(cm, num_pix / 4, PARTITION_HORZ_A, &tree->horizontala[0]);
|
||||
alloc_mode_context(cm, num_pix / 4, PARTITION_HORZ_A, &tree->horizontala[1]);
|
||||
alloc_mode_context(cm, num_pix / 2, PARTITION_HORZ_A, &tree->horizontala[2]);
|
||||
alloc_mode_context(cm, num_pix / 2, PARTITION_HORZ_B, &tree->horizontalb[0]);
|
||||
alloc_mode_context(cm, num_pix / 4, PARTITION_HORZ_B, &tree->horizontalb[1]);
|
||||
alloc_mode_context(cm, num_pix / 4, PARTITION_HORZ_B, &tree->horizontalb[2]);
|
||||
alloc_mode_context(cm, num_pix / 4, PARTITION_VERT_A, &tree->verticala[0]);
|
||||
alloc_mode_context(cm, num_pix / 4, PARTITION_VERT_A, &tree->verticala[1]);
|
||||
alloc_mode_context(cm, num_pix / 2, PARTITION_VERT_A, &tree->verticala[2]);
|
||||
alloc_mode_context(cm, num_pix / 2, PARTITION_VERT_B, &tree->verticalb[0]);
|
||||
alloc_mode_context(cm, num_pix / 4, PARTITION_VERT_B, &tree->verticalb[1]);
|
||||
alloc_mode_context(cm, num_pix / 4, PARTITION_VERT_B, &tree->verticalb[2]);
|
||||
for (int i = 0; i < 4; ++i) {
|
||||
alloc_mode_context(cm, num_4x4_blk / 4, PARTITION_HORZ_4,
|
||||
alloc_mode_context(cm, num_pix / 4, PARTITION_HORZ_4,
|
||||
&tree->horizontal4[i]);
|
||||
alloc_mode_context(cm, num_4x4_blk / 4, PARTITION_HORZ_4,
|
||||
&tree->vertical4[i]);
|
||||
alloc_mode_context(cm, num_pix / 4, PARTITION_HORZ_4, &tree->vertical4[i]);
|
||||
}
|
||||
#if CONFIG_SUPERTX
|
||||
alloc_mode_context(cm, num_4x4_blk, PARTITION_HORZ,
|
||||
&tree->horizontal_supertx);
|
||||
alloc_mode_context(cm, num_4x4_blk, PARTITION_VERT, &tree->vertical_supertx);
|
||||
alloc_mode_context(cm, num_4x4_blk, PARTITION_SPLIT, &tree->split_supertx);
|
||||
alloc_mode_context(cm, num_4x4_blk, PARTITION_HORZ_A,
|
||||
&tree->horizontala_supertx);
|
||||
alloc_mode_context(cm, num_4x4_blk, PARTITION_HORZ_B,
|
||||
&tree->horizontalb_supertx);
|
||||
alloc_mode_context(cm, num_4x4_blk, PARTITION_VERT_A,
|
||||
&tree->verticala_supertx);
|
||||
alloc_mode_context(cm, num_4x4_blk, PARTITION_VERT_B,
|
||||
&tree->verticalb_supertx);
|
||||
alloc_mode_context(cm, num_pix, PARTITION_HORZ, &tree->horizontal_supertx);
|
||||
alloc_mode_context(cm, num_pix, PARTITION_VERT, &tree->vertical_supertx);
|
||||
alloc_mode_context(cm, num_pix, PARTITION_SPLIT, &tree->split_supertx);
|
||||
alloc_mode_context(cm, num_pix, PARTITION_HORZ_A, &tree->horizontala_supertx);
|
||||
alloc_mode_context(cm, num_pix, PARTITION_HORZ_B, &tree->horizontalb_supertx);
|
||||
alloc_mode_context(cm, num_pix, PARTITION_VERT_A, &tree->verticala_supertx);
|
||||
alloc_mode_context(cm, num_pix, PARTITION_VERT_B, &tree->verticalb_supertx);
|
||||
#endif // CONFIG_SUPERTX
|
||||
#else
|
||||
alloc_mode_context(cm, num_4x4_blk, &tree->none);
|
||||
alloc_mode_context(cm, num_4x4_blk / 2, &tree->horizontal[0]);
|
||||
alloc_mode_context(cm, num_4x4_blk / 2, &tree->vertical[0]);
|
||||
alloc_mode_context(cm, num_pix, &tree->none);
|
||||
alloc_mode_context(cm, num_pix / 2, &tree->horizontal[0]);
|
||||
alloc_mode_context(cm, num_pix / 2, &tree->vertical[0]);
|
||||
#if CONFIG_SUPERTX
|
||||
alloc_mode_context(cm, num_4x4_blk, &tree->horizontal_supertx);
|
||||
alloc_mode_context(cm, num_4x4_blk, &tree->vertical_supertx);
|
||||
alloc_mode_context(cm, num_4x4_blk, &tree->split_supertx);
|
||||
alloc_mode_context(cm, num_pix, &tree->horizontal_supertx);
|
||||
alloc_mode_context(cm, num_pix, &tree->vertical_supertx);
|
||||
alloc_mode_context(cm, num_pix, &tree->split_supertx);
|
||||
#endif
|
||||
|
||||
if (num_4x4_blk > 4) {
|
||||
alloc_mode_context(cm, num_4x4_blk / 2, &tree->horizontal[1]);
|
||||
alloc_mode_context(cm, num_4x4_blk / 2, &tree->vertical[1]);
|
||||
if (num_pix > 16) {
|
||||
alloc_mode_context(cm, num_pix / 2, &tree->horizontal[1]);
|
||||
alloc_mode_context(cm, num_pix / 2, &tree->vertical[1]);
|
||||
} else {
|
||||
memset(&tree->horizontal[1], 0, sizeof(tree->horizontal[1]));
|
||||
memset(&tree->vertical[1], 0, sizeof(tree->vertical[1]));
|
||||
|
|
@ -217,8 +197,6 @@ static void free_tree_contexts(PC_TREE *tree) {
|
|||
// represents the state of our search.
|
||||
void av1_setup_pc_tree(AV1_COMMON *cm, ThreadData *td) {
|
||||
int i, j;
|
||||
// TODO(jingning): The pc_tree allocation is redundant. We can take out all
|
||||
// the leaf nodes after cb4x4 mode is enabled.
|
||||
#if CONFIG_CB4X4
|
||||
#if CONFIG_EXT_PARTITION
|
||||
const int tree_nodes_inc = 1024;
|
||||
|
|
@ -239,20 +217,21 @@ void av1_setup_pc_tree(AV1_COMMON *cm, ThreadData *td) {
|
|||
#endif // CONFIG_EXT_PARTITION
|
||||
int pc_tree_index = 0;
|
||||
PC_TREE *this_pc;
|
||||
PICK_MODE_CONTEXT *this_leaf;
|
||||
int square_index = 1;
|
||||
int nodes;
|
||||
|
||||
#if !CONFIG_CB4X4
|
||||
aom_free(td->leaf_tree);
|
||||
CHECK_MEM_ERROR(cm, td->leaf_tree,
|
||||
aom_calloc(leaf_nodes, sizeof(*td->leaf_tree)));
|
||||
PICK_MODE_CONTEXT *this_leaf = &td->leaf_tree[0];
|
||||
#endif
|
||||
aom_free(td->pc_tree);
|
||||
CHECK_MEM_ERROR(cm, td->pc_tree,
|
||||
aom_calloc(tree_nodes, sizeof(*td->pc_tree)));
|
||||
|
||||
this_pc = &td->pc_tree[0];
|
||||
this_leaf = &td->leaf_tree[0];
|
||||
|
||||
#if !CONFIG_CB4X4
|
||||
// 4x4 blocks smaller than 8x8 but in the same 8x8 block share the same
|
||||
// context so we only need to allocate 1 for each 8x8 block.
|
||||
for (i = 0; i < leaf_nodes; ++i) {
|
||||
|
|
@ -262,6 +241,7 @@ void av1_setup_pc_tree(AV1_COMMON *cm, ThreadData *td) {
|
|||
alloc_mode_context(cm, 16, &td->leaf_tree[i]);
|
||||
#endif
|
||||
}
|
||||
#endif
|
||||
|
||||
// Sets up all the leaf nodes in the tree.
|
||||
for (pc_tree_index = 0; pc_tree_index < leaf_nodes; ++pc_tree_index) {
|
||||
|
|
@ -272,8 +252,10 @@ void av1_setup_pc_tree(AV1_COMMON *cm, ThreadData *td) {
|
|||
#else
|
||||
alloc_tree_contexts(cm, tree, 4);
|
||||
#endif
|
||||
#if !CONFIG_CB4X4
|
||||
tree->leaf_split[0] = this_leaf++;
|
||||
for (j = 1; j < 4; j++) tree->leaf_split[j] = tree->leaf_split[0];
|
||||
#endif
|
||||
}
|
||||
|
||||
// Each node has 4 leaf nodes, fill each block_size level of the tree
|
||||
|
|
@ -311,29 +293,28 @@ void av1_free_pc_tree(ThreadData *td) {
|
|||
#else
|
||||
const int tree_nodes_inc = 256;
|
||||
#endif // CONFIG_EXT_PARTITION
|
||||
const int leaf_factor = 4;
|
||||
#else
|
||||
const int tree_nodes_inc = 0;
|
||||
const int leaf_factor = 1;
|
||||
#endif
|
||||
|
||||
#if CONFIG_EXT_PARTITION
|
||||
const int leaf_nodes = 256 * leaf_factor;
|
||||
const int tree_nodes = tree_nodes_inc + 256 + 64 + 16 + 4 + 1;
|
||||
#else
|
||||
const int leaf_nodes = 64 * leaf_factor;
|
||||
const int tree_nodes = tree_nodes_inc + 64 + 16 + 4 + 1;
|
||||
#endif // CONFIG_EXT_PARTITION
|
||||
int i;
|
||||
|
||||
// Set up all 4x4 mode contexts
|
||||
for (i = 0; i < leaf_nodes; ++i) free_mode_context(&td->leaf_tree[i]);
|
||||
|
||||
// Sets up all the leaf nodes in the tree.
|
||||
for (i = 0; i < tree_nodes; ++i) free_tree_contexts(&td->pc_tree[i]);
|
||||
|
||||
aom_free(td->pc_tree);
|
||||
td->pc_tree = NULL;
|
||||
#if !CONFIG_CB4X4
|
||||
const int leaf_factor = 1;
|
||||
#if CONFIG_EXT_PARTITION
|
||||
const int leaf_nodes = 256 * leaf_factor;
|
||||
#else
|
||||
const int leaf_nodes = 64 * leaf_factor;
|
||||
#endif // CONFIG_EXT_PARTITION
|
||||
for (i = 0; i < leaf_nodes; ++i) free_mode_context(&td->leaf_tree[i]);
|
||||
aom_free(td->leaf_tree);
|
||||
td->leaf_tree = NULL;
|
||||
#endif
|
||||
}
|
||||
|
|
|
|||
6
third_party/aom/av1/encoder/context_tree.h
vendored
6
third_party/aom/av1/encoder/context_tree.h
vendored
|
|
@ -27,9 +27,10 @@ struct ThreadData;
|
|||
typedef struct {
|
||||
MODE_INFO mic;
|
||||
MB_MODE_INFO_EXT mbmi_ext;
|
||||
#if CONFIG_PALETTE
|
||||
uint8_t *color_index_map[2];
|
||||
#endif // CONFIG_PALETTE
|
||||
#if CONFIG_MRC_TX
|
||||
uint8_t *mrc_mask;
|
||||
#endif // CONFIG_MRC_TX
|
||||
#if CONFIG_VAR_TX
|
||||
uint8_t *blk_skip[MAX_MB_PLANE];
|
||||
#endif
|
||||
|
|
@ -84,6 +85,7 @@ typedef struct PC_TREE {
|
|||
PICK_MODE_CONTEXT horizontal4[4];
|
||||
PICK_MODE_CONTEXT vertical4[4];
|
||||
#endif
|
||||
// TODO(jingning): remove leaf_split[] when cb4x4 experiment flag is removed.
|
||||
union {
|
||||
struct PC_TREE *split[4];
|
||||
PICK_MODE_CONTEXT *leaf_split[4];
|
||||
|
|
|
|||
888
third_party/aom/av1/encoder/dct.c
vendored
888
third_party/aom/av1/encoder/dct.c
vendored
File diff suppressed because it is too large
Load diff
1878
third_party/aom/av1/encoder/encodeframe.c
vendored
1878
third_party/aom/av1/encoder/encodeframe.c
vendored
File diff suppressed because it is too large
Load diff
1
third_party/aom/av1/encoder/encodeframe.h
vendored
1
third_party/aom/av1/encoder/encodeframe.h
vendored
|
|
@ -41,7 +41,6 @@ void av1_update_tx_type_count(const struct AV1Common *cm, MACROBLOCKD *xd,
|
|||
#endif
|
||||
BLOCK_SIZE bsize, TX_SIZE tx_size,
|
||||
FRAME_COUNTS *counts);
|
||||
|
||||
#ifdef __cplusplus
|
||||
} // extern "C"
|
||||
#endif
|
||||
|
|
|
|||
797
third_party/aom/av1/encoder/encodemb.c
vendored
797
third_party/aom/av1/encoder/encodemb.c
vendored
File diff suppressed because it is too large
Load diff
15
third_party/aom/av1/encoder/encodemb.h
vendored
15
third_party/aom/av1/encoder/encodemb.h
vendored
|
|
@ -56,15 +56,17 @@ void av1_xform_quant(const AV1_COMMON *cm, MACROBLOCK *x, int plane, int block,
|
|||
int av1_optimize_b(const AV1_COMMON *cm, MACROBLOCK *mb, int plane, int blk_row,
|
||||
int blk_col, int block, BLOCK_SIZE plane_bsize,
|
||||
TX_SIZE tx_size, const ENTROPY_CONTEXT *a,
|
||||
const ENTROPY_CONTEXT *l);
|
||||
const ENTROPY_CONTEXT *l, int fast_mode);
|
||||
|
||||
void av1_subtract_txb(MACROBLOCK *x, int plane, BLOCK_SIZE plane_bsize,
|
||||
int blk_col, int blk_row, TX_SIZE tx_size);
|
||||
|
||||
void av1_subtract_plane(MACROBLOCK *x, BLOCK_SIZE bsize, int plane);
|
||||
|
||||
#if !CONFIG_PVQ
|
||||
void av1_set_txb_context(MACROBLOCK *x, int plane, int block, TX_SIZE tx_size,
|
||||
ENTROPY_CONTEXT *a, ENTROPY_CONTEXT *l);
|
||||
#endif
|
||||
|
||||
void av1_encode_block_intra(int plane, int block, int blk_row, int blk_col,
|
||||
BLOCK_SIZE plane_bsize, TX_SIZE tx_size, void *arg);
|
||||
|
|
@ -79,7 +81,7 @@ PVQ_SKIP_TYPE av1_pvq_encode_helper(MACROBLOCK *x, tran_low_t *const coeff,
|
|||
tran_low_t *ref_coeff,
|
||||
tran_low_t *const dqcoeff, uint16_t *eob,
|
||||
const int16_t *quant, int plane,
|
||||
int tx_size, TX_TYPE tx_type, int *rate,
|
||||
TX_SIZE tx_size, TX_TYPE tx_type, int *rate,
|
||||
int speed, PVQ_INFO *pvq_info);
|
||||
|
||||
void av1_store_pvq_enc_info(PVQ_INFO *pvq_info, int *qg, int *theta, int *k,
|
||||
|
|
@ -87,15 +89,6 @@ void av1_store_pvq_enc_info(PVQ_INFO *pvq_info, int *qg, int *theta, int *k,
|
|||
int *size, int skip_rest, int skip_dir, int bs);
|
||||
#endif
|
||||
|
||||
#if CONFIG_DPCM_INTRA
|
||||
void av1_encode_block_intra_dpcm(const AV1_COMMON *cm, MACROBLOCK *x,
|
||||
PREDICTION_MODE mode, int plane, int block,
|
||||
int blk_row, int blk_col,
|
||||
BLOCK_SIZE plane_bsize, TX_SIZE tx_size,
|
||||
TX_TYPE tx_type, ENTROPY_CONTEXT *ta,
|
||||
ENTROPY_CONTEXT *tl, int8_t *skip);
|
||||
#endif // CONFIG_DPCM_INTRA
|
||||
|
||||
#ifdef __cplusplus
|
||||
} // extern "C"
|
||||
#endif
|
||||
|
|
|
|||
115
third_party/aom/av1/encoder/encodemv.c
vendored
115
third_party/aom/av1/encoder/encodemv.c
vendored
|
|
@ -62,17 +62,22 @@ static void encode_mv_component(aom_writer *w, int comp, nmv_component *mvcomp,
|
|||
} else {
|
||||
int i;
|
||||
const int n = mv_class + CLASS0_BITS - 1; // number of bits
|
||||
#if CONFIG_NEW_MULTISYMBOL
|
||||
for (i = 0; i < n; ++i)
|
||||
aom_write_symbol(w, (d >> i) & 1, mvcomp->bits_cdf[(i + 1) / 2], 2);
|
||||
#else
|
||||
for (i = 0; i < n; ++i) aom_write(w, (d >> i) & 1, mvcomp->bits[i]);
|
||||
#endif
|
||||
}
|
||||
|
||||
// Fractional bits
|
||||
#if CONFIG_INTRABC
|
||||
#if CONFIG_INTRABC || CONFIG_AMVR
|
||||
if (precision > MV_SUBPEL_NONE)
|
||||
#endif // CONFIG_INTRABC
|
||||
#endif // CONFIG_INTRABC || CONFIG_AMVR
|
||||
{
|
||||
aom_write_symbol(w, fr, mv_class == MV_CLASS_0 ? mvcomp->class0_fp_cdf[d]
|
||||
: mvcomp->fp_cdf,
|
||||
MV_FP_SIZE);
|
||||
aom_write_symbol(
|
||||
w, fr,
|
||||
mv_class == MV_CLASS_0 ? mvcomp->class0_fp_cdf[d] : mvcomp->fp_cdf,
|
||||
MV_FP_SIZE);
|
||||
}
|
||||
|
||||
// High precision bit
|
||||
|
|
@ -129,9 +134,9 @@ static void build_nmv_component_cost_table(int *mvcost,
|
|||
const int b = c + CLASS0_BITS - 1; /* number of bits */
|
||||
for (i = 0; i < b; ++i) cost += bits_cost[i][((d >> i) & 1)];
|
||||
}
|
||||
#if CONFIG_INTRABC
|
||||
#if CONFIG_INTRABC || CONFIG_AMVR
|
||||
if (precision > MV_SUBPEL_NONE)
|
||||
#endif // CONFIG_INTRABC
|
||||
#endif // CONFIG_INTRABC || CONFIG_AMVR
|
||||
{
|
||||
if (c == MV_CLASS_0) {
|
||||
cost += class0_fp_cost[d][f];
|
||||
|
|
@ -165,6 +170,11 @@ void av1_write_nmv_probs(AV1_COMMON *cm, int usehp, aom_writer *w,
|
|||
nmv_context_counts *const nmv_counts) {
|
||||
int i;
|
||||
int nmv_ctx = 0;
|
||||
#if CONFIG_AMVR
|
||||
if (cm->cur_frame_mv_precision_level) {
|
||||
return;
|
||||
}
|
||||
#endif
|
||||
for (nmv_ctx = 0; nmv_ctx < NMV_CONTEXTS; ++nmv_ctx) {
|
||||
nmv_context *const mvc = &cm->fc->nmvc[nmv_ctx];
|
||||
nmv_context_counts *const counts = &nmv_counts[nmv_ctx];
|
||||
|
|
@ -184,6 +194,11 @@ void av1_encode_mv(AV1_COMP *cpi, aom_writer *w, const MV *mv, const MV *ref,
|
|||
nmv_context *mvctx, int usehp) {
|
||||
const MV diff = { mv->row - ref->row, mv->col - ref->col };
|
||||
const MV_JOINT_TYPE j = av1_get_mv_joint(&diff);
|
||||
#if CONFIG_AMVR
|
||||
if (cpi->common.cur_frame_mv_precision_level) {
|
||||
usehp = MV_SUBPEL_NONE;
|
||||
}
|
||||
#endif
|
||||
aom_write_symbol(w, j, mvctx->joint_cdf, MV_JOINTS);
|
||||
if (mv_joint_vertical(j))
|
||||
encode_mv_component(w, diff.row, &mvctx->comps[0], usehp);
|
||||
|
|
@ -222,10 +237,14 @@ void av1_build_nmv_cost_table(int *mvjoint, int *mvcost[2],
|
|||
build_nmv_component_cost_table(mvcost[1], &ctx->comps[1], precision);
|
||||
}
|
||||
|
||||
#if CONFIG_EXT_INTER
|
||||
static void inc_mvs(const MB_MODE_INFO *mbmi, const MB_MODE_INFO_EXT *mbmi_ext,
|
||||
const int_mv mvs[2], const int_mv pred_mvs[2],
|
||||
nmv_context_counts *nmv_counts) {
|
||||
nmv_context_counts *nmv_counts
|
||||
#if CONFIG_AMVR
|
||||
,
|
||||
MvSubpelPrecision precision
|
||||
#endif
|
||||
) {
|
||||
int i;
|
||||
PREDICTION_MODE mode = mbmi->mode;
|
||||
|
||||
|
|
@ -240,7 +259,11 @@ static void inc_mvs(const MB_MODE_INFO *mbmi, const MB_MODE_INFO_EXT *mbmi_ext,
|
|||
mbmi_ext->ref_mv_stack[rf_type], i, mbmi->ref_mv_idx);
|
||||
nmv_context_counts *counts = &nmv_counts[nmv_ctx];
|
||||
(void)pred_mvs;
|
||||
#if CONFIG_AMVR
|
||||
av1_inc_mv(&diff, counts, precision);
|
||||
#else
|
||||
av1_inc_mv(&diff, counts, 1);
|
||||
#endif
|
||||
}
|
||||
} else if (mode == NEAREST_NEWMV || mode == NEAR_NEWMV) {
|
||||
const MV *ref = &mbmi_ext->ref_mvs[mbmi->ref_frame[1]][0].as_mv;
|
||||
|
|
@ -251,7 +274,11 @@ static void inc_mvs(const MB_MODE_INFO *mbmi, const MB_MODE_INFO_EXT *mbmi_ext,
|
|||
av1_nmv_ctx(mbmi_ext->ref_mv_count[rf_type],
|
||||
mbmi_ext->ref_mv_stack[rf_type], 1, mbmi->ref_mv_idx);
|
||||
nmv_context_counts *counts = &nmv_counts[nmv_ctx];
|
||||
#if CONFIG_AMVR
|
||||
av1_inc_mv(&diff, counts, precision);
|
||||
#else
|
||||
av1_inc_mv(&diff, counts, 1);
|
||||
#endif
|
||||
} else if (mode == NEW_NEARESTMV || mode == NEW_NEARMV) {
|
||||
const MV *ref = &mbmi_ext->ref_mvs[mbmi->ref_frame[0]][0].as_mv;
|
||||
const MV diff = { mvs[0].as_mv.row - ref->row,
|
||||
|
|
@ -261,7 +288,11 @@ static void inc_mvs(const MB_MODE_INFO *mbmi, const MB_MODE_INFO_EXT *mbmi_ext,
|
|||
av1_nmv_ctx(mbmi_ext->ref_mv_count[rf_type],
|
||||
mbmi_ext->ref_mv_stack[rf_type], 0, mbmi->ref_mv_idx);
|
||||
nmv_context_counts *counts = &nmv_counts[nmv_ctx];
|
||||
#if CONFIG_AMVR
|
||||
av1_inc_mv(&diff, counts, precision);
|
||||
#else
|
||||
av1_inc_mv(&diff, counts, 1);
|
||||
#endif
|
||||
#if CONFIG_COMPOUND_SINGLEREF
|
||||
} else {
|
||||
assert( // mode == SR_NEAREST_NEWMV ||
|
||||
|
|
@ -288,7 +319,12 @@ static void inc_mvs(const MB_MODE_INFO *mbmi, const MB_MODE_INFO_EXT *mbmi_ext,
|
|||
|
||||
static void inc_mvs_sub8x8(const MODE_INFO *mi, int block, const int_mv mvs[2],
|
||||
const MB_MODE_INFO_EXT *mbmi_ext,
|
||||
nmv_context_counts *nmv_counts) {
|
||||
nmv_context_counts *nmv_counts
|
||||
#if CONFIG_AMVR
|
||||
,
|
||||
MvSubpelPrecision precision
|
||||
#endif
|
||||
) {
|
||||
int i;
|
||||
PREDICTION_MODE mode = mi->bmi[block].as_mode;
|
||||
const MB_MODE_INFO *mbmi = &mi->mbmi;
|
||||
|
|
@ -303,7 +339,11 @@ static void inc_mvs_sub8x8(const MODE_INFO *mi, int block, const int_mv mvs[2],
|
|||
av1_nmv_ctx(mbmi_ext->ref_mv_count[rf_type],
|
||||
mbmi_ext->ref_mv_stack[rf_type], i, mbmi->ref_mv_idx);
|
||||
nmv_context_counts *counts = &nmv_counts[nmv_ctx];
|
||||
#if CONFIG_AMVR
|
||||
av1_inc_mv(&diff, counts, precision);
|
||||
#else
|
||||
av1_inc_mv(&diff, counts, 1);
|
||||
#endif
|
||||
}
|
||||
} else if (mode == NEAREST_NEWMV || mode == NEAR_NEWMV) {
|
||||
const MV *ref = &mi->bmi[block].ref_mv[1].as_mv;
|
||||
|
|
@ -314,7 +354,11 @@ static void inc_mvs_sub8x8(const MODE_INFO *mi, int block, const int_mv mvs[2],
|
|||
av1_nmv_ctx(mbmi_ext->ref_mv_count[rf_type],
|
||||
mbmi_ext->ref_mv_stack[rf_type], 1, mbmi->ref_mv_idx);
|
||||
nmv_context_counts *counts = &nmv_counts[nmv_ctx];
|
||||
#if CONFIG_AMVR
|
||||
av1_inc_mv(&diff, counts, precision);
|
||||
#else
|
||||
av1_inc_mv(&diff, counts, 1);
|
||||
#endif
|
||||
} else if (mode == NEW_NEARESTMV || mode == NEW_NEARMV) {
|
||||
const MV *ref = &mi->bmi[block].ref_mv[0].as_mv;
|
||||
const MV diff = { mvs[0].as_mv.row - ref->row,
|
||||
|
|
@ -324,28 +368,13 @@ static void inc_mvs_sub8x8(const MODE_INFO *mi, int block, const int_mv mvs[2],
|
|||
av1_nmv_ctx(mbmi_ext->ref_mv_count[rf_type],
|
||||
mbmi_ext->ref_mv_stack[rf_type], 0, mbmi->ref_mv_idx);
|
||||
nmv_context_counts *counts = &nmv_counts[nmv_ctx];
|
||||
#if CONFIG_AMVR
|
||||
av1_inc_mv(&diff, counts, precision);
|
||||
#else
|
||||
av1_inc_mv(&diff, counts, 1);
|
||||
#endif
|
||||
}
|
||||
}
|
||||
#else // !CONFIG_EXT_INTER
|
||||
static void inc_mvs(const MB_MODE_INFO *mbmi, const MB_MODE_INFO_EXT *mbmi_ext,
|
||||
const int_mv mvs[2], const int_mv pred_mvs[2],
|
||||
nmv_context_counts *nmv_counts) {
|
||||
int i;
|
||||
|
||||
for (i = 0; i < 1 + has_second_ref(mbmi); ++i) {
|
||||
int8_t rf_type = av1_ref_frame_type(mbmi->ref_frame);
|
||||
int nmv_ctx =
|
||||
av1_nmv_ctx(mbmi_ext->ref_mv_count[rf_type],
|
||||
mbmi_ext->ref_mv_stack[rf_type], i, mbmi->ref_mv_idx);
|
||||
nmv_context_counts *counts = &nmv_counts[nmv_ctx];
|
||||
const MV *ref = &pred_mvs[i].as_mv;
|
||||
const MV diff = { mvs[i].as_mv.row - ref->row,
|
||||
mvs[i].as_mv.col - ref->col };
|
||||
av1_inc_mv(&diff, counts, 1);
|
||||
}
|
||||
}
|
||||
#endif // CONFIG_EXT_INTER
|
||||
|
||||
void av1_update_mv_count(ThreadData *td) {
|
||||
const MACROBLOCKD *xd = &td->mb.e_mbd;
|
||||
|
|
@ -357,6 +386,12 @@ void av1_update_mv_count(ThreadData *td) {
|
|||
#else
|
||||
const int unify_bsize = 0;
|
||||
#endif
|
||||
#if CONFIG_AMVR
|
||||
MvSubpelPrecision precision = 1;
|
||||
if (xd->cur_frame_mv_precision_level) {
|
||||
precision = MV_SUBPEL_NONE;
|
||||
}
|
||||
#endif
|
||||
|
||||
if (mbmi->sb_type < BLOCK_8X8 && !unify_bsize) {
|
||||
const int num_4x4_w = num_4x4_blocks_wide_lookup[mbmi->sb_type];
|
||||
|
|
@ -367,22 +402,24 @@ void av1_update_mv_count(ThreadData *td) {
|
|||
for (idx = 0; idx < 2; idx += num_4x4_w) {
|
||||
const int i = idy * 2 + idx;
|
||||
|
||||
#if CONFIG_EXT_INTER
|
||||
if (have_newmv_in_inter_mode(mi->bmi[i].as_mode))
|
||||
inc_mvs_sub8x8(mi, i, mi->bmi[i].as_mv, mbmi_ext, td->counts->mv);
|
||||
|
||||
#if CONFIG_AMVR
|
||||
inc_mvs_sub8x8(mi, i, mi->bmi[i].as_mv, mbmi_ext, td->counts->mv,
|
||||
precision);
|
||||
#else
|
||||
if (mi->bmi[i].as_mode == NEWMV)
|
||||
inc_mvs(mbmi, mbmi_ext, mi->bmi[i].as_mv, mi->bmi[i].pred_mv,
|
||||
td->counts->mv);
|
||||
#endif // CONFIG_EXT_INTER
|
||||
inc_mvs_sub8x8(mi, i, mi->bmi[i].as_mv, mbmi_ext, td->counts->mv);
|
||||
#endif
|
||||
}
|
||||
}
|
||||
} else {
|
||||
#if CONFIG_EXT_INTER
|
||||
if (have_newmv_in_inter_mode(mbmi->mode))
|
||||
|
||||
#if CONFIG_AMVR
|
||||
inc_mvs(mbmi, mbmi_ext, mbmi->mv, mbmi->pred_mv, td->counts->mv,
|
||||
precision);
|
||||
#else
|
||||
if (mbmi->mode == NEWMV)
|
||||
#endif // CONFIG_EXT_INTER
|
||||
inc_mvs(mbmi, mbmi_ext, mbmi->mv, mbmi->pred_mv, td->counts->mv);
|
||||
#endif
|
||||
}
|
||||
}
|
||||
|
|
|
|||
1705
third_party/aom/av1/encoder/encoder.c
vendored
1705
third_party/aom/av1/encoder/encoder.c
vendored
File diff suppressed because it is too large
Load diff
208
third_party/aom/av1/encoder/encoder.h
vendored
208
third_party/aom/av1/encoder/encoder.h
vendored
|
|
@ -53,23 +53,20 @@
|
|||
extern "C" {
|
||||
#endif
|
||||
|
||||
#if CONFIG_SPEED_REFS
|
||||
#define MIN_SPEED_REFS_BLKSIZE BLOCK_16X16
|
||||
#endif // CONFIG_SPEED_REFS
|
||||
|
||||
typedef struct {
|
||||
int nmv_vec_cost[NMV_CONTEXTS][MV_JOINTS];
|
||||
int nmv_costs[NMV_CONTEXTS][2][MV_VALS];
|
||||
int nmv_costs_hp[NMV_CONTEXTS][2][MV_VALS];
|
||||
|
||||
// 0 = Intra, Last, GF, ARF
|
||||
signed char last_ref_lf_deltas[TOTAL_REFS_PER_FRAME];
|
||||
int8_t last_ref_lf_deltas[TOTAL_REFS_PER_FRAME];
|
||||
// 0 = ZERO_MV, MV
|
||||
signed char last_mode_lf_deltas[MAX_MODE_LF_DELTAS];
|
||||
int8_t last_mode_lf_deltas[MAX_MODE_LF_DELTAS];
|
||||
|
||||
FRAME_CONTEXT fc;
|
||||
} CODING_CONTEXT;
|
||||
|
||||
#if !CONFIG_NO_FRAME_CONTEXT_SIGNALING
|
||||
typedef enum {
|
||||
// regular inter frame
|
||||
REGULAR_FRAME = 0,
|
||||
|
|
@ -86,6 +83,7 @@ typedef enum {
|
|||
EXT_ARF_FRAME = 5
|
||||
#endif
|
||||
} FRAME_CONTEXT_INDEX;
|
||||
#endif
|
||||
|
||||
typedef enum {
|
||||
NORMAL = 0,
|
||||
|
|
@ -105,8 +103,9 @@ typedef enum {
|
|||
FRAMEFLAGS_GOLDEN = 1 << 1,
|
||||
#if CONFIG_EXT_REFS
|
||||
FRAMEFLAGS_BWDREF = 1 << 2,
|
||||
// TODO(zoeliu): To determine whether a frame flag is needed for ALTREF2_FRAME
|
||||
FRAMEFLAGS_ALTREF = 1 << 3,
|
||||
#else
|
||||
#else // !CONFIG_EXT_REFS
|
||||
FRAMEFLAGS_ALTREF = 1 << 2,
|
||||
#endif // CONFIG_EXT_REFS
|
||||
} FRAMETYPE_FLAGS;
|
||||
|
|
@ -116,7 +115,7 @@ typedef enum {
|
|||
VARIANCE_AQ = 1,
|
||||
COMPLEXITY_AQ = 2,
|
||||
CYCLIC_REFRESH_AQ = 3,
|
||||
#if CONFIG_DELTA_Q && !CONFIG_EXT_DELTA_Q
|
||||
#if !CONFIG_EXT_DELTA_Q
|
||||
DELTA_AQ = 4,
|
||||
#endif
|
||||
AQ_MODE_COUNT // This should always be the last member of the enum
|
||||
|
|
@ -131,14 +130,20 @@ typedef enum {
|
|||
#endif
|
||||
typedef enum {
|
||||
RESIZE_NONE = 0, // No frame resizing allowed.
|
||||
RESIZE_FIXED = 1, // All frames are coded at the specified dimension.
|
||||
RESIZE_DYNAMIC = 2 // Coded size of each frame is determined by the codec.
|
||||
RESIZE_FIXED = 1, // All frames are coded at the specified scale.
|
||||
RESIZE_RANDOM = 2, // All frames are coded at a random scale.
|
||||
RESIZE_MODES
|
||||
} RESIZE_MODE;
|
||||
#if CONFIG_FRAME_SUPERRES
|
||||
typedef enum {
|
||||
SUPERRES_NONE = 0,
|
||||
SUPERRES_FIXED = 1,
|
||||
SUPERRES_DYNAMIC = 2
|
||||
SUPERRES_NONE = 0, // No frame superres allowed
|
||||
SUPERRES_FIXED = 1, // All frames are coded at the specified scale,
|
||||
// and super-resolved.
|
||||
SUPERRES_RANDOM = 2, // All frames are coded at a random scale,
|
||||
// and super-resolved.
|
||||
SUPERRES_QTHRESH = 3, // Superres scale for a frame is determined based on
|
||||
// q_index
|
||||
SUPERRES_MODES
|
||||
} SUPERRES_MODE;
|
||||
#endif // CONFIG_FRAME_SUPERRES
|
||||
|
||||
|
|
@ -201,6 +206,9 @@ typedef struct AV1EncoderConfig {
|
|||
int using_qm;
|
||||
int qm_minlevel;
|
||||
int qm_maxlevel;
|
||||
#endif
|
||||
#if CONFIG_DIST_8X8
|
||||
int using_dist_8x8;
|
||||
#endif
|
||||
unsigned int num_tile_groups;
|
||||
unsigned int mtu;
|
||||
|
|
@ -210,14 +218,16 @@ typedef struct AV1EncoderConfig {
|
|||
#endif
|
||||
// Internal frame size scaling.
|
||||
RESIZE_MODE resize_mode;
|
||||
uint8_t resize_scale_numerator;
|
||||
uint8_t resize_kf_scale_numerator;
|
||||
uint8_t resize_scale_denominator;
|
||||
uint8_t resize_kf_scale_denominator;
|
||||
|
||||
#if CONFIG_FRAME_SUPERRES
|
||||
// Frame Super-Resolution size scaling.
|
||||
SUPERRES_MODE superres_mode;
|
||||
uint8_t superres_scale_numerator;
|
||||
uint8_t superres_kf_scale_numerator;
|
||||
uint8_t superres_scale_denominator;
|
||||
uint8_t superres_kf_scale_denominator;
|
||||
int superres_qthresh;
|
||||
int superres_kf_qthresh;
|
||||
#endif // CONFIG_FRAME_SUPERRES
|
||||
|
||||
// Enable feature to reduce the frame quantization every x frames.
|
||||
|
|
@ -255,6 +265,12 @@ typedef struct AV1EncoderConfig {
|
|||
|
||||
int tile_columns;
|
||||
int tile_rows;
|
||||
#if CONFIG_MAX_TILE
|
||||
int tile_width_count;
|
||||
int tile_height_count;
|
||||
int tile_widths[MAX_TILE_COLS];
|
||||
int tile_heights[MAX_TILE_ROWS];
|
||||
#endif
|
||||
#if CONFIG_DEPENDENT_HORZTILES
|
||||
int dependent_horz_tiles;
|
||||
#endif
|
||||
|
|
@ -277,10 +293,8 @@ typedef struct AV1EncoderConfig {
|
|||
int use_highbitdepth;
|
||||
#endif
|
||||
aom_color_space_t color_space;
|
||||
#if CONFIG_COLORSPACE_HEADERS
|
||||
aom_transfer_function_t transfer_function;
|
||||
aom_chroma_sample_position_t chroma_sample_position;
|
||||
#endif
|
||||
int color_range;
|
||||
int render_width;
|
||||
int render_height;
|
||||
|
|
@ -320,7 +334,6 @@ typedef struct TileDataEnc {
|
|||
} TileDataEnc;
|
||||
|
||||
typedef struct RD_COUNTS {
|
||||
av1_coeff_count coef_counts[TX_SIZES][PLANE_TYPES];
|
||||
int64_t comp_pred_diff[REFERENCE_MODES];
|
||||
#if CONFIG_GLOBAL_MOTION
|
||||
// Stores number of 4x4 blocks using global motion per reference frame.
|
||||
|
|
@ -334,8 +347,9 @@ typedef struct ThreadData {
|
|||
MACROBLOCK mb;
|
||||
RD_COUNTS rd_counts;
|
||||
FRAME_COUNTS *counts;
|
||||
|
||||
#if !CONFIG_CB4X4
|
||||
PICK_MODE_CONTEXT *leaf_tree;
|
||||
#endif
|
||||
PC_TREE *pc_tree;
|
||||
PC_TREE *pc_root[MAX_MIB_SIZE_LOG2 - MIN_MIB_SIZE_LOG2 + 1];
|
||||
#if CONFIG_MOTION_VAR
|
||||
|
|
@ -345,9 +359,7 @@ typedef struct ThreadData {
|
|||
uint8_t *left_pred_buf;
|
||||
#endif
|
||||
|
||||
#if CONFIG_PALETTE
|
||||
PALETTE_BUFFER *palette_buffer;
|
||||
#endif // CONFIG_PALETTE
|
||||
} ThreadData;
|
||||
|
||||
struct EncWorkerData;
|
||||
|
|
@ -381,6 +393,9 @@ typedef struct AV1_COMP {
|
|||
QUANTS quants;
|
||||
ThreadData td;
|
||||
MB_MODE_INFO_EXT *mbmi_ext_base;
|
||||
#if CONFIG_LV_MAP
|
||||
CB_COEFF_BUFFER *coeff_buffer_base;
|
||||
#endif
|
||||
Dequants dequants;
|
||||
AV1_COMMON common;
|
||||
AV1EncoderConfig oxcf;
|
||||
|
|
@ -396,6 +411,15 @@ typedef struct AV1_COMP {
|
|||
|
||||
// For a still frame, this flag is set to 1 to skip partition search.
|
||||
int partition_search_skippable_frame;
|
||||
#if CONFIG_AMVR
|
||||
double csm_rate_array[32];
|
||||
double m_rate_array[32];
|
||||
int rate_size;
|
||||
int rate_index;
|
||||
hash_table *previsou_hash_table;
|
||||
int previsous_index;
|
||||
int cur_poc; // DebugInfo
|
||||
#endif
|
||||
|
||||
int scaled_ref_idx[TOTAL_REFS_PER_FRAME];
|
||||
#if CONFIG_EXT_REFS
|
||||
|
|
@ -405,9 +429,14 @@ typedef struct AV1_COMP {
|
|||
#endif // CONFIG_EXT_REFS
|
||||
int gld_fb_idx;
|
||||
#if CONFIG_EXT_REFS
|
||||
int bwd_fb_idx; // BWD_REF_FRAME
|
||||
#endif // CONFIG_EXT_REFS
|
||||
int bwd_fb_idx; // BWDREF_FRAME
|
||||
int alt2_fb_idx; // ALTREF2_FRAME
|
||||
#endif // CONFIG_EXT_REFS
|
||||
int alt_fb_idx;
|
||||
#if CONFIG_EXT_REFS
|
||||
int ext_fb_idx; // extra ref frame buffer index
|
||||
int refresh_fb_idx; // ref frame buffer index to refresh
|
||||
#endif // CONFIG_EXT_REFS
|
||||
|
||||
int last_show_frame_buf_idx; // last show frame buffer index
|
||||
|
||||
|
|
@ -415,6 +444,7 @@ typedef struct AV1_COMP {
|
|||
int refresh_golden_frame;
|
||||
#if CONFIG_EXT_REFS
|
||||
int refresh_bwd_ref_frame;
|
||||
int refresh_alt2_ref_frame;
|
||||
#endif // CONFIG_EXT_REFS
|
||||
int refresh_alt_ref_frame;
|
||||
|
||||
|
|
@ -441,6 +471,11 @@ typedef struct AV1_COMP {
|
|||
|
||||
CODING_CONTEXT coding_context;
|
||||
|
||||
#if CONFIG_GLOBAL_MOTION
|
||||
int gmtype_cost[TRANS_TYPES];
|
||||
int gmparams_cost[TOTAL_REFS_PER_FRAME];
|
||||
#endif // CONFIG_GLOBAL_MOTION
|
||||
|
||||
int nmv_costs[NMV_CONTEXTS][2][MV_VALS];
|
||||
int nmv_costs_hp[NMV_CONTEXTS][2][MV_VALS];
|
||||
|
||||
|
|
@ -534,77 +569,17 @@ typedef struct AV1_COMP {
|
|||
// number of MBs in the current frame when the frame is
|
||||
// scaled.
|
||||
|
||||
// When resize is triggered through external control, the desired width/height
|
||||
// are stored here until use in the next frame coded. They are effective only
|
||||
// for
|
||||
// one frame and are reset after use.
|
||||
int resize_pending_width;
|
||||
int resize_pending_height;
|
||||
|
||||
int frame_flags;
|
||||
|
||||
search_site_config ss_cfg;
|
||||
|
||||
int mbmode_cost[BLOCK_SIZE_GROUPS][INTRA_MODES];
|
||||
int newmv_mode_cost[NEWMV_MODE_CONTEXTS][2];
|
||||
int zeromv_mode_cost[ZEROMV_MODE_CONTEXTS][2];
|
||||
int refmv_mode_cost[REFMV_MODE_CONTEXTS][2];
|
||||
int drl_mode_cost0[DRL_MODE_CONTEXTS][2];
|
||||
|
||||
unsigned int inter_mode_cost[INTER_MODE_CONTEXTS][INTER_MODES];
|
||||
#if CONFIG_EXT_INTER
|
||||
unsigned int inter_compound_mode_cost[INTER_MODE_CONTEXTS]
|
||||
[INTER_COMPOUND_MODES];
|
||||
#if CONFIG_COMPOUND_SINGLEREF
|
||||
unsigned int inter_singleref_comp_mode_cost[INTER_MODE_CONTEXTS]
|
||||
[INTER_SINGLEREF_COMP_MODES];
|
||||
#endif // CONFIG_COMPOUND_SINGLEREF
|
||||
#if CONFIG_INTERINTRA
|
||||
unsigned int interintra_mode_cost[BLOCK_SIZE_GROUPS][INTERINTRA_MODES];
|
||||
#endif // CONFIG_INTERINTRA
|
||||
#endif // CONFIG_EXT_INTER
|
||||
#if CONFIG_MOTION_VAR || CONFIG_WARPED_MOTION
|
||||
int motion_mode_cost[BLOCK_SIZES_ALL][MOTION_MODES];
|
||||
#if CONFIG_MOTION_VAR && CONFIG_WARPED_MOTION
|
||||
int motion_mode_cost1[BLOCK_SIZES_ALL][2];
|
||||
#endif // CONFIG_MOTION_VAR && CONFIG_WARPED_MOTION
|
||||
#if CONFIG_MOTION_VAR && CONFIG_NCOBMC_ADAPT_WEIGHT
|
||||
int ncobmc_mode_cost[ADAPT_OVERLAP_BLOCKS][MAX_NCOBMC_MODES];
|
||||
#endif // CONFIG_MOTION_VAR && CONFIG_NCOBMC_ADAPT_WEIGHT
|
||||
#endif // CONFIG_MOTION_VAR || CONFIG_WARPED_MOTION
|
||||
int intra_uv_mode_cost[INTRA_MODES][UV_INTRA_MODES];
|
||||
int y_mode_costs[INTRA_MODES][INTRA_MODES][INTRA_MODES];
|
||||
int switchable_interp_costs[SWITCHABLE_FILTER_CONTEXTS][SWITCHABLE_FILTERS];
|
||||
#if CONFIG_EXT_PARTITION_TYPES
|
||||
int partition_cost[PARTITION_CONTEXTS + CONFIG_UNPOISON_PARTITION_CTX]
|
||||
[EXT_PARTITION_TYPES];
|
||||
#else
|
||||
int partition_cost[PARTITION_CONTEXTS + CONFIG_UNPOISON_PARTITION_CTX]
|
||||
[PARTITION_TYPES];
|
||||
#endif
|
||||
#if CONFIG_PALETTE
|
||||
int palette_y_size_cost[PALETTE_BLOCK_SIZES][PALETTE_SIZES];
|
||||
int palette_uv_size_cost[PALETTE_BLOCK_SIZES][PALETTE_SIZES];
|
||||
int palette_y_color_cost[PALETTE_SIZES][PALETTE_COLOR_INDEX_CONTEXTS]
|
||||
[PALETTE_COLORS];
|
||||
int palette_uv_color_cost[PALETTE_SIZES][PALETTE_COLOR_INDEX_CONTEXTS]
|
||||
[PALETTE_COLORS];
|
||||
#endif // CONFIG_PALETTE
|
||||
int tx_size_cost[TX_SIZES - 1][TX_SIZE_CONTEXTS][TX_SIZES];
|
||||
#if CONFIG_EXT_TX
|
||||
int inter_tx_type_costs[EXT_TX_SETS_INTER][EXT_TX_SIZES][TX_TYPES];
|
||||
int intra_tx_type_costs[EXT_TX_SETS_INTRA][EXT_TX_SIZES][INTRA_MODES]
|
||||
[TX_TYPES];
|
||||
#else
|
||||
int intra_tx_type_costs[EXT_TX_SIZES][TX_TYPES][TX_TYPES];
|
||||
int inter_tx_type_costs[EXT_TX_SIZES][TX_TYPES];
|
||||
#endif // CONFIG_EXT_TX
|
||||
#if CONFIG_EXT_INTRA
|
||||
#if CONFIG_INTRA_INTERP
|
||||
int intra_filter_cost[INTRA_FILTERS + 1][INTRA_FILTERS];
|
||||
#endif // CONFIG_INTRA_INTERP
|
||||
#endif // CONFIG_EXT_INTRA
|
||||
#if CONFIG_LOOP_RESTORATION
|
||||
int switchable_restore_cost[RESTORE_SWITCHABLE_TYPES];
|
||||
#endif // CONFIG_LOOP_RESTORATION
|
||||
#if CONFIG_GLOBAL_MOTION
|
||||
int gmtype_cost[TRANS_TYPES];
|
||||
int gmparams_cost[TOTAL_REFS_PER_FRAME];
|
||||
#endif // CONFIG_GLOBAL_MOTION
|
||||
|
||||
int multi_arf_allowed;
|
||||
int multi_arf_enabled;
|
||||
int multi_arf_last_grp_enabled;
|
||||
|
|
@ -639,25 +614,24 @@ typedef struct AV1_COMP {
|
|||
int is_arf_filter_off[MAX_EXT_ARFS + 1];
|
||||
int num_extra_arfs;
|
||||
int arf_map[MAX_EXT_ARFS + 1];
|
||||
int arf_pos_in_gf[MAX_EXT_ARFS + 1];
|
||||
int arf_pos_for_ovrly[MAX_EXT_ARFS + 1];
|
||||
#endif // CONFIG_EXT_REFS
|
||||
#if CONFIG_GLOBAL_MOTION
|
||||
int global_motion_search_done;
|
||||
#endif
|
||||
#if CONFIG_REFERENCE_BUFFER
|
||||
SequenceHeader seq_params;
|
||||
#endif
|
||||
#if CONFIG_LV_MAP
|
||||
tran_low_t *tcoeff_buf[MAX_MB_PLANE];
|
||||
#endif
|
||||
|
||||
#if CONFIG_SPEED_REFS
|
||||
int sb_scanning_pass_idx;
|
||||
#endif // CONFIG_SPEED_REFS
|
||||
|
||||
#if CONFIG_FLEX_REFS
|
||||
#if CONFIG_EXT_REFS
|
||||
int extra_arf_allowed;
|
||||
int bwd_ref_allowed;
|
||||
#endif // CONFIG_FLEX_REFS
|
||||
#endif // CONFIG_EXT_REFS
|
||||
|
||||
#if CONFIG_BGSPRITE
|
||||
int bgsprite_allowed;
|
||||
#endif // CONFIG_BGSPRITE
|
||||
} AV1_COMP;
|
||||
|
||||
void av1_initialize_enc(void);
|
||||
|
|
@ -686,11 +660,9 @@ int av1_use_as_reference(AV1_COMP *cpi, int ref_frame_flags);
|
|||
|
||||
void av1_update_reference(AV1_COMP *cpi, int ref_frame_flags);
|
||||
|
||||
int av1_copy_reference_enc(AV1_COMP *cpi, AOM_REFFRAME ref_frame_flag,
|
||||
YV12_BUFFER_CONFIG *sd);
|
||||
int av1_copy_reference_enc(AV1_COMP *cpi, int idx, YV12_BUFFER_CONFIG *sd);
|
||||
|
||||
int av1_set_reference_enc(AV1_COMP *cpi, AOM_REFFRAME ref_frame_flag,
|
||||
YV12_BUFFER_CONFIG *sd);
|
||||
int av1_set_reference_enc(AV1_COMP *cpi, int idx, YV12_BUFFER_CONFIG *sd);
|
||||
|
||||
int av1_update_entropy(AV1_COMP *cpi, int update);
|
||||
|
||||
|
|
@ -701,14 +673,8 @@ int av1_get_active_map(AV1_COMP *cpi, unsigned char *map, int rows, int cols);
|
|||
int av1_set_internal_size(AV1_COMP *cpi, AOM_SCALING horiz_mode,
|
||||
AOM_SCALING vert_mode);
|
||||
|
||||
// Returns 1 if the assigned width or height was <= 0.
|
||||
int av1_set_size_literal(AV1_COMP *cpi, int width, int height);
|
||||
|
||||
int av1_get_quantizer(struct AV1_COMP *cpi);
|
||||
|
||||
void av1_full_to_model_counts(av1_coeff_count_model *model_count,
|
||||
av1_coeff_count *full_count);
|
||||
|
||||
static INLINE int frame_is_kf_gf_arf(const AV1_COMP *cpi) {
|
||||
return frame_is_intra_only(&cpi->common) || cpi->refresh_alt_ref_frame ||
|
||||
(cpi->refresh_golden_frame && !cpi->rc.is_src_frame_alt_ref);
|
||||
|
|
@ -727,6 +693,8 @@ static INLINE int get_ref_frame_map_idx(const AV1_COMP *cpi,
|
|||
#if CONFIG_EXT_REFS
|
||||
else if (ref_frame == BWDREF_FRAME)
|
||||
return cpi->bwd_fb_idx;
|
||||
else if (ref_frame == ALTREF2_FRAME)
|
||||
return cpi->alt2_fb_idx;
|
||||
#endif // CONFIG_EXT_REFS
|
||||
else
|
||||
return cpi->alt_fb_idx;
|
||||
|
|
@ -739,6 +707,17 @@ static INLINE int get_ref_frame_buf_idx(const AV1_COMP *cpi,
|
|||
return (map_idx != INVALID_IDX) ? cm->ref_frame_map[map_idx] : INVALID_IDX;
|
||||
}
|
||||
|
||||
#if CONFIG_HASH_ME
|
||||
static INLINE hash_table *get_ref_frame_hash_map(const AV1_COMP *cpi,
|
||||
MV_REFERENCE_FRAME ref_frame) {
|
||||
const AV1_COMMON *const cm = &cpi->common;
|
||||
const int buf_idx = get_ref_frame_buf_idx(cpi, ref_frame);
|
||||
return buf_idx != INVALID_IDX
|
||||
? &cm->buffer_pool->frame_bufs[buf_idx].hash_table
|
||||
: NULL;
|
||||
}
|
||||
#endif
|
||||
|
||||
static INLINE YV12_BUFFER_CONFIG *get_ref_frame_buffer(
|
||||
const AV1_COMP *cpi, MV_REFERENCE_FRAME ref_frame) {
|
||||
const AV1_COMMON *const cm = &cpi->common;
|
||||
|
|
@ -781,13 +760,6 @@ static INLINE unsigned int allocated_tokens(TileInfo tile) {
|
|||
return get_token_alloc(tile_mb_rows, tile_mb_cols);
|
||||
}
|
||||
|
||||
void av1_alloc_compressor_data(AV1_COMP *cpi);
|
||||
|
||||
void av1_scale_references(AV1_COMP *cpi);
|
||||
|
||||
void av1_update_reference_frames(AV1_COMP *cpi);
|
||||
|
||||
void av1_set_high_precision_mv(AV1_COMP *cpi, int allow_high_precision_mv);
|
||||
#if CONFIG_TEMPMV_SIGNALING
|
||||
void av1_set_temporal_mv_prediction(AV1_COMP *cpi, int allow_tempmv_prediction);
|
||||
#endif
|
||||
|
|
|
|||
1468
third_party/aom/av1/encoder/encodetxb.c
vendored
1468
third_party/aom/av1/encoder/encodetxb.c
vendored
File diff suppressed because it is too large
Load diff
18
third_party/aom/av1/encoder/encodetxb.h
vendored
18
third_party/aom/av1/encoder/encodetxb.h
vendored
|
|
@ -31,6 +31,7 @@ typedef struct TxbInfo {
|
|||
int shift;
|
||||
TX_SIZE tx_size;
|
||||
TX_SIZE txs_ctx;
|
||||
TX_TYPE tx_type;
|
||||
int bwl;
|
||||
int stride;
|
||||
int height;
|
||||
|
|
@ -39,20 +40,21 @@ typedef struct TxbInfo {
|
|||
const SCAN_ORDER *scan_order;
|
||||
TXB_CTX *txb_ctx;
|
||||
int64_t rdmult;
|
||||
const LV_MAP_CTX_TABLE *coeff_ctx_table;
|
||||
} TxbInfo;
|
||||
|
||||
typedef struct TxbCache {
|
||||
int nz_count_arr[MAX_TX_SQUARE];
|
||||
int nz_ctx_arr[MAX_TX_SQUARE][2];
|
||||
int nz_ctx_arr[MAX_TX_SQUARE];
|
||||
int base_count_arr[NUM_BASE_LEVELS][MAX_TX_SQUARE];
|
||||
int base_mag_arr[MAX_TX_SQUARE]
|
||||
[2]; // [0]: max magnitude [1]: num of max magnitude
|
||||
int base_ctx_arr[NUM_BASE_LEVELS][MAX_TX_SQUARE][2]; // [1]: not used
|
||||
int base_ctx_arr[NUM_BASE_LEVELS][MAX_TX_SQUARE];
|
||||
|
||||
int br_count_arr[MAX_TX_SQUARE];
|
||||
int br_mag_arr[MAX_TX_SQUARE]
|
||||
[2]; // [0]: max magnitude [1]: num of max magnitude
|
||||
int br_ctx_arr[MAX_TX_SQUARE][2]; // [1]: not used
|
||||
int br_ctx_arr[MAX_TX_SQUARE];
|
||||
} TxbCache;
|
||||
|
||||
typedef struct TxbProbs {
|
||||
|
|
@ -62,11 +64,14 @@ typedef struct TxbProbs {
|
|||
const aom_prob *coeff_lps;
|
||||
const aom_prob *eob_flag;
|
||||
const aom_prob *txb_skip;
|
||||
#if BR_NODE
|
||||
const aom_prob *coeff_br;
|
||||
#endif
|
||||
} TxbProbs;
|
||||
|
||||
void av1_alloc_txb_buf(AV1_COMP *cpi);
|
||||
void av1_free_txb_buf(AV1_COMP *cpi);
|
||||
int av1_cost_coeffs_txb(const AV1_COMP *const cpi, MACROBLOCK *x, int plane,
|
||||
int av1_cost_coeffs_txb(const AV1_COMMON *const cm, MACROBLOCK *x, int plane,
|
||||
int blk_row, int blk_col, int block, TX_SIZE tx_size,
|
||||
TXB_CTX *txb_ctx);
|
||||
void av1_write_coeffs_txb(const AV1_COMMON *const cm, MACROBLOCKD *xd,
|
||||
|
|
@ -90,6 +95,9 @@ void av1_update_and_record_txb_context(int plane, int block, int blk_row,
|
|||
int blk_col, BLOCK_SIZE plane_bsize,
|
||||
TX_SIZE tx_size, void *arg);
|
||||
|
||||
void av1_set_coeff_buffer(const AV1_COMP *const cpi, MACROBLOCK *const x,
|
||||
int mi_row, int mi_col);
|
||||
|
||||
#if CONFIG_TXK_SEL
|
||||
int64_t av1_search_txk_type(const AV1_COMP *cpi, MACROBLOCK *x, int plane,
|
||||
int block, int blk_row, int blk_col,
|
||||
|
|
@ -99,7 +107,7 @@ int64_t av1_search_txk_type(const AV1_COMP *cpi, MACROBLOCK *x, int plane,
|
|||
#endif
|
||||
int av1_optimize_txb(const AV1_COMMON *cm, MACROBLOCK *x, int plane,
|
||||
int blk_row, int blk_col, int block, TX_SIZE tx_size,
|
||||
TXB_CTX *txb_ctx);
|
||||
TXB_CTX *txb_ctx, int fast_mode);
|
||||
#ifdef __cplusplus
|
||||
}
|
||||
#endif
|
||||
|
|
|
|||
33
third_party/aom/av1/encoder/ethread.c
vendored
33
third_party/aom/av1/encoder/ethread.c
vendored
|
|
@ -15,13 +15,11 @@
|
|||
#include "aom_dsp/aom_dsp_common.h"
|
||||
|
||||
static void accumulate_rd_opt(ThreadData *td, ThreadData *td_t) {
|
||||
int i, j, k, l, m, n;
|
||||
|
||||
for (i = 0; i < REFERENCE_MODES; i++)
|
||||
for (int i = 0; i < REFERENCE_MODES; i++)
|
||||
td->rd_counts.comp_pred_diff[i] += td_t->rd_counts.comp_pred_diff[i];
|
||||
|
||||
#if CONFIG_GLOBAL_MOTION
|
||||
for (i = 0; i < TOTAL_REFS_PER_FRAME; i++)
|
||||
for (int i = 0; i < TOTAL_REFS_PER_FRAME; i++)
|
||||
td->rd_counts.global_motion_used[i] +=
|
||||
td_t->rd_counts.global_motion_used[i];
|
||||
#endif // CONFIG_GLOBAL_MOTION
|
||||
|
|
@ -29,15 +27,6 @@ static void accumulate_rd_opt(ThreadData *td, ThreadData *td_t) {
|
|||
td->rd_counts.compound_ref_used_flag |=
|
||||
td_t->rd_counts.compound_ref_used_flag;
|
||||
td->rd_counts.single_ref_used_flag |= td_t->rd_counts.single_ref_used_flag;
|
||||
|
||||
for (i = 0; i < TX_SIZES; i++)
|
||||
for (j = 0; j < PLANE_TYPES; j++)
|
||||
for (k = 0; k < REF_TYPES; k++)
|
||||
for (l = 0; l < COEF_BANDS; l++)
|
||||
for (m = 0; m < COEFF_CONTEXTS; m++)
|
||||
for (n = 0; n < ENTROPY_TOKENS; n++)
|
||||
td->rd_counts.coef_counts[i][j][k][l][m][n] +=
|
||||
td_t->rd_counts.coef_counts[i][j][k][l][m][n];
|
||||
}
|
||||
|
||||
static int enc_worker_hook(EncWorkerData *const thread_data, void *unused) {
|
||||
|
|
@ -92,8 +81,10 @@ void av1_encode_tiles_mt(AV1_COMP *cpi) {
|
|||
aom_memalign(32, sizeof(*thread_data->td)));
|
||||
av1_zero(*thread_data->td);
|
||||
|
||||
// Set up pc_tree.
|
||||
// Set up pc_tree.
|
||||
#if !CONFIG_CB4X4
|
||||
thread_data->td->leaf_tree = NULL;
|
||||
#endif
|
||||
thread_data->td->pc_tree = NULL;
|
||||
av1_setup_pc_tree(cm, thread_data->td);
|
||||
|
||||
|
|
@ -105,12 +96,14 @@ void av1_encode_tiles_mt(AV1_COMP *cpi) {
|
|||
#endif
|
||||
CHECK_MEM_ERROR(cm, thread_data->td->above_pred_buf,
|
||||
(uint8_t *)aom_memalign(
|
||||
16, buf_scaler * MAX_MB_PLANE * MAX_SB_SQUARE *
|
||||
sizeof(*thread_data->td->above_pred_buf)));
|
||||
16,
|
||||
buf_scaler * MAX_MB_PLANE * MAX_SB_SQUARE *
|
||||
sizeof(*thread_data->td->above_pred_buf)));
|
||||
CHECK_MEM_ERROR(cm, thread_data->td->left_pred_buf,
|
||||
(uint8_t *)aom_memalign(
|
||||
16, buf_scaler * MAX_MB_PLANE * MAX_SB_SQUARE *
|
||||
sizeof(*thread_data->td->left_pred_buf)));
|
||||
16,
|
||||
buf_scaler * MAX_MB_PLANE * MAX_SB_SQUARE *
|
||||
sizeof(*thread_data->td->left_pred_buf)));
|
||||
CHECK_MEM_ERROR(
|
||||
cm, thread_data->td->wsrc_buf,
|
||||
(int32_t *)aom_memalign(
|
||||
|
|
@ -124,12 +117,10 @@ void av1_encode_tiles_mt(AV1_COMP *cpi) {
|
|||
CHECK_MEM_ERROR(cm, thread_data->td->counts,
|
||||
aom_calloc(1, sizeof(*thread_data->td->counts)));
|
||||
|
||||
#if CONFIG_PALETTE
|
||||
// Allocate buffers used by palette coding mode.
|
||||
CHECK_MEM_ERROR(
|
||||
cm, thread_data->td->palette_buffer,
|
||||
aom_memalign(16, sizeof(*thread_data->td->palette_buffer)));
|
||||
#endif // CONFIG_PALETTE
|
||||
|
||||
// Create threads
|
||||
if (!winterface->reset(worker))
|
||||
|
|
@ -169,10 +160,8 @@ void av1_encode_tiles_mt(AV1_COMP *cpi) {
|
|||
sizeof(cpi->common.counts));
|
||||
}
|
||||
|
||||
#if CONFIG_PALETTE
|
||||
if (i < num_workers - 1)
|
||||
thread_data->td->mb.palette_buffer = thread_data->td->palette_buffer;
|
||||
#endif // CONFIG_PALETTE
|
||||
}
|
||||
|
||||
// Encode a frame
|
||||
|
|
|
|||
1359
third_party/aom/av1/encoder/firstpass.c
vendored
1359
third_party/aom/av1/encoder/firstpass.c
vendored
File diff suppressed because it is too large
Load diff
40
third_party/aom/av1/encoder/firstpass.h
vendored
40
third_party/aom/av1/encoder/firstpass.h
vendored
|
|
@ -12,6 +12,8 @@
|
|||
#ifndef AV1_ENCODER_FIRSTPASS_H_
|
||||
#define AV1_ENCODER_FIRSTPASS_H_
|
||||
|
||||
#include "av1/common/enums.h"
|
||||
#include "av1/common/onyxc_int.h"
|
||||
#include "av1/encoder/lookahead.h"
|
||||
#include "av1/encoder/ratectrl.h"
|
||||
|
||||
|
|
@ -45,19 +47,24 @@ typedef struct {
|
|||
// NOTE: Currently each BFG contains one backward ref (BWF) frame plus a certain
|
||||
// number of bi-predictive frames.
|
||||
#define BFG_INTERVAL 2
|
||||
// The maximum number of extra ALT_REF's
|
||||
// NOTE: This number cannot be greater than 2 or the reference frame buffer will
|
||||
// overflow.
|
||||
#define MAX_EXT_ARFS 2
|
||||
#define MIN_EXT_ARF_INTERVAL 4
|
||||
#endif // CONFIG_EXT_REFS
|
||||
// The maximum number of extra ALTREF's except ALTREF_FRAME
|
||||
// NOTE: REF_FRAMES indicates the maximum number of frames that may be buffered
|
||||
// to serve as references. Currently REF_FRAMES == 8.
|
||||
#define USE_GF16_MULTI_LAYER 0
|
||||
|
||||
#if USE_GF16_MULTI_LAYER
|
||||
#define MAX_EXT_ARFS (REF_FRAMES - BWDREF_FRAME)
|
||||
#else // !USE_GF16_MULTI_LAYER
|
||||
#define MAX_EXT_ARFS (REF_FRAMES - BWDREF_FRAME - 1)
|
||||
#endif // USE_GF16_MULTI_LAYER
|
||||
|
||||
#define MIN_EXT_ARF_INTERVAL 4
|
||||
|
||||
#if CONFIG_FLEX_REFS
|
||||
#define MIN_ZERO_MOTION 0.95
|
||||
#define MAX_SR_CODED_ERROR 40
|
||||
#define MAX_RAW_ERR_VAR 2000
|
||||
#define MIN_MV_IN_OUT 0.4
|
||||
#endif // CONFIG_FLEX_REFS
|
||||
#endif // CONFIG_EXT_REFS
|
||||
|
||||
#define VLOW_MOTION_THRESHOLD 950
|
||||
|
||||
|
|
@ -84,10 +91,10 @@ typedef struct {
|
|||
double new_mv_count;
|
||||
double duration;
|
||||
double count;
|
||||
#if CONFIG_FLEX_REFS
|
||||
#if CONFIG_EXT_REFS || CONFIG_BGSPRITE
|
||||
// standard deviation for (0, 0) motion prediction error
|
||||
double raw_error_stdev;
|
||||
#endif // CONFIG_FLEX_REFS
|
||||
#endif // CONFIG_EXT_REFS
|
||||
} FIRSTPASS_STATS;
|
||||
|
||||
typedef enum {
|
||||
|
|
@ -101,8 +108,9 @@ typedef enum {
|
|||
LAST_BIPRED_UPDATE = 6, // Last Bi-predictive Frame
|
||||
BIPRED_UPDATE = 7, // Bi-predictive Frame, but not the last one
|
||||
INTNL_OVERLAY_UPDATE = 8, // Internal Overlay Frame
|
||||
FRAME_UPDATE_TYPES = 9
|
||||
#else
|
||||
INTNL_ARF_UPDATE = 9, // Internal Altref Frame (candidate for ALTREF2)
|
||||
FRAME_UPDATE_TYPES = 10
|
||||
#else // !CONFIG_EXT_REFS
|
||||
FRAME_UPDATE_TYPES = 5
|
||||
#endif // CONFIG_EXT_REFS
|
||||
} FRAME_UPDATE_TYPE;
|
||||
|
|
@ -124,6 +132,9 @@ typedef struct {
|
|||
#if CONFIG_EXT_REFS
|
||||
unsigned char brf_src_offset[(MAX_LAG_BUFFERS * 2) + 1];
|
||||
unsigned char bidir_pred_enabled[(MAX_LAG_BUFFERS * 2) + 1];
|
||||
unsigned char ref_fb_idx_map[(MAX_LAG_BUFFERS * 2) + 1][REF_FRAMES];
|
||||
unsigned char refresh_idx[(MAX_LAG_BUFFERS * 2) + 1];
|
||||
unsigned char refresh_flag[(MAX_LAG_BUFFERS * 2) + 1];
|
||||
#endif // CONFIG_EXT_REFS
|
||||
int bit_allocation[(MAX_LAG_BUFFERS * 2) + 1];
|
||||
} GF_GROUP;
|
||||
|
|
@ -183,12 +194,15 @@ void av1_end_first_pass(struct AV1_COMP *cpi);
|
|||
|
||||
void av1_init_second_pass(struct AV1_COMP *cpi);
|
||||
void av1_rc_get_second_pass_params(struct AV1_COMP *cpi);
|
||||
void av1_twopass_postencode_update(struct AV1_COMP *cpi);
|
||||
|
||||
// Post encode update of the rate control parameters for 2-pass
|
||||
void av1_twopass_postencode_update(struct AV1_COMP *cpi);
|
||||
|
||||
#if CONFIG_EXT_REFS
|
||||
#if USE_GF16_MULTI_LAYER
|
||||
void av1_ref_frame_map_idx_updates(struct AV1_COMP *cpi, int gf_frame_index);
|
||||
#endif // USE_GF16_MULTI_LAYER
|
||||
|
||||
static INLINE int get_number_of_extra_arfs(int interval, int arf_pending) {
|
||||
if (arf_pending && MAX_EXT_ARFS > 0)
|
||||
return interval >= MIN_EXT_ARF_INTERVAL * (MAX_EXT_ARFS + 1)
|
||||
|
|
|
|||
30
third_party/aom/av1/encoder/global_motion.c
vendored
30
third_party/aom/av1/encoder/global_motion.c
vendored
|
|
@ -244,14 +244,18 @@ static unsigned char *downconvert_frame(YV12_BUFFER_CONFIG *frm,
|
|||
int bit_depth) {
|
||||
int i, j;
|
||||
uint16_t *orig_buf = CONVERT_TO_SHORTPTR(frm->y_buffer);
|
||||
uint8_t *buf = malloc(frm->y_height * frm->y_stride * sizeof(*buf));
|
||||
|
||||
for (i = 0; i < frm->y_height; ++i)
|
||||
for (j = 0; j < frm->y_width; ++j)
|
||||
buf[i * frm->y_stride + j] =
|
||||
orig_buf[i * frm->y_stride + j] >> (bit_depth - 8);
|
||||
|
||||
return buf;
|
||||
uint8_t *buf_8bit = frm->y_buffer_8bit;
|
||||
assert(buf_8bit);
|
||||
if (!frm->buf_8bit_valid) {
|
||||
for (i = 0; i < frm->y_height; ++i) {
|
||||
for (j = 0; j < frm->y_width; ++j) {
|
||||
buf_8bit[i * frm->y_stride + j] =
|
||||
orig_buf[i * frm->y_stride + j] >> (bit_depth - 8);
|
||||
}
|
||||
}
|
||||
frm->buf_8bit_valid = 1;
|
||||
}
|
||||
return buf_8bit;
|
||||
}
|
||||
#endif
|
||||
|
||||
|
|
@ -274,16 +278,10 @@ int compute_global_motion_feature_based(
|
|||
if (frm->flags & YV12_FLAG_HIGHBITDEPTH) {
|
||||
// The frame buffer is 16-bit, so we need to convert to 8 bits for the
|
||||
// following code. We cache the result until the frame is released.
|
||||
if (frm->y_buffer_8bit)
|
||||
frm_buffer = frm->y_buffer_8bit;
|
||||
else
|
||||
frm_buffer = frm->y_buffer_8bit = downconvert_frame(frm, bit_depth);
|
||||
frm_buffer = downconvert_frame(frm, bit_depth);
|
||||
}
|
||||
if (ref->flags & YV12_FLAG_HIGHBITDEPTH) {
|
||||
if (ref->y_buffer_8bit)
|
||||
ref_buffer = ref->y_buffer_8bit;
|
||||
else
|
||||
ref_buffer = ref->y_buffer_8bit = downconvert_frame(ref, bit_depth);
|
||||
ref_buffer = downconvert_frame(ref, bit_depth);
|
||||
}
|
||||
#endif
|
||||
|
||||
|
|
|
|||
69
third_party/aom/av1/encoder/hash.c
vendored
Normal file
69
third_party/aom/av1/encoder/hash.c
vendored
Normal file
|
|
@ -0,0 +1,69 @@
|
|||
/*
|
||||
* Copyright (c) 2016, Alliance for Open Media. All rights reserved
|
||||
*
|
||||
* This source code is subject to the terms of the BSD 2 Clause License and
|
||||
* the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
|
||||
* was not distributed with this source code in the LICENSE file, you can
|
||||
* obtain it at www.aomedia.org/license/software. If the Alliance for Open
|
||||
* Media Patent License 1.0 was not distributed with this source code in the
|
||||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#include "av1/encoder/hash.h"
|
||||
|
||||
static void crc_calculator_process_data(CRC_CALCULATOR *p_crc_calculator,
|
||||
uint8_t *pData, uint32_t dataLength) {
|
||||
for (uint32_t i = 0; i < dataLength; i++) {
|
||||
const uint8_t index =
|
||||
(p_crc_calculator->remainder >> (p_crc_calculator->bits - 8)) ^
|
||||
pData[i];
|
||||
p_crc_calculator->remainder <<= 8;
|
||||
p_crc_calculator->remainder ^= p_crc_calculator->table[index];
|
||||
}
|
||||
}
|
||||
|
||||
void crc_calculator_reset(CRC_CALCULATOR *p_crc_calculator) {
|
||||
p_crc_calculator->remainder = 0;
|
||||
}
|
||||
|
||||
static uint32_t crc_calculator_get_crc(CRC_CALCULATOR *p_crc_calculator) {
|
||||
return p_crc_calculator->remainder & p_crc_calculator->final_result_mask;
|
||||
}
|
||||
|
||||
static void crc_calculator_init_table(CRC_CALCULATOR *p_crc_calculator) {
|
||||
const uint32_t high_bit = 1 << (p_crc_calculator->bits - 1);
|
||||
const uint32_t byte_high_bit = 1 << (8 - 1);
|
||||
|
||||
for (uint32_t value = 0; value < 256; value++) {
|
||||
uint32_t remainder = 0;
|
||||
for (uint8_t mask = byte_high_bit; mask != 0; mask >>= 1) {
|
||||
if (value & mask) {
|
||||
remainder ^= high_bit;
|
||||
}
|
||||
|
||||
if (remainder & high_bit) {
|
||||
remainder <<= 1;
|
||||
remainder ^= p_crc_calculator->trunc_poly;
|
||||
} else {
|
||||
remainder <<= 1;
|
||||
}
|
||||
}
|
||||
p_crc_calculator->table[value] = remainder;
|
||||
}
|
||||
}
|
||||
|
||||
void av1_crc_calculator_init(CRC_CALCULATOR *p_crc_calculator, uint32_t bits,
|
||||
uint32_t truncPoly) {
|
||||
p_crc_calculator->remainder = 0;
|
||||
p_crc_calculator->bits = bits;
|
||||
p_crc_calculator->trunc_poly = truncPoly;
|
||||
p_crc_calculator->final_result_mask = (1 << bits) - 1;
|
||||
crc_calculator_init_table(p_crc_calculator);
|
||||
}
|
||||
|
||||
uint32_t av1_get_crc_value(CRC_CALCULATOR *p_crc_calculator, uint8_t *p,
|
||||
int length) {
|
||||
crc_calculator_reset(p_crc_calculator);
|
||||
crc_calculator_process_data(p_crc_calculator, p, length);
|
||||
return crc_calculator_get_crc(p_crc_calculator);
|
||||
}
|
||||
42
third_party/aom/av1/encoder/hash.h
vendored
Normal file
42
third_party/aom/av1/encoder/hash.h
vendored
Normal file
|
|
@ -0,0 +1,42 @@
|
|||
/*
|
||||
* Copyright (c) 2016, Alliance for Open Media. All rights reserved
|
||||
*
|
||||
* This source code is subject to the terms of the BSD 2 Clause License and
|
||||
* the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
|
||||
* was not distributed with this source code in the LICENSE file, you can
|
||||
* obtain it at www.aomedia.org/license/software. If the Alliance for Open
|
||||
* Media Patent License 1.0 was not distributed with this source code in the
|
||||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#ifndef AV1_ENCODER_HASH_H_
|
||||
#define AV1_ENCODER_HASH_H_
|
||||
|
||||
#include "./aom_config.h"
|
||||
#include "aom/aom_integer.h"
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
|
||||
typedef struct _crc_calculator {
|
||||
uint32_t remainder;
|
||||
uint32_t trunc_poly;
|
||||
uint32_t bits;
|
||||
uint32_t table[256];
|
||||
uint32_t final_result_mask;
|
||||
} CRC_CALCULATOR;
|
||||
|
||||
// Initialize the crc calculator. It must be executed at least once before
|
||||
// calling av1_get_crc_value().
|
||||
void av1_crc_calculator_init(CRC_CALCULATOR *p_crc_calculator, uint32_t bits,
|
||||
uint32_t truncPoly);
|
||||
|
||||
uint32_t av1_get_crc_value(CRC_CALCULATOR *p_crc_calculator, uint8_t *p,
|
||||
int length);
|
||||
|
||||
#ifdef __cplusplus
|
||||
} // extern "C"
|
||||
#endif
|
||||
|
||||
#endif // AV1_ENCODER_HASH_H_
|
||||
380
third_party/aom/av1/encoder/hash_motion.c
vendored
Normal file
380
third_party/aom/av1/encoder/hash_motion.c
vendored
Normal file
|
|
@ -0,0 +1,380 @@
|
|||
#include <assert.h>
|
||||
#include "av1/encoder/hash.h"
|
||||
#include "av1/encoder/hash_motion.h"
|
||||
#include "./av1_rtcd.h"
|
||||
|
||||
static const int crc_bits = 16;
|
||||
static const int block_size_bits = 3;
|
||||
static CRC_CALCULATOR crc_calculator1;
|
||||
static CRC_CALCULATOR crc_calculator2;
|
||||
static int g_crc_initialized = 0;
|
||||
|
||||
static void hash_table_clear_all(hash_table *p_hash_table) {
|
||||
if (p_hash_table->p_lookup_table == NULL) {
|
||||
return;
|
||||
}
|
||||
int max_addr = 1 << (crc_bits + block_size_bits);
|
||||
for (int i = 0; i < max_addr; i++) {
|
||||
if (p_hash_table->p_lookup_table[i] != NULL) {
|
||||
vector_destroy(p_hash_table->p_lookup_table[i]);
|
||||
aom_free(p_hash_table->p_lookup_table[i]);
|
||||
p_hash_table->p_lookup_table[i] = NULL;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TODO(youzhou@microsoft.com): is higher than 8 bits screen content supported?
|
||||
// If yes, fix this function
|
||||
static void get_pixels_in_1D_char_array_by_block_2x2(uint8_t *y_src, int stride,
|
||||
uint8_t *p_pixels_in1D) {
|
||||
uint8_t *p_pel = y_src;
|
||||
int index = 0;
|
||||
for (int i = 0; i < 2; i++) {
|
||||
for (int j = 0; j < 2; j++) {
|
||||
p_pixels_in1D[index++] = p_pel[j];
|
||||
}
|
||||
p_pel += stride;
|
||||
}
|
||||
}
|
||||
|
||||
static int is_block_2x2_row_same_value(uint8_t *p) {
|
||||
if (p[0] != p[1] || p[2] != p[3]) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
return 1;
|
||||
}
|
||||
|
||||
static int is_block_2x2_col_same_value(uint8_t *p) {
|
||||
if ((p[0] != p[2]) || (p[1] != p[3])) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
return 1;
|
||||
}
|
||||
|
||||
// the hash value (hash_value1 consists two parts, the first 3 bits relate to
|
||||
// the block size and the remaining 16 bits are the crc values. This fuction
|
||||
// is used to get the first 3 bits.
|
||||
static int hash_block_size_to_index(int block_size) {
|
||||
switch (block_size) {
|
||||
case 4: return 0;
|
||||
case 8: return 1;
|
||||
case 16: return 2;
|
||||
case 32: return 3;
|
||||
case 64: return 4;
|
||||
default: return -1;
|
||||
}
|
||||
}
|
||||
|
||||
void av1_hash_table_init(hash_table *p_hash_table) {
|
||||
if (g_crc_initialized == 0) {
|
||||
av1_crc_calculator_init(&crc_calculator1, 24, 0x5D6DCB);
|
||||
av1_crc_calculator_init(&crc_calculator2, 24, 0x864CFB);
|
||||
g_crc_initialized = 1;
|
||||
}
|
||||
p_hash_table->p_lookup_table = NULL;
|
||||
}
|
||||
|
||||
void av1_hash_table_destroy(hash_table *p_hash_table) {
|
||||
hash_table_clear_all(p_hash_table);
|
||||
aom_free(p_hash_table->p_lookup_table);
|
||||
p_hash_table->p_lookup_table = NULL;
|
||||
}
|
||||
|
||||
void av1_hash_table_create(hash_table *p_hash_table) {
|
||||
if (p_hash_table->p_lookup_table != NULL) {
|
||||
hash_table_clear_all(p_hash_table);
|
||||
return;
|
||||
}
|
||||
const int max_addr = 1 << (crc_bits + block_size_bits);
|
||||
p_hash_table->p_lookup_table =
|
||||
(Vector **)aom_malloc(sizeof(p_hash_table->p_lookup_table[0]) * max_addr);
|
||||
memset(p_hash_table->p_lookup_table, 0,
|
||||
sizeof(p_hash_table->p_lookup_table[0]) * max_addr);
|
||||
}
|
||||
|
||||
static void hash_table_add_to_table(hash_table *p_hash_table,
|
||||
uint32_t hash_value,
|
||||
block_hash *curr_block_hash) {
|
||||
if (p_hash_table->p_lookup_table[hash_value] == NULL) {
|
||||
p_hash_table->p_lookup_table[hash_value] =
|
||||
aom_malloc(sizeof(p_hash_table->p_lookup_table[0][0]));
|
||||
vector_setup(p_hash_table->p_lookup_table[hash_value], 10,
|
||||
sizeof(curr_block_hash[0]));
|
||||
vector_push_back(p_hash_table->p_lookup_table[hash_value], curr_block_hash);
|
||||
} else {
|
||||
vector_push_back(p_hash_table->p_lookup_table[hash_value], curr_block_hash);
|
||||
}
|
||||
}
|
||||
|
||||
int32_t av1_hash_table_count(hash_table *p_hash_table, uint32_t hash_value) {
|
||||
if (p_hash_table->p_lookup_table[hash_value] == NULL) {
|
||||
return 0;
|
||||
} else {
|
||||
return (int32_t)(p_hash_table->p_lookup_table[hash_value]->size);
|
||||
}
|
||||
}
|
||||
|
||||
Iterator av1_hash_get_first_iterator(hash_table *p_hash_table,
|
||||
uint32_t hash_value) {
|
||||
assert(av1_hash_table_count(p_hash_table, hash_value) > 0);
|
||||
return vector_begin(p_hash_table->p_lookup_table[hash_value]);
|
||||
}
|
||||
|
||||
int32_t av1_has_exact_match(hash_table *p_hash_table, uint32_t hash_value1,
|
||||
uint32_t hash_value2) {
|
||||
if (p_hash_table->p_lookup_table[hash_value1] == NULL) {
|
||||
return 0;
|
||||
}
|
||||
Iterator iterator = vector_begin(p_hash_table->p_lookup_table[hash_value1]);
|
||||
Iterator last = vector_end(p_hash_table->p_lookup_table[hash_value1]);
|
||||
for (; !iterator_equals(&iterator, &last); iterator_increment(&iterator)) {
|
||||
if ((*(block_hash *)iterator_get(&iterator)).hash_value2 == hash_value2) {
|
||||
return 1;
|
||||
}
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
void av1_generate_block_2x2_hash_value(const YV12_BUFFER_CONFIG *picture,
|
||||
uint32_t *pic_block_hash[2],
|
||||
int8_t *pic_block_same_info[3]) {
|
||||
const int width = 2;
|
||||
const int height = 2;
|
||||
const int x_end = picture->y_crop_width - width + 1;
|
||||
const int y_end = picture->y_crop_height - height + 1;
|
||||
|
||||
const int length = width * 2;
|
||||
uint8_t p[4];
|
||||
|
||||
int pos = 0;
|
||||
for (int y_pos = 0; y_pos < y_end; y_pos++) {
|
||||
for (int x_pos = 0; x_pos < x_end; x_pos++) {
|
||||
get_pixels_in_1D_char_array_by_block_2x2(
|
||||
picture->y_buffer + y_pos * picture->y_stride + x_pos,
|
||||
picture->y_stride, p);
|
||||
pic_block_same_info[0][pos] = is_block_2x2_row_same_value(p);
|
||||
pic_block_same_info[1][pos] = is_block_2x2_col_same_value(p);
|
||||
|
||||
pic_block_hash[0][pos] =
|
||||
av1_get_crc_value(&crc_calculator1, p, length * sizeof(p[0]));
|
||||
pic_block_hash[1][pos] =
|
||||
av1_get_crc_value(&crc_calculator2, p, length * sizeof(p[0]));
|
||||
|
||||
pos++;
|
||||
}
|
||||
pos += width - 1;
|
||||
}
|
||||
}
|
||||
|
||||
void av1_generate_block_hash_value(const YV12_BUFFER_CONFIG *picture,
|
||||
int block_size,
|
||||
uint32_t *src_pic_block_hash[2],
|
||||
uint32_t *dst_pic_block_hash[2],
|
||||
int8_t *src_pic_block_same_info[3],
|
||||
int8_t *dst_pic_block_same_info[3]) {
|
||||
const int pic_width = picture->y_crop_width;
|
||||
const int x_end = picture->y_crop_width - block_size + 1;
|
||||
const int y_end = picture->y_crop_height - block_size + 1;
|
||||
|
||||
const int src_size = block_size >> 1;
|
||||
const int quad_size = block_size >> 2;
|
||||
|
||||
uint32_t p[4];
|
||||
const int length = sizeof(p);
|
||||
|
||||
int pos = 0;
|
||||
for (int y_pos = 0; y_pos < y_end; y_pos++) {
|
||||
for (int x_pos = 0; x_pos < x_end; x_pos++) {
|
||||
p[0] = src_pic_block_hash[0][pos];
|
||||
p[1] = src_pic_block_hash[0][pos + src_size];
|
||||
p[2] = src_pic_block_hash[0][pos + src_size * pic_width];
|
||||
p[3] = src_pic_block_hash[0][pos + src_size * pic_width + src_size];
|
||||
dst_pic_block_hash[0][pos] =
|
||||
av1_get_crc_value(&crc_calculator1, (uint8_t *)p, length);
|
||||
|
||||
p[0] = src_pic_block_hash[1][pos];
|
||||
p[1] = src_pic_block_hash[1][pos + src_size];
|
||||
p[2] = src_pic_block_hash[1][pos + src_size * pic_width];
|
||||
p[3] = src_pic_block_hash[1][pos + src_size * pic_width + src_size];
|
||||
dst_pic_block_hash[1][pos] =
|
||||
av1_get_crc_value(&crc_calculator2, (uint8_t *)p, length);
|
||||
|
||||
dst_pic_block_same_info[0][pos] =
|
||||
src_pic_block_same_info[0][pos] &&
|
||||
src_pic_block_same_info[0][pos + quad_size] &&
|
||||
src_pic_block_same_info[0][pos + src_size] &&
|
||||
src_pic_block_same_info[0][pos + src_size * pic_width] &&
|
||||
src_pic_block_same_info[0][pos + src_size * pic_width + quad_size] &&
|
||||
src_pic_block_same_info[0][pos + src_size * pic_width + src_size];
|
||||
|
||||
dst_pic_block_same_info[1][pos] =
|
||||
src_pic_block_same_info[1][pos] &&
|
||||
src_pic_block_same_info[1][pos + src_size] &&
|
||||
src_pic_block_same_info[1][pos + quad_size * pic_width] &&
|
||||
src_pic_block_same_info[1][pos + quad_size * pic_width + src_size] &&
|
||||
src_pic_block_same_info[1][pos + src_size * pic_width] &&
|
||||
src_pic_block_same_info[1][pos + src_size * pic_width + src_size];
|
||||
pos++;
|
||||
}
|
||||
pos += block_size - 1;
|
||||
}
|
||||
|
||||
if (block_size >= 4) {
|
||||
const int size_minus1 = block_size - 1;
|
||||
pos = 0;
|
||||
for (int y_pos = 0; y_pos < y_end; y_pos++) {
|
||||
for (int x_pos = 0; x_pos < x_end; x_pos++) {
|
||||
dst_pic_block_same_info[2][pos] =
|
||||
(!dst_pic_block_same_info[0][pos] &&
|
||||
!dst_pic_block_same_info[1][pos]) ||
|
||||
(((x_pos & size_minus1) == 0) && ((y_pos & size_minus1) == 0));
|
||||
pos++;
|
||||
}
|
||||
pos += block_size - 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void av1_add_to_hash_map_by_row_with_precal_data(hash_table *p_hash_table,
|
||||
uint32_t *pic_hash[2],
|
||||
int8_t *pic_is_same,
|
||||
int pic_width, int pic_height,
|
||||
int block_size) {
|
||||
const int x_end = pic_width - block_size + 1;
|
||||
const int y_end = pic_height - block_size + 1;
|
||||
|
||||
const int8_t *src_is_added = pic_is_same;
|
||||
const uint32_t *src_hash[2] = { pic_hash[0], pic_hash[1] };
|
||||
|
||||
int add_value = hash_block_size_to_index(block_size);
|
||||
assert(add_value >= 0);
|
||||
add_value <<= crc_bits;
|
||||
const int crc_mask = (1 << crc_bits) - 1;
|
||||
|
||||
for (int x_pos = 0; x_pos < x_end; x_pos++) {
|
||||
for (int y_pos = 0; y_pos < y_end; y_pos++) {
|
||||
const int pos = y_pos * pic_width + x_pos;
|
||||
// valid data
|
||||
if (src_is_added[pos]) {
|
||||
block_hash curr_block_hash;
|
||||
curr_block_hash.x = x_pos;
|
||||
curr_block_hash.y = y_pos;
|
||||
|
||||
const uint32_t hash_value1 = (src_hash[0][pos] & crc_mask) + add_value;
|
||||
curr_block_hash.hash_value2 = src_hash[1][pos];
|
||||
|
||||
hash_table_add_to_table(p_hash_table, hash_value1, &curr_block_hash);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
int av1_hash_is_horizontal_perfect(const YV12_BUFFER_CONFIG *picture,
|
||||
int block_size, int x_start, int y_start) {
|
||||
const int stride = picture->y_stride;
|
||||
const uint8_t *p = picture->y_buffer + y_start * stride + x_start;
|
||||
|
||||
for (int i = 0; i < block_size; i++) {
|
||||
for (int j = 1; j < block_size; j++) {
|
||||
if (p[j] != p[0]) {
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
p += stride;
|
||||
}
|
||||
|
||||
return 1;
|
||||
}
|
||||
|
||||
int av1_hash_is_vertical_perfect(const YV12_BUFFER_CONFIG *picture,
|
||||
int block_size, int x_start, int y_start) {
|
||||
const int stride = picture->y_stride;
|
||||
const uint8_t *p = picture->y_buffer + y_start * stride + x_start;
|
||||
|
||||
for (int i = 0; i < block_size; i++) {
|
||||
for (int j = 1; j < block_size; j++) {
|
||||
if (p[j * stride + i] != p[i]) {
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return 1;
|
||||
}
|
||||
|
||||
// global buffer for hash value calculation of a block
|
||||
// used only in av1_get_block_hash_value()
|
||||
static uint32_t hash_value_buffer[2][2][1024]; // [first hash/second hash]
|
||||
// [two buffers used ping-pong]
|
||||
// [num of 2x2 blocks in 64x64]
|
||||
|
||||
void av1_get_block_hash_value(uint8_t *y_src, int stride, int block_size,
|
||||
uint32_t *hash_value1, uint32_t *hash_value2) {
|
||||
uint8_t pixel_to_hash[4];
|
||||
uint32_t to_hash[4];
|
||||
const int add_value = hash_block_size_to_index(block_size) << crc_bits;
|
||||
assert(add_value >= 0);
|
||||
const int crc_mask = (1 << crc_bits) - 1;
|
||||
|
||||
// 2x2 subblock hash values in current CU
|
||||
int sub_block_in_width = (block_size >> 1);
|
||||
for (int y_pos = 0; y_pos < block_size; y_pos += 2) {
|
||||
for (int x_pos = 0; x_pos < block_size; x_pos += 2) {
|
||||
int pos = (y_pos >> 1) * sub_block_in_width + (x_pos >> 1);
|
||||
get_pixels_in_1D_char_array_by_block_2x2(y_src + y_pos * stride + x_pos,
|
||||
stride, pixel_to_hash);
|
||||
|
||||
hash_value_buffer[0][0][pos] = av1_get_crc_value(
|
||||
&crc_calculator1, pixel_to_hash, sizeof(pixel_to_hash));
|
||||
hash_value_buffer[1][0][pos] = av1_get_crc_value(
|
||||
&crc_calculator2, pixel_to_hash, sizeof(pixel_to_hash));
|
||||
}
|
||||
}
|
||||
|
||||
int src_sub_block_in_width = sub_block_in_width;
|
||||
sub_block_in_width >>= 1;
|
||||
|
||||
int src_idx = 1;
|
||||
int dst_idx = 0;
|
||||
|
||||
// 4x4 subblock hash values to current block hash values
|
||||
for (int sub_width = 4; sub_width <= block_size; sub_width *= 2) {
|
||||
src_idx = 1 - src_idx;
|
||||
dst_idx = 1 - dst_idx;
|
||||
|
||||
int dst_pos = 0;
|
||||
for (int y_pos = 0; y_pos < sub_block_in_width; y_pos++) {
|
||||
for (int x_pos = 0; x_pos < sub_block_in_width; x_pos++) {
|
||||
int srcPos = (y_pos << 1) * src_sub_block_in_width + (x_pos << 1);
|
||||
|
||||
to_hash[0] = hash_value_buffer[0][src_idx][srcPos];
|
||||
to_hash[1] = hash_value_buffer[0][src_idx][srcPos + 1];
|
||||
to_hash[2] =
|
||||
hash_value_buffer[0][src_idx][srcPos + src_sub_block_in_width];
|
||||
to_hash[3] =
|
||||
hash_value_buffer[0][src_idx][srcPos + src_sub_block_in_width + 1];
|
||||
|
||||
hash_value_buffer[0][dst_idx][dst_pos] = av1_get_crc_value(
|
||||
&crc_calculator1, (uint8_t *)to_hash, sizeof(to_hash));
|
||||
|
||||
to_hash[0] = hash_value_buffer[1][src_idx][srcPos];
|
||||
to_hash[1] = hash_value_buffer[1][src_idx][srcPos + 1];
|
||||
to_hash[2] =
|
||||
hash_value_buffer[1][src_idx][srcPos + src_sub_block_in_width];
|
||||
to_hash[3] =
|
||||
hash_value_buffer[1][src_idx][srcPos + src_sub_block_in_width + 1];
|
||||
hash_value_buffer[1][dst_idx][dst_pos] = av1_get_crc_value(
|
||||
&crc_calculator2, (uint8_t *)to_hash, sizeof(to_hash));
|
||||
dst_pos++;
|
||||
}
|
||||
}
|
||||
|
||||
src_sub_block_in_width = sub_block_in_width;
|
||||
sub_block_in_width >>= 1;
|
||||
}
|
||||
|
||||
*hash_value1 = (hash_value_buffer[0][dst_idx][0] & crc_mask) + add_value;
|
||||
*hash_value2 = hash_value_buffer[1][dst_idx][0];
|
||||
}
|
||||
72
third_party/aom/av1/encoder/hash_motion.h
vendored
Normal file
72
third_party/aom/av1/encoder/hash_motion.h
vendored
Normal file
|
|
@ -0,0 +1,72 @@
|
|||
/*
|
||||
* Copyright (c) 2016, Alliance for Open Media. All rights reserved
|
||||
*
|
||||
* This source code is subject to the terms of the BSD 2 Clause License and
|
||||
* the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
|
||||
* was not distributed with this source code in the LICENSE file, you can
|
||||
* obtain it at www.aomedia.org/license/software. If the Alliance for Open
|
||||
* Media Patent License 1.0 was not distributed with this source code in the
|
||||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#ifndef AV1_ENCODER_HASH_MOTION_H_
|
||||
#define AV1_ENCODER_HASH_MOTION_H_
|
||||
|
||||
#include "./aom_config.h"
|
||||
#include "aom/aom_integer.h"
|
||||
#include "aom_scale/yv12config.h"
|
||||
#include "third_party/vector/vector.h"
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
|
||||
// store a block's hash info.
|
||||
// x and y are the position from the top left of the picture
|
||||
// hash_value2 is used to store the second hash value
|
||||
typedef struct _block_hash {
|
||||
int16_t x;
|
||||
int16_t y;
|
||||
uint32_t hash_value2;
|
||||
} block_hash;
|
||||
|
||||
typedef struct _hash_table { Vector **p_lookup_table; } hash_table;
|
||||
|
||||
void av1_hash_table_init(hash_table *p_hash_table);
|
||||
void av1_hash_table_destroy(hash_table *p_hash_table);
|
||||
void av1_hash_table_create(hash_table *p_hash_table);
|
||||
int32_t av1_hash_table_count(hash_table *p_hash_table, uint32_t hash_value);
|
||||
Iterator av1_hash_get_first_iterator(hash_table *p_hash_table,
|
||||
uint32_t hash_value);
|
||||
int32_t av1_has_exact_match(hash_table *p_hash_table, uint32_t hash_value1,
|
||||
uint32_t hash_value2);
|
||||
void av1_generate_block_2x2_hash_value(const YV12_BUFFER_CONFIG *picture,
|
||||
uint32_t *pic_block_hash[2],
|
||||
int8_t *pic_block_same_info[3]);
|
||||
void av1_generate_block_hash_value(const YV12_BUFFER_CONFIG *picture,
|
||||
int block_size,
|
||||
uint32_t *src_pic_block_hash[2],
|
||||
uint32_t *dst_pic_block_hash[2],
|
||||
int8_t *src_pic_block_same_info[3],
|
||||
int8_t *dst_pic_block_same_info[3]);
|
||||
void av1_add_to_hash_map_by_row_with_precal_data(hash_table *p_hash_table,
|
||||
uint32_t *pic_hash[2],
|
||||
int8_t *pic_is_same,
|
||||
int pic_width, int pic_height,
|
||||
int block_size);
|
||||
|
||||
// check whether the block starts from (x_start, y_start) with the size of
|
||||
// block_size x block_size has the same color in all rows
|
||||
int av1_hash_is_horizontal_perfect(const YV12_BUFFER_CONFIG *picture,
|
||||
int block_size, int x_start, int y_start);
|
||||
// check whether the block starts from (x_start, y_start) with the size of
|
||||
// block_size x block_size has the same color in all columns
|
||||
int av1_hash_is_vertical_perfect(const YV12_BUFFER_CONFIG *picture,
|
||||
int block_size, int x_start, int y_start);
|
||||
void av1_get_block_hash_value(uint8_t *y_src, int stride, int block_size,
|
||||
uint32_t *hash_value1, uint32_t *hash_value2);
|
||||
|
||||
#ifdef __cplusplus
|
||||
} // extern "C"
|
||||
#endif
|
||||
|
||||
#endif // AV1_ENCODER_HASH_MOTION_H_
|
||||
141
third_party/aom/av1/encoder/hybrid_fwd_txfm.c
vendored
141
third_party/aom/av1/encoder/hybrid_fwd_txfm.c
vendored
|
|
@ -51,7 +51,7 @@ static void fwd_txfm_4x4(const int16_t *src_diff, tran_low_t *coeff,
|
|||
return;
|
||||
}
|
||||
|
||||
#if CONFIG_LGT
|
||||
#if CONFIG_LGT || CONFIG_DAALA_DCT4
|
||||
// only C version has LGTs
|
||||
av1_fht4x4_c(src_diff, coeff, diff_stride, txfm_param);
|
||||
#else
|
||||
|
|
@ -107,7 +107,7 @@ static void fwd_txfm_32x16(const int16_t *src_diff, tran_low_t *coeff,
|
|||
|
||||
static void fwd_txfm_8x8(const int16_t *src_diff, tran_low_t *coeff,
|
||||
int diff_stride, TxfmParam *txfm_param) {
|
||||
#if CONFIG_LGT
|
||||
#if CONFIG_LGT || CONFIG_DAALA_DCT8
|
||||
av1_fht8x8_c(src_diff, coeff, diff_stride, txfm_param);
|
||||
#else
|
||||
av1_fht8x8(src_diff, coeff, diff_stride, txfm_param);
|
||||
|
|
@ -116,7 +116,11 @@ static void fwd_txfm_8x8(const int16_t *src_diff, tran_low_t *coeff,
|
|||
|
||||
static void fwd_txfm_16x16(const int16_t *src_diff, tran_low_t *coeff,
|
||||
int diff_stride, TxfmParam *txfm_param) {
|
||||
#if CONFIG_DAALA_DCT16
|
||||
av1_fht16x16_c(src_diff, coeff, diff_stride, txfm_param);
|
||||
#else
|
||||
av1_fht16x16(src_diff, coeff, diff_stride, txfm_param);
|
||||
#endif // CONFIG_DAALA_DCT16
|
||||
}
|
||||
|
||||
static void fwd_txfm_32x32(const int16_t *src_diff, tran_low_t *coeff,
|
||||
|
|
@ -136,11 +140,31 @@ static void fwd_txfm_64x64(const int16_t *src_diff, tran_low_t *coeff,
|
|||
int diff_stride, TxfmParam *txfm_param) {
|
||||
#if CONFIG_EXT_TX
|
||||
if (txfm_param->tx_type == IDTX)
|
||||
av1_fwd_idtx_c(src_diff, coeff, diff_stride, 64, txfm_param->tx_type);
|
||||
av1_fwd_idtx_c(src_diff, coeff, diff_stride, 64, 64, txfm_param->tx_type);
|
||||
else
|
||||
#endif
|
||||
av1_fht64x64(src_diff, coeff, diff_stride, txfm_param);
|
||||
}
|
||||
|
||||
static void fwd_txfm_32x64(const int16_t *src_diff, tran_low_t *coeff,
|
||||
int diff_stride, TxfmParam *txfm_param) {
|
||||
#if CONFIG_EXT_TX
|
||||
if (txfm_param->tx_type == IDTX)
|
||||
av1_fwd_idtx_c(src_diff, coeff, diff_stride, 32, 64, txfm_param->tx_type);
|
||||
else
|
||||
#endif
|
||||
av1_fht32x64(src_diff, coeff, diff_stride, txfm_param);
|
||||
}
|
||||
|
||||
static void fwd_txfm_64x32(const int16_t *src_diff, tran_low_t *coeff,
|
||||
int diff_stride, TxfmParam *txfm_param) {
|
||||
#if CONFIG_EXT_TX
|
||||
if (txfm_param->tx_type == IDTX)
|
||||
av1_fwd_idtx_c(src_diff, coeff, diff_stride, 64, 32, txfm_param->tx_type);
|
||||
else
|
||||
#endif
|
||||
av1_fht64x32(src_diff, coeff, diff_stride, txfm_param);
|
||||
}
|
||||
#endif // CONFIG_TX64X64
|
||||
|
||||
#if CONFIG_RECT_TX_EXT && (CONFIG_EXT_TX || CONFIG_VAR_TX)
|
||||
|
|
@ -211,7 +235,7 @@ static void highbd_fwd_txfm_2x2(const int16_t *src_diff, tran_low_t *coeff,
|
|||
static void highbd_fwd_txfm_4x4(const int16_t *src_diff, tran_low_t *coeff,
|
||||
int diff_stride, TxfmParam *txfm_param) {
|
||||
int32_t *dst_coeff = (int32_t *)coeff;
|
||||
const int tx_type = txfm_param->tx_type;
|
||||
const TX_TYPE tx_type = txfm_param->tx_type;
|
||||
const int bd = txfm_param->bd;
|
||||
if (txfm_param->lossless) {
|
||||
assert(tx_type == DCT_DCT);
|
||||
|
|
@ -296,7 +320,7 @@ static void highbd_fwd_txfm_32x16(const int16_t *src_diff, tran_low_t *coeff,
|
|||
static void highbd_fwd_txfm_8x8(const int16_t *src_diff, tran_low_t *coeff,
|
||||
int diff_stride, TxfmParam *txfm_param) {
|
||||
int32_t *dst_coeff = (int32_t *)coeff;
|
||||
const int tx_type = txfm_param->tx_type;
|
||||
const TX_TYPE tx_type = txfm_param->tx_type;
|
||||
const int bd = txfm_param->bd;
|
||||
switch (tx_type) {
|
||||
case DCT_DCT:
|
||||
|
|
@ -334,7 +358,7 @@ static void highbd_fwd_txfm_8x8(const int16_t *src_diff, tran_low_t *coeff,
|
|||
static void highbd_fwd_txfm_16x16(const int16_t *src_diff, tran_low_t *coeff,
|
||||
int diff_stride, TxfmParam *txfm_param) {
|
||||
int32_t *dst_coeff = (int32_t *)coeff;
|
||||
const int tx_type = txfm_param->tx_type;
|
||||
const TX_TYPE tx_type = txfm_param->tx_type;
|
||||
const int bd = txfm_param->bd;
|
||||
switch (tx_type) {
|
||||
case DCT_DCT:
|
||||
|
|
@ -372,7 +396,7 @@ static void highbd_fwd_txfm_16x16(const int16_t *src_diff, tran_low_t *coeff,
|
|||
static void highbd_fwd_txfm_32x32(const int16_t *src_diff, tran_low_t *coeff,
|
||||
int diff_stride, TxfmParam *txfm_param) {
|
||||
int32_t *dst_coeff = (int32_t *)coeff;
|
||||
const int tx_type = txfm_param->tx_type;
|
||||
const TX_TYPE tx_type = txfm_param->tx_type;
|
||||
const int bd = txfm_param->bd;
|
||||
switch (tx_type) {
|
||||
case DCT_DCT:
|
||||
|
|
@ -408,10 +432,89 @@ static void highbd_fwd_txfm_32x32(const int16_t *src_diff, tran_low_t *coeff,
|
|||
}
|
||||
|
||||
#if CONFIG_TX64X64
|
||||
static void highbd_fwd_txfm_32x64(const int16_t *src_diff, tran_low_t *coeff,
|
||||
int diff_stride, TxfmParam *txfm_param) {
|
||||
int32_t *dst_coeff = (int32_t *)coeff;
|
||||
const TX_TYPE tx_type = txfm_param->tx_type;
|
||||
const int bd = txfm_param->bd;
|
||||
switch (tx_type) {
|
||||
case DCT_DCT:
|
||||
av1_fwd_txfm2d_32x64_c(src_diff, dst_coeff, diff_stride, tx_type, bd);
|
||||
break;
|
||||
#if CONFIG_EXT_TX
|
||||
case ADST_DCT:
|
||||
case DCT_ADST:
|
||||
case ADST_ADST:
|
||||
case FLIPADST_DCT:
|
||||
case DCT_FLIPADST:
|
||||
case FLIPADST_FLIPADST:
|
||||
case ADST_FLIPADST:
|
||||
case FLIPADST_ADST:
|
||||
case V_DCT:
|
||||
case H_DCT:
|
||||
case V_ADST:
|
||||
case H_ADST:
|
||||
case V_FLIPADST:
|
||||
case H_FLIPADST:
|
||||
// TODO(sarahparker)
|
||||
// I've deleted the 64x64 implementations that existed in lieu
|
||||
// of adst, flipadst and identity for simplicity but will bring back
|
||||
// in a later change. This shouldn't impact performance since
|
||||
// DCT_DCT is the only extended type currently allowed for 64x64,
|
||||
// as dictated by get_ext_tx_set_type in blockd.h.
|
||||
av1_fwd_txfm2d_32x64_c(src_diff, dst_coeff, diff_stride, DCT_DCT, bd);
|
||||
break;
|
||||
case IDTX:
|
||||
av1_fwd_idtx_c(src_diff, dst_coeff, diff_stride, 32, 64, tx_type);
|
||||
break;
|
||||
#endif // CONFIG_EXT_TX
|
||||
default: assert(0); break;
|
||||
}
|
||||
}
|
||||
|
||||
static void highbd_fwd_txfm_64x32(const int16_t *src_diff, tran_low_t *coeff,
|
||||
int diff_stride, TxfmParam *txfm_param) {
|
||||
int32_t *dst_coeff = (int32_t *)coeff;
|
||||
const TX_TYPE tx_type = txfm_param->tx_type;
|
||||
const int bd = txfm_param->bd;
|
||||
switch (tx_type) {
|
||||
case DCT_DCT:
|
||||
av1_fwd_txfm2d_64x32_c(src_diff, dst_coeff, diff_stride, tx_type, bd);
|
||||
break;
|
||||
#if CONFIG_EXT_TX
|
||||
case ADST_DCT:
|
||||
case DCT_ADST:
|
||||
case ADST_ADST:
|
||||
case FLIPADST_DCT:
|
||||
case DCT_FLIPADST:
|
||||
case FLIPADST_FLIPADST:
|
||||
case ADST_FLIPADST:
|
||||
case FLIPADST_ADST:
|
||||
case V_DCT:
|
||||
case H_DCT:
|
||||
case V_ADST:
|
||||
case H_ADST:
|
||||
case V_FLIPADST:
|
||||
case H_FLIPADST:
|
||||
// TODO(sarahparker)
|
||||
// I've deleted the 64x64 implementations that existed in lieu
|
||||
// of adst, flipadst and identity for simplicity but will bring back
|
||||
// in a later change. This shouldn't impact performance since
|
||||
// DCT_DCT is the only extended type currently allowed for 64x64,
|
||||
// as dictated by get_ext_tx_set_type in blockd.h.
|
||||
av1_fwd_txfm2d_64x32_c(src_diff, dst_coeff, diff_stride, DCT_DCT, bd);
|
||||
break;
|
||||
case IDTX:
|
||||
av1_fwd_idtx_c(src_diff, dst_coeff, diff_stride, 64, 32, tx_type);
|
||||
break;
|
||||
#endif // CONFIG_EXT_TX
|
||||
default: assert(0); break;
|
||||
}
|
||||
}
|
||||
static void highbd_fwd_txfm_64x64(const int16_t *src_diff, tran_low_t *coeff,
|
||||
int diff_stride, TxfmParam *txfm_param) {
|
||||
int32_t *dst_coeff = (int32_t *)coeff;
|
||||
const int tx_type = txfm_param->tx_type;
|
||||
const TX_TYPE tx_type = txfm_param->tx_type;
|
||||
const int bd = txfm_param->bd;
|
||||
switch (tx_type) {
|
||||
case DCT_DCT:
|
||||
|
|
@ -441,7 +544,7 @@ static void highbd_fwd_txfm_64x64(const int16_t *src_diff, tran_low_t *coeff,
|
|||
av1_fwd_txfm2d_64x64_c(src_diff, dst_coeff, diff_stride, DCT_DCT, bd);
|
||||
break;
|
||||
case IDTX:
|
||||
av1_fwd_idtx_c(src_diff, dst_coeff, diff_stride, 64, tx_type);
|
||||
av1_fwd_idtx_c(src_diff, dst_coeff, diff_stride, 64, 64, tx_type);
|
||||
break;
|
||||
#endif // CONFIG_EXT_TX
|
||||
default: assert(0); break;
|
||||
|
|
@ -452,11 +555,25 @@ static void highbd_fwd_txfm_64x64(const int16_t *src_diff, tran_low_t *coeff,
|
|||
void av1_fwd_txfm(const int16_t *src_diff, tran_low_t *coeff, int diff_stride,
|
||||
TxfmParam *txfm_param) {
|
||||
const TX_SIZE tx_size = txfm_param->tx_size;
|
||||
#if CONFIG_LGT_FROM_PRED
|
||||
if (txfm_param->use_lgt) {
|
||||
// if use_lgt is 1, it will override tx_type
|
||||
assert(is_lgt_allowed(txfm_param->mode, tx_size));
|
||||
flgt2d_from_pred_c(src_diff, coeff, diff_stride, txfm_param);
|
||||
return;
|
||||
}
|
||||
#endif // CONFIG_LGT_FROM_PRED
|
||||
switch (tx_size) {
|
||||
#if CONFIG_TX64X64
|
||||
case TX_64X64:
|
||||
fwd_txfm_64x64(src_diff, coeff, diff_stride, txfm_param);
|
||||
break;
|
||||
case TX_32X64:
|
||||
fwd_txfm_32x64(src_diff, coeff, diff_stride, txfm_param);
|
||||
break;
|
||||
case TX_64X32:
|
||||
fwd_txfm_64x32(src_diff, coeff, diff_stride, txfm_param);
|
||||
break;
|
||||
#endif // CONFIG_TX64X64
|
||||
case TX_32X32:
|
||||
fwd_txfm_32x32(src_diff, coeff, diff_stride, txfm_param);
|
||||
|
|
@ -509,6 +626,12 @@ void av1_highbd_fwd_txfm(const int16_t *src_diff, tran_low_t *coeff,
|
|||
case TX_64X64:
|
||||
highbd_fwd_txfm_64x64(src_diff, coeff, diff_stride, txfm_param);
|
||||
break;
|
||||
case TX_32X64:
|
||||
highbd_fwd_txfm_32x64(src_diff, coeff, diff_stride, txfm_param);
|
||||
break;
|
||||
case TX_64X32:
|
||||
highbd_fwd_txfm_64x32(src_diff, coeff, diff_stride, txfm_param);
|
||||
break;
|
||||
#endif // CONFIG_TX64X64
|
||||
case TX_32X32:
|
||||
highbd_fwd_txfm_32x32(src_diff, coeff, diff_stride, txfm_param);
|
||||
|
|
|
|||
137
third_party/aom/av1/encoder/k_means_template.h
vendored
Normal file
137
third_party/aom/av1/encoder/k_means_template.h
vendored
Normal file
|
|
@ -0,0 +1,137 @@
|
|||
/*
|
||||
* Copyright (c) 2016, Alliance for Open Media. All rights reserved
|
||||
*
|
||||
* This source code is subject to the terms of the BSD 2 Clause License and
|
||||
* the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
|
||||
* was not distributed with this source code in the LICENSE file, you can
|
||||
* obtain it at www.aomedia.org/license/software. If the Alliance for Open
|
||||
* Media Patent License 1.0 was not distributed with this source code in the
|
||||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#include <assert.h>
|
||||
#include <stdint.h>
|
||||
#include <string.h>
|
||||
|
||||
#include "av1/encoder/palette.h"
|
||||
#include "av1/encoder/random.h"
|
||||
|
||||
#ifndef AV1_K_MEANS_DIM
|
||||
#error "This template requires AV1_K_MEANS_DIM to be defined"
|
||||
#endif
|
||||
|
||||
#define RENAME_(x, y) AV1_K_MEANS_RENAME(x, y)
|
||||
#define RENAME(x) RENAME_(x, AV1_K_MEANS_DIM)
|
||||
|
||||
static float RENAME(calc_dist)(const float *p1, const float *p2) {
|
||||
float dist = 0;
|
||||
int i;
|
||||
for (i = 0; i < AV1_K_MEANS_DIM; ++i) {
|
||||
const float diff = p1[i] - p2[i];
|
||||
dist += diff * diff;
|
||||
}
|
||||
return dist;
|
||||
}
|
||||
|
||||
void RENAME(av1_calc_indices)(const float *data, const float *centroids,
|
||||
uint8_t *indices, int n, int k) {
|
||||
int i, j;
|
||||
for (i = 0; i < n; ++i) {
|
||||
float min_dist = RENAME(calc_dist)(data + i * AV1_K_MEANS_DIM, centroids);
|
||||
indices[i] = 0;
|
||||
for (j = 1; j < k; ++j) {
|
||||
const float this_dist = RENAME(calc_dist)(
|
||||
data + i * AV1_K_MEANS_DIM, centroids + j * AV1_K_MEANS_DIM);
|
||||
if (this_dist < min_dist) {
|
||||
min_dist = this_dist;
|
||||
indices[i] = j;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static void RENAME(calc_centroids)(const float *data, float *centroids,
|
||||
const uint8_t *indices, int n, int k) {
|
||||
int i, j, index;
|
||||
int count[PALETTE_MAX_SIZE];
|
||||
unsigned int rand_state = (unsigned int)data[0];
|
||||
|
||||
assert(n <= 32768);
|
||||
|
||||
memset(count, 0, sizeof(count[0]) * k);
|
||||
memset(centroids, 0, sizeof(centroids[0]) * k * AV1_K_MEANS_DIM);
|
||||
|
||||
for (i = 0; i < n; ++i) {
|
||||
index = indices[i];
|
||||
assert(index < k);
|
||||
++count[index];
|
||||
for (j = 0; j < AV1_K_MEANS_DIM; ++j) {
|
||||
centroids[index * AV1_K_MEANS_DIM + j] += data[i * AV1_K_MEANS_DIM + j];
|
||||
}
|
||||
}
|
||||
|
||||
for (i = 0; i < k; ++i) {
|
||||
if (count[i] == 0) {
|
||||
memcpy(centroids + i * AV1_K_MEANS_DIM,
|
||||
data + (lcg_rand16(&rand_state) % n) * AV1_K_MEANS_DIM,
|
||||
sizeof(centroids[0]) * AV1_K_MEANS_DIM);
|
||||
} else {
|
||||
const float norm = 1.0f / count[i];
|
||||
for (j = 0; j < AV1_K_MEANS_DIM; ++j)
|
||||
centroids[i * AV1_K_MEANS_DIM + j] *= norm;
|
||||
}
|
||||
}
|
||||
|
||||
// Round to nearest integers.
|
||||
for (i = 0; i < k * AV1_K_MEANS_DIM; ++i) {
|
||||
centroids[i] = roundf(centroids[i]);
|
||||
}
|
||||
}
|
||||
|
||||
static float RENAME(calc_total_dist)(const float *data, const float *centroids,
|
||||
const uint8_t *indices, int n, int k) {
|
||||
float dist = 0;
|
||||
int i;
|
||||
(void)k;
|
||||
|
||||
for (i = 0; i < n; ++i)
|
||||
dist += RENAME(calc_dist)(data + i * AV1_K_MEANS_DIM,
|
||||
centroids + indices[i] * AV1_K_MEANS_DIM);
|
||||
|
||||
return dist;
|
||||
}
|
||||
|
||||
void RENAME(av1_k_means)(const float *data, float *centroids, uint8_t *indices,
|
||||
int n, int k, int max_itr) {
|
||||
int i;
|
||||
float this_dist;
|
||||
float pre_centroids[2 * PALETTE_MAX_SIZE];
|
||||
uint8_t pre_indices[MAX_SB_SQUARE];
|
||||
|
||||
RENAME(av1_calc_indices)(data, centroids, indices, n, k);
|
||||
this_dist = RENAME(calc_total_dist)(data, centroids, indices, n, k);
|
||||
|
||||
for (i = 0; i < max_itr; ++i) {
|
||||
const float pre_dist = this_dist;
|
||||
memcpy(pre_centroids, centroids,
|
||||
sizeof(pre_centroids[0]) * k * AV1_K_MEANS_DIM);
|
||||
memcpy(pre_indices, indices, sizeof(pre_indices[0]) * n);
|
||||
|
||||
RENAME(calc_centroids)(data, centroids, indices, n, k);
|
||||
RENAME(av1_calc_indices)(data, centroids, indices, n, k);
|
||||
this_dist = RENAME(calc_total_dist)(data, centroids, indices, n, k);
|
||||
|
||||
if (this_dist > pre_dist) {
|
||||
memcpy(centroids, pre_centroids,
|
||||
sizeof(pre_centroids[0]) * k * AV1_K_MEANS_DIM);
|
||||
memcpy(indices, pre_indices, sizeof(pre_indices[0]) * n);
|
||||
break;
|
||||
}
|
||||
if (!memcmp(centroids, pre_centroids,
|
||||
sizeof(pre_centroids[0]) * k * AV1_K_MEANS_DIM))
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
#undef RENAME_
|
||||
#undef RENAME
|
||||
36
third_party/aom/av1/encoder/mbgraph.c
vendored
36
third_party/aom/av1/encoder/mbgraph.c
vendored
|
|
@ -47,32 +47,32 @@ static unsigned int do_16x16_motion_iteration(AV1_COMP *cpi, const MV *ref_mv,
|
|||
av1_hex_search(x, &ref_full, step_param, x->errorperbit, 0,
|
||||
cond_cost_list(cpi, cost_list), &v_fn_ptr, 0, ref_mv);
|
||||
|
||||
// Try sub-pixel MC
|
||||
// if (bestsme > error_thresh && bestsme < INT_MAX)
|
||||
// Try sub-pixel MC
|
||||
// if (bestsme > error_thresh && bestsme < INT_MAX)
|
||||
#if CONFIG_AMVR
|
||||
if (cpi->common.cur_frame_mv_precision_level == 1) {
|
||||
x->best_mv.as_mv.row *= 8;
|
||||
x->best_mv.as_mv.col *= 8;
|
||||
} else {
|
||||
#else
|
||||
{
|
||||
#endif
|
||||
int distortion;
|
||||
unsigned int sse;
|
||||
cpi->find_fractional_mv_step(
|
||||
x, ref_mv, cpi->common.allow_high_precision_mv, x->errorperbit,
|
||||
&v_fn_ptr, 0, mv_sf->subpel_iters_per_step,
|
||||
cond_cost_list(cpi, cost_list), NULL, NULL, &distortion, &sse, NULL,
|
||||
#if CONFIG_EXT_INTER
|
||||
NULL, 0, 0,
|
||||
#endif
|
||||
0, 0, 0);
|
||||
cpi->find_fractional_mv_step(x, ref_mv, cpi->common.allow_high_precision_mv,
|
||||
x->errorperbit, &v_fn_ptr, 0,
|
||||
mv_sf->subpel_iters_per_step,
|
||||
cond_cost_list(cpi, cost_list), NULL, NULL,
|
||||
&distortion, &sse, NULL, NULL, 0, 0, 0, 0, 0);
|
||||
}
|
||||
|
||||
#if CONFIG_EXT_INTER
|
||||
if (has_second_ref(&xd->mi[0]->mbmi))
|
||||
xd->mi[0]->mbmi.mode = NEW_NEWMV;
|
||||
else
|
||||
#endif // CONFIG_EXT_INTER
|
||||
xd->mi[0]->mbmi.mode = NEWMV;
|
||||
|
||||
xd->mi[0]->mbmi.mv[0] = x->best_mv;
|
||||
#if CONFIG_EXT_INTER
|
||||
xd->mi[0]->mbmi.ref_frame[1] = NONE_FRAME;
|
||||
#endif // CONFIG_EXT_INTER
|
||||
|
||||
av1_build_inter_predictors_sby(&cpi->common, xd, mb_row, mb_col, NULL,
|
||||
BLOCK_16X16);
|
||||
|
|
@ -136,6 +136,7 @@ static int do_16x16_zerozero_search(AV1_COMP *cpi, int_mv *dst_mv) {
|
|||
return err;
|
||||
}
|
||||
static int find_best_16x16_intra(AV1_COMP *cpi, PREDICTION_MODE *pbest_mode) {
|
||||
const AV1_COMMON *cm = &cpi->common;
|
||||
MACROBLOCK *const x = &cpi->td.mb;
|
||||
MACROBLOCKD *const xd = &x->e_mbd;
|
||||
PREDICTION_MODE best_mode = -1, mode;
|
||||
|
|
@ -147,9 +148,10 @@ static int find_best_16x16_intra(AV1_COMP *cpi, PREDICTION_MODE *pbest_mode) {
|
|||
unsigned int err;
|
||||
|
||||
xd->mi[0]->mbmi.mode = mode;
|
||||
av1_predict_intra_block(xd, 16, 16, BLOCK_16X16, mode, x->plane[0].src.buf,
|
||||
x->plane[0].src.stride, xd->plane[0].dst.buf,
|
||||
xd->plane[0].dst.stride, 0, 0, 0);
|
||||
av1_predict_intra_block(cm, xd, 16, 16, BLOCK_16X16, mode,
|
||||
x->plane[0].src.buf, x->plane[0].src.stride,
|
||||
xd->plane[0].dst.buf, xd->plane[0].dst.stride, 0, 0,
|
||||
0);
|
||||
err = aom_sad16x16(x->plane[0].src.buf, x->plane[0].src.stride,
|
||||
xd->plane[0].dst.buf, xd->plane[0].dst.stride);
|
||||
|
||||
|
|
|
|||
362
third_party/aom/av1/encoder/mcomp.c
vendored
362
third_party/aom/av1/encoder/mcomp.c
vendored
|
|
@ -176,7 +176,6 @@ static INLINE const uint8_t *pre(const uint8_t *buf, int stride, int r, int c) {
|
|||
}
|
||||
|
||||
/* checks if (r, c) has better score than previous best */
|
||||
#if CONFIG_EXT_INTER
|
||||
#define CHECK_BETTER(v, r, c) \
|
||||
if (c >= minc && c <= maxc && r >= minr && r <= maxr) { \
|
||||
MV this_mv = { r, c }; \
|
||||
|
|
@ -202,34 +201,10 @@ static INLINE const uint8_t *pre(const uint8_t *buf, int stride, int r, int c) {
|
|||
} else { \
|
||||
v = INT_MAX; \
|
||||
}
|
||||
#else
|
||||
#define CHECK_BETTER(v, r, c) \
|
||||
if (c >= minc && c <= maxc && r >= minr && r <= maxr) { \
|
||||
MV this_mv = { r, c }; \
|
||||
v = mv_err_cost(&this_mv, ref_mv, mvjcost, mvcost, error_per_bit); \
|
||||
if (second_pred == NULL) \
|
||||
thismse = vfp->svf(pre(y, y_stride, r, c), y_stride, sp(c), sp(r), \
|
||||
src_address, src_stride, &sse); \
|
||||
else \
|
||||
thismse = vfp->svaf(pre(y, y_stride, r, c), y_stride, sp(c), sp(r), \
|
||||
src_address, src_stride, &sse, second_pred); \
|
||||
v += thismse; \
|
||||
if (v < besterr) { \
|
||||
besterr = v; \
|
||||
br = r; \
|
||||
bc = c; \
|
||||
*distortion = thismse; \
|
||||
*sse1 = sse; \
|
||||
} \
|
||||
} else { \
|
||||
v = INT_MAX; \
|
||||
}
|
||||
#endif // CONFIG_EXT_INTER
|
||||
|
||||
#define CHECK_BETTER0(v, r, c) CHECK_BETTER(v, r, c)
|
||||
|
||||
/* checks if (r, c) has better score than previous best */
|
||||
#if CONFIG_EXT_INTER
|
||||
#define CHECK_BETTER1(v, r, c) \
|
||||
if (c >= minc && c <= maxc && r >= minr && r <= maxr) { \
|
||||
MV this_mv = { r, c }; \
|
||||
|
|
@ -249,26 +224,6 @@ static INLINE const uint8_t *pre(const uint8_t *buf, int stride, int r, int c) {
|
|||
} else { \
|
||||
v = INT_MAX; \
|
||||
}
|
||||
#else
|
||||
#define CHECK_BETTER1(v, r, c) \
|
||||
if (c >= minc && c <= maxc && r >= minr && r <= maxr) { \
|
||||
MV this_mv = { r, c }; \
|
||||
thismse = upsampled_pref_error(xd, vfp, src_address, src_stride, \
|
||||
pre(y, y_stride, r, c), y_stride, sp(c), \
|
||||
sp(r), second_pred, w, h, &sse); \
|
||||
v = mv_err_cost(&this_mv, ref_mv, mvjcost, mvcost, error_per_bit); \
|
||||
v += thismse; \
|
||||
if (v < besterr) { \
|
||||
besterr = v; \
|
||||
br = r; \
|
||||
bc = c; \
|
||||
*distortion = thismse; \
|
||||
*sse1 = sse; \
|
||||
} \
|
||||
} else { \
|
||||
v = INT_MAX; \
|
||||
}
|
||||
#endif // CONFIG_EXT_INTER
|
||||
|
||||
#define FIRST_LEVEL_CHECKS \
|
||||
{ \
|
||||
|
|
@ -372,35 +327,28 @@ static unsigned int setup_center_error(
|
|||
const MACROBLOCKD *xd, const MV *bestmv, const MV *ref_mv,
|
||||
int error_per_bit, const aom_variance_fn_ptr_t *vfp,
|
||||
const uint8_t *const src, const int src_stride, const uint8_t *const y,
|
||||
int y_stride, const uint8_t *second_pred,
|
||||
#if CONFIG_EXT_INTER
|
||||
const uint8_t *mask, int mask_stride, int invert_mask,
|
||||
#endif
|
||||
int w, int h, int offset, int *mvjcost, int *mvcost[2], unsigned int *sse1,
|
||||
int *distortion) {
|
||||
int y_stride, const uint8_t *second_pred, const uint8_t *mask,
|
||||
int mask_stride, int invert_mask, int w, int h, int offset, int *mvjcost,
|
||||
int *mvcost[2], unsigned int *sse1, int *distortion) {
|
||||
unsigned int besterr;
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
if (second_pred != NULL) {
|
||||
if (xd->cur_buf->flags & YV12_FLAG_HIGHBITDEPTH) {
|
||||
DECLARE_ALIGNED(16, uint16_t, comp_pred16[MAX_SB_SQUARE]);
|
||||
#if CONFIG_EXT_INTER
|
||||
if (mask)
|
||||
aom_highbd_comp_mask_pred(comp_pred16, second_pred, w, h, y + offset,
|
||||
y_stride, mask, mask_stride, invert_mask);
|
||||
else
|
||||
#endif
|
||||
aom_highbd_comp_avg_pred(comp_pred16, second_pred, w, h, y + offset,
|
||||
y_stride);
|
||||
besterr =
|
||||
vfp->vf(CONVERT_TO_BYTEPTR(comp_pred16), w, src, src_stride, sse1);
|
||||
} else {
|
||||
DECLARE_ALIGNED(16, uint8_t, comp_pred[MAX_SB_SQUARE]);
|
||||
#if CONFIG_EXT_INTER
|
||||
if (mask)
|
||||
aom_comp_mask_pred(comp_pred, second_pred, w, h, y + offset, y_stride,
|
||||
mask, mask_stride, invert_mask);
|
||||
else
|
||||
#endif
|
||||
aom_comp_avg_pred(comp_pred, second_pred, w, h, y + offset, y_stride);
|
||||
besterr = vfp->vf(comp_pred, w, src, src_stride, sse1);
|
||||
}
|
||||
|
|
@ -413,12 +361,10 @@ static unsigned int setup_center_error(
|
|||
(void)xd;
|
||||
if (second_pred != NULL) {
|
||||
DECLARE_ALIGNED(16, uint8_t, comp_pred[MAX_SB_SQUARE]);
|
||||
#if CONFIG_EXT_INTER
|
||||
if (mask)
|
||||
aom_comp_mask_pred(comp_pred, second_pred, w, h, y + offset, y_stride,
|
||||
mask, mask_stride, invert_mask);
|
||||
else
|
||||
#endif
|
||||
aom_comp_avg_pred(comp_pred, second_pred, w, h, y + offset, y_stride);
|
||||
besterr = vfp->vf(comp_pred, w, src, src_stride, sse1);
|
||||
} else {
|
||||
|
|
@ -458,19 +404,13 @@ int av1_find_best_sub_pixel_tree_pruned_evenmore(
|
|||
MACROBLOCK *x, const MV *ref_mv, int allow_hp, int error_per_bit,
|
||||
const aom_variance_fn_ptr_t *vfp, int forced_stop, int iters_per_step,
|
||||
int *cost_list, int *mvjcost, int *mvcost[2], int *distortion,
|
||||
unsigned int *sse1, const uint8_t *second_pred,
|
||||
#if CONFIG_EXT_INTER
|
||||
const uint8_t *mask, int mask_stride, int invert_mask,
|
||||
#endif
|
||||
int w, int h, int use_upsampled_ref) {
|
||||
unsigned int *sse1, const uint8_t *second_pred, const uint8_t *mask,
|
||||
int mask_stride, int invert_mask, int w, int h, int use_upsampled_ref) {
|
||||
SETUP_SUBPEL_SEARCH;
|
||||
besterr =
|
||||
setup_center_error(xd, bestmv, ref_mv, error_per_bit, vfp, src_address,
|
||||
src_stride, y, y_stride, second_pred,
|
||||
#if CONFIG_EXT_INTER
|
||||
mask, mask_stride, invert_mask,
|
||||
#endif
|
||||
w, h, offset, mvjcost, mvcost, sse1, distortion);
|
||||
besterr = setup_center_error(xd, bestmv, ref_mv, error_per_bit, vfp,
|
||||
src_address, src_stride, y, y_stride,
|
||||
second_pred, mask, mask_stride, invert_mask, w,
|
||||
h, offset, mvjcost, mvcost, sse1, distortion);
|
||||
(void)halfiters;
|
||||
(void)quarteriters;
|
||||
(void)eighthiters;
|
||||
|
|
@ -531,21 +471,15 @@ int av1_find_best_sub_pixel_tree_pruned_more(
|
|||
MACROBLOCK *x, const MV *ref_mv, int allow_hp, int error_per_bit,
|
||||
const aom_variance_fn_ptr_t *vfp, int forced_stop, int iters_per_step,
|
||||
int *cost_list, int *mvjcost, int *mvcost[2], int *distortion,
|
||||
unsigned int *sse1, const uint8_t *second_pred,
|
||||
#if CONFIG_EXT_INTER
|
||||
const uint8_t *mask, int mask_stride, int invert_mask,
|
||||
#endif
|
||||
int w, int h, int use_upsampled_ref) {
|
||||
unsigned int *sse1, const uint8_t *second_pred, const uint8_t *mask,
|
||||
int mask_stride, int invert_mask, int w, int h, int use_upsampled_ref) {
|
||||
SETUP_SUBPEL_SEARCH;
|
||||
(void)use_upsampled_ref;
|
||||
|
||||
besterr =
|
||||
setup_center_error(xd, bestmv, ref_mv, error_per_bit, vfp, src_address,
|
||||
src_stride, y, y_stride, second_pred,
|
||||
#if CONFIG_EXT_INTER
|
||||
mask, mask_stride, invert_mask,
|
||||
#endif
|
||||
w, h, offset, mvjcost, mvcost, sse1, distortion);
|
||||
besterr = setup_center_error(xd, bestmv, ref_mv, error_per_bit, vfp,
|
||||
src_address, src_stride, y, y_stride,
|
||||
second_pred, mask, mask_stride, invert_mask, w,
|
||||
h, offset, mvjcost, mvcost, sse1, distortion);
|
||||
if (cost_list && cost_list[0] != INT_MAX && cost_list[1] != INT_MAX &&
|
||||
cost_list[2] != INT_MAX && cost_list[3] != INT_MAX &&
|
||||
cost_list[4] != INT_MAX && is_cost_list_wellbehaved(cost_list)) {
|
||||
|
|
@ -600,21 +534,15 @@ int av1_find_best_sub_pixel_tree_pruned(
|
|||
MACROBLOCK *x, const MV *ref_mv, int allow_hp, int error_per_bit,
|
||||
const aom_variance_fn_ptr_t *vfp, int forced_stop, int iters_per_step,
|
||||
int *cost_list, int *mvjcost, int *mvcost[2], int *distortion,
|
||||
unsigned int *sse1, const uint8_t *second_pred,
|
||||
#if CONFIG_EXT_INTER
|
||||
const uint8_t *mask, int mask_stride, int invert_mask,
|
||||
#endif
|
||||
int w, int h, int use_upsampled_ref) {
|
||||
unsigned int *sse1, const uint8_t *second_pred, const uint8_t *mask,
|
||||
int mask_stride, int invert_mask, int w, int h, int use_upsampled_ref) {
|
||||
SETUP_SUBPEL_SEARCH;
|
||||
(void)use_upsampled_ref;
|
||||
|
||||
besterr =
|
||||
setup_center_error(xd, bestmv, ref_mv, error_per_bit, vfp, src_address,
|
||||
src_stride, y, y_stride, second_pred,
|
||||
#if CONFIG_EXT_INTER
|
||||
mask, mask_stride, invert_mask,
|
||||
#endif
|
||||
w, h, offset, mvjcost, mvcost, sse1, distortion);
|
||||
besterr = setup_center_error(xd, bestmv, ref_mv, error_per_bit, vfp,
|
||||
src_address, src_stride, y, y_stride,
|
||||
second_pred, mask, mask_stride, invert_mask, w,
|
||||
h, offset, mvjcost, mvcost, sse1, distortion);
|
||||
if (cost_list && cost_list[0] != INT_MAX && cost_list[1] != INT_MAX &&
|
||||
cost_list[2] != INT_MAX && cost_list[3] != INT_MAX &&
|
||||
cost_list[4] != INT_MAX) {
|
||||
|
|
@ -696,26 +624,24 @@ static const MV search_step_table[12] = {
|
|||
};
|
||||
/* clang-format on */
|
||||
|
||||
static int upsampled_pref_error(
|
||||
const MACROBLOCKD *xd, const aom_variance_fn_ptr_t *vfp,
|
||||
const uint8_t *const src, const int src_stride, const uint8_t *const y,
|
||||
int y_stride, int subpel_x_q3, int subpel_y_q3, const uint8_t *second_pred,
|
||||
#if CONFIG_EXT_INTER
|
||||
const uint8_t *mask, int mask_stride, int invert_mask,
|
||||
#endif
|
||||
int w, int h, unsigned int *sse) {
|
||||
static int upsampled_pref_error(const MACROBLOCKD *xd,
|
||||
const aom_variance_fn_ptr_t *vfp,
|
||||
const uint8_t *const src, const int src_stride,
|
||||
const uint8_t *const y, int y_stride,
|
||||
int subpel_x_q3, int subpel_y_q3,
|
||||
const uint8_t *second_pred, const uint8_t *mask,
|
||||
int mask_stride, int invert_mask, int w, int h,
|
||||
unsigned int *sse) {
|
||||
unsigned int besterr;
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
if (xd->cur_buf->flags & YV12_FLAG_HIGHBITDEPTH) {
|
||||
DECLARE_ALIGNED(16, uint16_t, pred16[MAX_SB_SQUARE]);
|
||||
if (second_pred != NULL) {
|
||||
#if CONFIG_EXT_INTER
|
||||
if (mask)
|
||||
aom_highbd_comp_mask_upsampled_pred(
|
||||
pred16, second_pred, w, h, subpel_x_q3, subpel_y_q3, y, y_stride,
|
||||
mask, mask_stride, invert_mask, xd->bd);
|
||||
else
|
||||
#endif
|
||||
aom_highbd_comp_avg_upsampled_pred(pred16, second_pred, w, h,
|
||||
subpel_x_q3, subpel_y_q3, y,
|
||||
y_stride, xd->bd);
|
||||
|
|
@ -732,13 +658,11 @@ static int upsampled_pref_error(
|
|||
(void)xd;
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
if (second_pred != NULL) {
|
||||
#if CONFIG_EXT_INTER
|
||||
if (mask)
|
||||
aom_comp_mask_upsampled_pred(pred, second_pred, w, h, subpel_x_q3,
|
||||
subpel_y_q3, y, y_stride, mask,
|
||||
mask_stride, invert_mask);
|
||||
else
|
||||
#endif
|
||||
aom_comp_avg_upsampled_pred(pred, second_pred, w, h, subpel_x_q3,
|
||||
subpel_y_q3, y, y_stride);
|
||||
} else {
|
||||
|
|
@ -756,18 +680,12 @@ static unsigned int upsampled_setup_center_error(
|
|||
const MACROBLOCKD *xd, const MV *bestmv, const MV *ref_mv,
|
||||
int error_per_bit, const aom_variance_fn_ptr_t *vfp,
|
||||
const uint8_t *const src, const int src_stride, const uint8_t *const y,
|
||||
int y_stride, const uint8_t *second_pred,
|
||||
#if CONFIG_EXT_INTER
|
||||
const uint8_t *mask, int mask_stride, int invert_mask,
|
||||
#endif
|
||||
int w, int h, int offset, int *mvjcost, int *mvcost[2], unsigned int *sse1,
|
||||
int *distortion) {
|
||||
int y_stride, const uint8_t *second_pred, const uint8_t *mask,
|
||||
int mask_stride, int invert_mask, int w, int h, int offset, int *mvjcost,
|
||||
int *mvcost[2], unsigned int *sse1, int *distortion) {
|
||||
unsigned int besterr = upsampled_pref_error(
|
||||
xd, vfp, src, src_stride, y + offset, y_stride, 0, 0, second_pred,
|
||||
#if CONFIG_EXT_INTER
|
||||
mask, mask_stride, invert_mask,
|
||||
#endif
|
||||
w, h, sse1);
|
||||
xd, vfp, src, src_stride, y + offset, y_stride, 0, 0, second_pred, mask,
|
||||
mask_stride, invert_mask, w, h, sse1);
|
||||
*distortion = besterr;
|
||||
besterr += mv_err_cost(bestmv, ref_mv, mvjcost, mvcost, error_per_bit);
|
||||
return besterr;
|
||||
|
|
@ -777,11 +695,8 @@ int av1_find_best_sub_pixel_tree(
|
|||
MACROBLOCK *x, const MV *ref_mv, int allow_hp, int error_per_bit,
|
||||
const aom_variance_fn_ptr_t *vfp, int forced_stop, int iters_per_step,
|
||||
int *cost_list, int *mvjcost, int *mvcost[2], int *distortion,
|
||||
unsigned int *sse1, const uint8_t *second_pred,
|
||||
#if CONFIG_EXT_INTER
|
||||
const uint8_t *mask, int mask_stride, int invert_mask,
|
||||
#endif
|
||||
int w, int h, int use_upsampled_ref) {
|
||||
unsigned int *sse1, const uint8_t *second_pred, const uint8_t *mask,
|
||||
int mask_stride, int invert_mask, int w, int h, int use_upsampled_ref) {
|
||||
const uint8_t *const src_address = x->plane[0].src.buf;
|
||||
const int src_stride = x->plane[0].src.stride;
|
||||
const MACROBLOCKD *xd = &x->e_mbd;
|
||||
|
|
@ -818,19 +733,13 @@ int av1_find_best_sub_pixel_tree(
|
|||
if (use_upsampled_ref)
|
||||
besterr = upsampled_setup_center_error(
|
||||
xd, bestmv, ref_mv, error_per_bit, vfp, src_address, src_stride, y,
|
||||
y_stride, second_pred,
|
||||
#if CONFIG_EXT_INTER
|
||||
mask, mask_stride, invert_mask,
|
||||
#endif
|
||||
w, h, offset, mvjcost, mvcost, sse1, distortion);
|
||||
y_stride, second_pred, mask, mask_stride, invert_mask, w, h, offset,
|
||||
mvjcost, mvcost, sse1, distortion);
|
||||
else
|
||||
besterr =
|
||||
setup_center_error(xd, bestmv, ref_mv, error_per_bit, vfp, src_address,
|
||||
src_stride, y, y_stride, second_pred,
|
||||
#if CONFIG_EXT_INTER
|
||||
mask, mask_stride, invert_mask,
|
||||
#endif
|
||||
w, h, offset, mvjcost, mvcost, sse1, distortion);
|
||||
besterr = setup_center_error(xd, bestmv, ref_mv, error_per_bit, vfp,
|
||||
src_address, src_stride, y, y_stride,
|
||||
second_pred, mask, mask_stride, invert_mask, w,
|
||||
h, offset, mvjcost, mvcost, sse1, distortion);
|
||||
|
||||
(void)cost_list; // to silence compiler warning
|
||||
|
||||
|
|
@ -845,22 +754,17 @@ int av1_find_best_sub_pixel_tree(
|
|||
if (use_upsampled_ref) {
|
||||
thismse = upsampled_pref_error(xd, vfp, src_address, src_stride,
|
||||
pre(y, y_stride, tr, tc), y_stride,
|
||||
sp(tc), sp(tr), second_pred,
|
||||
#if CONFIG_EXT_INTER
|
||||
mask, mask_stride, invert_mask,
|
||||
#endif
|
||||
w, h, &sse);
|
||||
sp(tc), sp(tr), second_pred, mask,
|
||||
mask_stride, invert_mask, w, h, &sse);
|
||||
} else {
|
||||
const uint8_t *const pre_address = pre(y, y_stride, tr, tc);
|
||||
if (second_pred == NULL)
|
||||
thismse = vfp->svf(pre_address, y_stride, sp(tc), sp(tr),
|
||||
src_address, src_stride, &sse);
|
||||
#if CONFIG_EXT_INTER
|
||||
else if (mask)
|
||||
thismse = vfp->msvf(pre_address, y_stride, sp(tc), sp(tr),
|
||||
src_address, src_stride, second_pred, mask,
|
||||
mask_stride, invert_mask, &sse);
|
||||
#endif
|
||||
else
|
||||
thismse = vfp->svaf(pre_address, y_stride, sp(tc), sp(tr),
|
||||
src_address, src_stride, &sse, second_pred);
|
||||
|
|
@ -892,23 +796,18 @@ int av1_find_best_sub_pixel_tree(
|
|||
if (use_upsampled_ref) {
|
||||
thismse = upsampled_pref_error(xd, vfp, src_address, src_stride,
|
||||
pre(y, y_stride, tr, tc), y_stride,
|
||||
sp(tc), sp(tr), second_pred,
|
||||
#if CONFIG_EXT_INTER
|
||||
mask, mask_stride, invert_mask,
|
||||
#endif
|
||||
w, h, &sse);
|
||||
sp(tc), sp(tr), second_pred, mask,
|
||||
mask_stride, invert_mask, w, h, &sse);
|
||||
} else {
|
||||
const uint8_t *const pre_address = pre(y, y_stride, tr, tc);
|
||||
|
||||
if (second_pred == NULL)
|
||||
thismse = vfp->svf(pre_address, y_stride, sp(tc), sp(tr), src_address,
|
||||
src_stride, &sse);
|
||||
#if CONFIG_EXT_INTER
|
||||
else if (mask)
|
||||
thismse = vfp->msvf(pre_address, y_stride, sp(tc), sp(tr),
|
||||
src_address, src_stride, second_pred, mask,
|
||||
mask_stride, invert_mask, &sse);
|
||||
#endif
|
||||
else
|
||||
thismse = vfp->svaf(pre_address, y_stride, sp(tc), sp(tr),
|
||||
src_address, src_stride, &sse, second_pred);
|
||||
|
|
@ -1225,6 +1124,7 @@ static int pattern_search(
|
|||
int thissad;
|
||||
int k = -1;
|
||||
const MV fcenter_mv = { center_mv->row >> 3, center_mv->col >> 3 };
|
||||
assert(search_param < MAX_MVSEARCH_STEPS);
|
||||
int best_init_s = search_param_to_steps[search_param];
|
||||
// adjust ref_mv to make sure it is within MV range
|
||||
clamp_mv(start_mv, x->mv_limits.col_min, x->mv_limits.col_max,
|
||||
|
|
@ -1493,7 +1393,6 @@ int av1_get_mvpred_av_var(const MACROBLOCK *x, const MV *best_mv,
|
|||
: 0);
|
||||
}
|
||||
|
||||
#if CONFIG_EXT_INTER
|
||||
int av1_get_mvpred_mask_var(const MACROBLOCK *x, const MV *best_mv,
|
||||
const MV *center_mv, const uint8_t *second_pred,
|
||||
const uint8_t *mask, int mask_stride,
|
||||
|
|
@ -1512,7 +1411,6 @@ int av1_get_mvpred_mask_var(const MACROBLOCK *x, const MV *best_mv,
|
|||
x->errorperbit)
|
||||
: 0);
|
||||
}
|
||||
#endif
|
||||
|
||||
int av1_hex_search(MACROBLOCK *x, MV *start_mv, int search_param,
|
||||
int sad_per_bit, int do_init_search, int *cost_list,
|
||||
|
|
@ -2481,11 +2379,9 @@ int av1_refining_search_sad(MACROBLOCK *x, MV *ref_mv, int error_per_bit,
|
|||
// mode, or when searching for one component of an ext-inter compound mode.
|
||||
int av1_refining_search_8p_c(MACROBLOCK *x, int error_per_bit, int search_range,
|
||||
const aom_variance_fn_ptr_t *fn_ptr,
|
||||
#if CONFIG_EXT_INTER
|
||||
const uint8_t *mask, int mask_stride,
|
||||
int invert_mask,
|
||||
#endif
|
||||
const MV *center_mv, const uint8_t *second_pred) {
|
||||
int invert_mask, const MV *center_mv,
|
||||
const uint8_t *second_pred) {
|
||||
const MV neighbors[8] = { { -1, 0 }, { 0, -1 }, { 0, 1 }, { 1, 0 },
|
||||
{ -1, -1 }, { 1, -1 }, { -1, 1 }, { 1, 1 } };
|
||||
const MACROBLOCKD *const xd = &x->e_mbd;
|
||||
|
|
@ -2498,14 +2394,12 @@ int av1_refining_search_8p_c(MACROBLOCK *x, int error_per_bit, int search_range,
|
|||
|
||||
clamp_mv(best_mv, x->mv_limits.col_min, x->mv_limits.col_max,
|
||||
x->mv_limits.row_min, x->mv_limits.row_max);
|
||||
#if CONFIG_EXT_INTER
|
||||
if (mask)
|
||||
best_sad = fn_ptr->msdf(what->buf, what->stride,
|
||||
get_buf_from_mv(in_what, best_mv), in_what->stride,
|
||||
second_pred, mask, mask_stride, invert_mask) +
|
||||
mvsad_err_cost(x, best_mv, &fcenter_mv, error_per_bit);
|
||||
else
|
||||
#endif
|
||||
best_sad =
|
||||
fn_ptr->sdaf(what->buf, what->stride, get_buf_from_mv(in_what, best_mv),
|
||||
in_what->stride, second_pred) +
|
||||
|
|
@ -2520,13 +2414,11 @@ int av1_refining_search_8p_c(MACROBLOCK *x, int error_per_bit, int search_range,
|
|||
|
||||
if (is_mv_in(&x->mv_limits, &mv)) {
|
||||
unsigned int sad;
|
||||
#if CONFIG_EXT_INTER
|
||||
if (mask)
|
||||
sad = fn_ptr->msdf(what->buf, what->stride,
|
||||
get_buf_from_mv(in_what, &mv), in_what->stride,
|
||||
second_pred, mask, mask_stride, invert_mask);
|
||||
else
|
||||
#endif
|
||||
sad = fn_ptr->sdaf(what->buf, what->stride,
|
||||
get_buf_from_mv(in_what, &mv), in_what->stride,
|
||||
second_pred);
|
||||
|
|
@ -2562,10 +2454,45 @@ static int is_exhaustive_allowed(const AV1_COMP *const cpi, MACROBLOCK *x) {
|
|||
(*x->ex_search_count_ptr <= max_ex) && !cpi->rc.is_src_frame_alt_ref;
|
||||
}
|
||||
|
||||
#if CONFIG_HASH_ME
|
||||
#define MAX_HASH_MV_TABLE_SIZE 5
|
||||
static void add_to_sort_table(block_hash block_hashes[MAX_HASH_MV_TABLE_SIZE],
|
||||
int costs[MAX_HASH_MV_TABLE_SIZE], int *existing,
|
||||
int max_size, block_hash curr_block,
|
||||
int curr_cost) {
|
||||
if (*existing < max_size) {
|
||||
block_hashes[*existing] = curr_block;
|
||||
costs[*existing] = curr_cost;
|
||||
(*existing)++;
|
||||
} else {
|
||||
int max_cost = 0;
|
||||
int max_cost_idx = 0;
|
||||
for (int i = 0; i < max_size; i++) {
|
||||
if (costs[i] > max_cost) {
|
||||
max_cost = costs[i];
|
||||
max_cost_idx = i;
|
||||
}
|
||||
}
|
||||
|
||||
if (curr_cost < max_cost) {
|
||||
block_hashes[max_cost_idx] = curr_block;
|
||||
costs[max_cost_idx] = curr_cost;
|
||||
}
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
#if CONFIG_HASH_ME
|
||||
int av1_full_pixel_search(const AV1_COMP *cpi, MACROBLOCK *x, BLOCK_SIZE bsize,
|
||||
MV *mvp_full, int step_param, int error_per_bit,
|
||||
int *cost_list, const MV *ref_mv, int var_max, int rd,
|
||||
int x_pos, int y_pos, int intra) {
|
||||
#else
|
||||
int av1_full_pixel_search(const AV1_COMP *cpi, MACROBLOCK *x, BLOCK_SIZE bsize,
|
||||
MV *mvp_full, int step_param, int error_per_bit,
|
||||
int *cost_list, const MV *ref_mv, int var_max,
|
||||
int rd) {
|
||||
#endif
|
||||
const SPEED_FEATURES *const sf = &cpi->sf;
|
||||
const SEARCH_METHODS method = sf->mv.search_method;
|
||||
const aom_variance_fn_ptr_t *fn_ptr = &cpi->fn_ptr[bsize];
|
||||
|
|
@ -2637,6 +2564,93 @@ int av1_full_pixel_search(const AV1_COMP *cpi, MACROBLOCK *x, BLOCK_SIZE bsize,
|
|||
if (method != NSTEP && rd && var < var_max)
|
||||
var = av1_get_mvpred_var(x, &x->best_mv.as_mv, ref_mv, fn_ptr, 1);
|
||||
|
||||
#if CONFIG_HASH_ME
|
||||
do {
|
||||
if (!cpi->common.allow_screen_content_tools) {
|
||||
break;
|
||||
}
|
||||
// already single ME
|
||||
// get block size and original buffer of current block
|
||||
const int block_height = block_size_high[bsize];
|
||||
const int block_width = block_size_wide[bsize];
|
||||
if (block_height == block_width && x_pos >= 0 && y_pos >= 0) {
|
||||
if (block_width == 4 || block_width == 8 || block_width == 16 ||
|
||||
block_width == 32 || block_width == 64) {
|
||||
uint8_t *what = x->plane[0].src.buf;
|
||||
const int what_stride = x->plane[0].src.stride;
|
||||
block_hash block_hashes[MAX_HASH_MV_TABLE_SIZE];
|
||||
int costs[MAX_HASH_MV_TABLE_SIZE];
|
||||
int existing = 0;
|
||||
int i;
|
||||
uint32_t hash_value1, hash_value2;
|
||||
MV best_hash_mv;
|
||||
int best_hash_cost = INT_MAX;
|
||||
|
||||
// for the hashMap
|
||||
hash_table *ref_frame_hash =
|
||||
intra ? &cpi->common.cur_frame->hash_table
|
||||
: get_ref_frame_hash_map(cpi,
|
||||
x->e_mbd.mi[0]->mbmi.ref_frame[0]);
|
||||
|
||||
av1_get_block_hash_value(what, what_stride, block_width, &hash_value1,
|
||||
&hash_value2);
|
||||
|
||||
const int count = av1_hash_table_count(ref_frame_hash, hash_value1);
|
||||
// for intra, at lest one matching can be found, itself.
|
||||
if (count <= (intra ? 1 : 0)) {
|
||||
break;
|
||||
}
|
||||
|
||||
Iterator iterator =
|
||||
av1_hash_get_first_iterator(ref_frame_hash, hash_value1);
|
||||
for (i = 0; i < count; i++, iterator_increment(&iterator)) {
|
||||
block_hash ref_block_hash = *(block_hash *)(iterator_get(&iterator));
|
||||
if (hash_value2 == ref_block_hash.hash_value2) {
|
||||
// for intra, make sure the prediction is from valid area
|
||||
// not predict from current block.
|
||||
// TODO(roger): check if the constrain is necessary
|
||||
if (intra &&
|
||||
ref_block_hash.y + block_height >
|
||||
((y_pos >> MAX_SB_SIZE_LOG2) << MAX_SB_SIZE_LOG2) &&
|
||||
ref_block_hash.x + block_width >
|
||||
((x_pos >> MAX_SB_SIZE_LOG2) << MAX_SB_SIZE_LOG2)) {
|
||||
continue;
|
||||
}
|
||||
int refCost =
|
||||
abs(ref_block_hash.x - x_pos) + abs(ref_block_hash.y - y_pos);
|
||||
add_to_sort_table(block_hashes, costs, &existing,
|
||||
MAX_HASH_MV_TABLE_SIZE, ref_block_hash, refCost);
|
||||
}
|
||||
}
|
||||
|
||||
if (existing == 0) {
|
||||
break;
|
||||
}
|
||||
|
||||
for (i = 0; i < existing; i++) {
|
||||
MV hash_mv;
|
||||
hash_mv.col = block_hashes[i].x - x_pos;
|
||||
hash_mv.row = block_hashes[i].y - y_pos;
|
||||
if (!is_mv_in(&x->mv_limits, &hash_mv)) {
|
||||
continue;
|
||||
}
|
||||
int currHashCost = av1_get_mvpred_var(x, &hash_mv, ref_mv, fn_ptr, 1);
|
||||
if (currHashCost < best_hash_cost) {
|
||||
best_hash_cost = currHashCost;
|
||||
best_hash_mv = hash_mv;
|
||||
}
|
||||
}
|
||||
|
||||
if (best_hash_cost < var) {
|
||||
x->second_best_mv = x->best_mv;
|
||||
x->best_mv.as_mv = best_hash_mv;
|
||||
var = best_hash_cost;
|
||||
}
|
||||
}
|
||||
}
|
||||
} while (0);
|
||||
#endif
|
||||
|
||||
return var;
|
||||
}
|
||||
|
||||
|
|
@ -3150,25 +3164,24 @@ int av1_return_max_sub_pixel_mv(
|
|||
MACROBLOCK *x, const MV *ref_mv, int allow_hp, int error_per_bit,
|
||||
const aom_variance_fn_ptr_t *vfp, int forced_stop, int iters_per_step,
|
||||
int *cost_list, int *mvjcost, int *mvcost[2], int *distortion,
|
||||
unsigned int *sse1, const uint8_t *second_pred,
|
||||
#if CONFIG_EXT_INTER
|
||||
const uint8_t *mask, int mask_stride, int invert_mask,
|
||||
#endif
|
||||
int w, int h, int use_upsampled_ref) {
|
||||
unsigned int *sse1, const uint8_t *second_pred, const uint8_t *mask,
|
||||
int mask_stride, int invert_mask, int w, int h, int use_upsampled_ref) {
|
||||
COMMON_MV_TEST;
|
||||
#if CONFIG_EXT_INTER
|
||||
(void)mask;
|
||||
(void)mask_stride;
|
||||
(void)invert_mask;
|
||||
#endif
|
||||
(void)minr;
|
||||
(void)minc;
|
||||
bestmv->row = maxr;
|
||||
bestmv->col = maxc;
|
||||
besterr = 0;
|
||||
// In the sub-pel motion search, if hp is not used, then the last bit of mv
|
||||
// has to be 0.
|
||||
// In the sub-pel motion search, if hp is not used, then the last bit of mv
|
||||
// has to be 0.
|
||||
#if CONFIG_AMVR
|
||||
lower_mv_precision(bestmv, allow_hp, 0);
|
||||
#else
|
||||
lower_mv_precision(bestmv, allow_hp);
|
||||
#endif
|
||||
return besterr;
|
||||
}
|
||||
// Return the minimum MV.
|
||||
|
|
@ -3176,24 +3189,23 @@ int av1_return_min_sub_pixel_mv(
|
|||
MACROBLOCK *x, const MV *ref_mv, int allow_hp, int error_per_bit,
|
||||
const aom_variance_fn_ptr_t *vfp, int forced_stop, int iters_per_step,
|
||||
int *cost_list, int *mvjcost, int *mvcost[2], int *distortion,
|
||||
unsigned int *sse1, const uint8_t *second_pred,
|
||||
#if CONFIG_EXT_INTER
|
||||
const uint8_t *mask, int mask_stride, int invert_mask,
|
||||
#endif
|
||||
int w, int h, int use_upsampled_ref) {
|
||||
unsigned int *sse1, const uint8_t *second_pred, const uint8_t *mask,
|
||||
int mask_stride, int invert_mask, int w, int h, int use_upsampled_ref) {
|
||||
COMMON_MV_TEST;
|
||||
(void)maxr;
|
||||
(void)maxc;
|
||||
#if CONFIG_EXT_INTER
|
||||
(void)mask;
|
||||
(void)mask_stride;
|
||||
(void)invert_mask;
|
||||
#endif
|
||||
bestmv->row = minr;
|
||||
bestmv->col = minc;
|
||||
besterr = 0;
|
||||
// In the sub-pel motion search, if hp is not used, then the last bit of mv
|
||||
// has to be 0.
|
||||
// In the sub-pel motion search, if hp is not used, then the last bit of mv
|
||||
// has to be 0.
|
||||
#if CONFIG_AMVR
|
||||
lower_mv_precision(bestmv, allow_hp, 0);
|
||||
#else
|
||||
lower_mv_precision(bestmv, allow_hp);
|
||||
#endif
|
||||
return besterr;
|
||||
}
|
||||
|
|
|
|||
21
third_party/aom/av1/encoder/mcomp.h
vendored
21
third_party/aom/av1/encoder/mcomp.h
vendored
|
|
@ -58,13 +58,11 @@ int av1_get_mvpred_var(const MACROBLOCK *x, const MV *best_mv,
|
|||
int av1_get_mvpred_av_var(const MACROBLOCK *x, const MV *best_mv,
|
||||
const MV *center_mv, const uint8_t *second_pred,
|
||||
const aom_variance_fn_ptr_t *vfp, int use_mvcost);
|
||||
#if CONFIG_EXT_INTER
|
||||
int av1_get_mvpred_mask_var(const MACROBLOCK *x, const MV *best_mv,
|
||||
const MV *center_mv, const uint8_t *second_pred,
|
||||
const uint8_t *mask, int mask_stride,
|
||||
int invert_mask, const aom_variance_fn_ptr_t *vfp,
|
||||
int use_mvcost);
|
||||
#endif
|
||||
|
||||
struct AV1_COMP;
|
||||
struct SPEED_FEATURES;
|
||||
|
|
@ -99,10 +97,8 @@ typedef int(fractional_mv_step_fp)(
|
|||
int forced_stop, // 0 - full, 1 - qtr only, 2 - half only
|
||||
int iters_per_step, int *cost_list, int *mvjcost, int *mvcost[2],
|
||||
int *distortion, unsigned int *sse1, const uint8_t *second_pred,
|
||||
#if CONFIG_EXT_INTER
|
||||
const uint8_t *mask, int mask_stride, int invert_mask,
|
||||
#endif
|
||||
int w, int h, int use_upsampled_ref);
|
||||
const uint8_t *mask, int mask_stride, int invert_mask, int w, int h,
|
||||
int use_upsampled_ref);
|
||||
|
||||
extern fractional_mv_step_fp av1_find_best_sub_pixel_tree;
|
||||
extern fractional_mv_step_fp av1_find_best_sub_pixel_tree_pruned;
|
||||
|
|
@ -123,18 +119,23 @@ typedef int (*av1_diamond_search_fn_t)(
|
|||
|
||||
int av1_refining_search_8p_c(MACROBLOCK *x, int error_per_bit, int search_range,
|
||||
const aom_variance_fn_ptr_t *fn_ptr,
|
||||
#if CONFIG_EXT_INTER
|
||||
const uint8_t *mask, int mask_stride,
|
||||
int invert_mask,
|
||||
#endif
|
||||
const MV *center_mv, const uint8_t *second_pred);
|
||||
int invert_mask, const MV *center_mv,
|
||||
const uint8_t *second_pred);
|
||||
|
||||
struct AV1_COMP;
|
||||
|
||||
#if CONFIG_HASH_ME
|
||||
int av1_full_pixel_search(const struct AV1_COMP *cpi, MACROBLOCK *x,
|
||||
BLOCK_SIZE bsize, MV *mvp_full, int step_param,
|
||||
int error_per_bit, int *cost_list, const MV *ref_mv,
|
||||
int var_max, int rd, int x_pos, int y_pos, int intra);
|
||||
#else
|
||||
int av1_full_pixel_search(const struct AV1_COMP *cpi, MACROBLOCK *x,
|
||||
BLOCK_SIZE bsize, MV *mvp_full, int step_param,
|
||||
int error_per_bit, int *cost_list, const MV *ref_mv,
|
||||
int var_max, int rd);
|
||||
#endif
|
||||
|
||||
#if CONFIG_MOTION_VAR
|
||||
int av1_obmc_full_pixel_diamond(const struct AV1_COMP *cpi, MACROBLOCK *x,
|
||||
|
|
|
|||
116
third_party/aom/av1/encoder/palette.c
vendored
116
third_party/aom/av1/encoder/palette.c
vendored
|
|
@ -14,116 +14,14 @@
|
|||
|
||||
#include "av1/encoder/cost.h"
|
||||
#include "av1/encoder/palette.h"
|
||||
#include "av1/encoder/random.h"
|
||||
|
||||
static float calc_dist(const float *p1, const float *p2, int dim) {
|
||||
float dist = 0;
|
||||
int i;
|
||||
for (i = 0; i < dim; ++i) {
|
||||
const float diff = p1[i] - p2[i];
|
||||
dist += diff * diff;
|
||||
}
|
||||
return dist;
|
||||
}
|
||||
|
||||
void av1_calc_indices(const float *data, const float *centroids,
|
||||
uint8_t *indices, int n, int k, int dim) {
|
||||
int i, j;
|
||||
for (i = 0; i < n; ++i) {
|
||||
float min_dist = calc_dist(data + i * dim, centroids, dim);
|
||||
indices[i] = 0;
|
||||
for (j = 1; j < k; ++j) {
|
||||
const float this_dist =
|
||||
calc_dist(data + i * dim, centroids + j * dim, dim);
|
||||
if (this_dist < min_dist) {
|
||||
min_dist = this_dist;
|
||||
indices[i] = j;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Generate a random number in the range [0, 32768).
|
||||
static unsigned int lcg_rand16(unsigned int *state) {
|
||||
*state = (unsigned int)(*state * 1103515245ULL + 12345);
|
||||
return *state / 65536 % 32768;
|
||||
}
|
||||
|
||||
static void calc_centroids(const float *data, float *centroids,
|
||||
const uint8_t *indices, int n, int k, int dim) {
|
||||
int i, j, index;
|
||||
int count[PALETTE_MAX_SIZE];
|
||||
unsigned int rand_state = (unsigned int)data[0];
|
||||
|
||||
assert(n <= 32768);
|
||||
|
||||
memset(count, 0, sizeof(count[0]) * k);
|
||||
memset(centroids, 0, sizeof(centroids[0]) * k * dim);
|
||||
|
||||
for (i = 0; i < n; ++i) {
|
||||
index = indices[i];
|
||||
assert(index < k);
|
||||
++count[index];
|
||||
for (j = 0; j < dim; ++j) {
|
||||
centroids[index * dim + j] += data[i * dim + j];
|
||||
}
|
||||
}
|
||||
|
||||
for (i = 0; i < k; ++i) {
|
||||
if (count[i] == 0) {
|
||||
memcpy(centroids + i * dim, data + (lcg_rand16(&rand_state) % n) * dim,
|
||||
sizeof(centroids[0]) * dim);
|
||||
} else {
|
||||
const float norm = 1.0f / count[i];
|
||||
for (j = 0; j < dim; ++j) centroids[i * dim + j] *= norm;
|
||||
}
|
||||
}
|
||||
|
||||
// Round to nearest integers.
|
||||
for (i = 0; i < k * dim; ++i) {
|
||||
centroids[i] = roundf(centroids[i]);
|
||||
}
|
||||
}
|
||||
|
||||
static float calc_total_dist(const float *data, const float *centroids,
|
||||
const uint8_t *indices, int n, int k, int dim) {
|
||||
float dist = 0;
|
||||
int i;
|
||||
(void)k;
|
||||
|
||||
for (i = 0; i < n; ++i)
|
||||
dist += calc_dist(data + i * dim, centroids + indices[i] * dim, dim);
|
||||
|
||||
return dist;
|
||||
}
|
||||
|
||||
void av1_k_means(const float *data, float *centroids, uint8_t *indices, int n,
|
||||
int k, int dim, int max_itr) {
|
||||
int i;
|
||||
float this_dist;
|
||||
float pre_centroids[2 * PALETTE_MAX_SIZE];
|
||||
uint8_t pre_indices[MAX_SB_SQUARE];
|
||||
|
||||
av1_calc_indices(data, centroids, indices, n, k, dim);
|
||||
this_dist = calc_total_dist(data, centroids, indices, n, k, dim);
|
||||
|
||||
for (i = 0; i < max_itr; ++i) {
|
||||
const float pre_dist = this_dist;
|
||||
memcpy(pre_centroids, centroids, sizeof(pre_centroids[0]) * k * dim);
|
||||
memcpy(pre_indices, indices, sizeof(pre_indices[0]) * n);
|
||||
|
||||
calc_centroids(data, centroids, indices, n, k, dim);
|
||||
av1_calc_indices(data, centroids, indices, n, k, dim);
|
||||
this_dist = calc_total_dist(data, centroids, indices, n, k, dim);
|
||||
|
||||
if (this_dist > pre_dist) {
|
||||
memcpy(centroids, pre_centroids, sizeof(pre_centroids[0]) * k * dim);
|
||||
memcpy(indices, pre_indices, sizeof(pre_indices[0]) * n);
|
||||
break;
|
||||
}
|
||||
if (!memcmp(centroids, pre_centroids, sizeof(pre_centroids[0]) * k * dim))
|
||||
break;
|
||||
}
|
||||
}
|
||||
#define AV1_K_MEANS_DIM 1
|
||||
#include "av1/encoder/k_means_template.h"
|
||||
#undef AV1_K_MEANS_DIM
|
||||
#define AV1_K_MEANS_DIM 2
|
||||
#include "av1/encoder/k_means_template.h"
|
||||
#undef AV1_K_MEANS_DIM
|
||||
|
||||
static int float_comparer(const void *a, const void *b) {
|
||||
const float fa = *(const float *)a;
|
||||
|
|
|
|||
40
third_party/aom/av1/encoder/palette.h
vendored
40
third_party/aom/av1/encoder/palette.h
vendored
|
|
@ -18,17 +18,49 @@
|
|||
extern "C" {
|
||||
#endif
|
||||
|
||||
#define AV1_K_MEANS_RENAME(func, dim) func##_dim##dim
|
||||
|
||||
void AV1_K_MEANS_RENAME(av1_calc_indices, 1)(const float *data,
|
||||
const float *centroids,
|
||||
uint8_t *indices, int n, int k);
|
||||
void AV1_K_MEANS_RENAME(av1_calc_indices, 2)(const float *data,
|
||||
const float *centroids,
|
||||
uint8_t *indices, int n, int k);
|
||||
void AV1_K_MEANS_RENAME(av1_k_means, 1)(const float *data, float *centroids,
|
||||
uint8_t *indices, int n, int k,
|
||||
int max_itr);
|
||||
void AV1_K_MEANS_RENAME(av1_k_means, 2)(const float *data, float *centroids,
|
||||
uint8_t *indices, int n, int k,
|
||||
int max_itr);
|
||||
|
||||
// Given 'n' 'data' points and 'k' 'centroids' each of dimension 'dim',
|
||||
// calculate the centroid 'indices' for the data points.
|
||||
void av1_calc_indices(const float *data, const float *centroids,
|
||||
uint8_t *indices, int n, int k, int dim);
|
||||
static INLINE void av1_calc_indices(const float *data, const float *centroids,
|
||||
uint8_t *indices, int n, int k, int dim) {
|
||||
if (dim == 1) {
|
||||
AV1_K_MEANS_RENAME(av1_calc_indices, 1)(data, centroids, indices, n, k);
|
||||
} else if (dim == 2) {
|
||||
AV1_K_MEANS_RENAME(av1_calc_indices, 2)(data, centroids, indices, n, k);
|
||||
} else {
|
||||
assert(0 && "Untemplated k means dimension");
|
||||
}
|
||||
}
|
||||
|
||||
// Given 'n' 'data' points and an initial guess of 'k' 'centroids' each of
|
||||
// dimension 'dim', runs up to 'max_itr' iterations of k-means algorithm to get
|
||||
// updated 'centroids' and the centroid 'indices' for elements in 'data'.
|
||||
// Note: the output centroids are rounded off to nearest integers.
|
||||
void av1_k_means(const float *data, float *centroids, uint8_t *indices, int n,
|
||||
int k, int dim, int max_itr);
|
||||
static INLINE void av1_k_means(const float *data, float *centroids,
|
||||
uint8_t *indices, int n, int k, int dim,
|
||||
int max_itr) {
|
||||
if (dim == 1) {
|
||||
AV1_K_MEANS_RENAME(av1_k_means, 1)(data, centroids, indices, n, k, max_itr);
|
||||
} else if (dim == 2) {
|
||||
AV1_K_MEANS_RENAME(av1_k_means, 2)(data, centroids, indices, n, k, max_itr);
|
||||
} else {
|
||||
assert(0 && "Untemplated k means dimension");
|
||||
}
|
||||
}
|
||||
|
||||
// Given a list of centroids, returns the unique number of centroids 'k', and
|
||||
// puts these unique centroids in first 'k' indices of 'centroids' array.
|
||||
|
|
|
|||
161
third_party/aom/av1/encoder/pickcdef.c
vendored
161
third_party/aom/av1/encoder/pickcdef.c
vendored
|
|
@ -19,11 +19,11 @@
|
|||
#include "av1/common/reconinter.h"
|
||||
#include "av1/encoder/encoder.h"
|
||||
|
||||
#define REDUCED_STRENGTHS 8
|
||||
#define REDUCED_TOTAL_STRENGTHS (REDUCED_STRENGTHS * CLPF_STRENGTHS)
|
||||
#define TOTAL_STRENGTHS (DERING_STRENGTHS * CLPF_STRENGTHS)
|
||||
#define REDUCED_PRI_STRENGTHS 8
|
||||
#define REDUCED_TOTAL_STRENGTHS (REDUCED_PRI_STRENGTHS * CDEF_SEC_STRENGTHS)
|
||||
#define TOTAL_STRENGTHS (CDEF_PRI_STRENGTHS * CDEF_SEC_STRENGTHS)
|
||||
|
||||
static int priconv[REDUCED_STRENGTHS] = { 0, 1, 2, 3, 4, 7, 12, 25 };
|
||||
static int priconv[REDUCED_PRI_STRENGTHS] = { 0, 1, 2, 3, 4, 7, 12, 25 };
|
||||
|
||||
/* Search for the best strength to add as an option, knowing we
|
||||
already selected nb_strengths options. */
|
||||
|
|
@ -68,11 +68,16 @@ static uint64_t search_one_dual(int *lev0, int *lev1, int nb_strengths,
|
|||
uint64_t (**mse)[TOTAL_STRENGTHS], int sb_count,
|
||||
int fast) {
|
||||
uint64_t tot_mse[TOTAL_STRENGTHS][TOTAL_STRENGTHS];
|
||||
#if !CONFIG_CDEF_SINGLEPASS
|
||||
const int total_strengths = fast ? REDUCED_TOTAL_STRENGTHS : TOTAL_STRENGTHS;
|
||||
#endif
|
||||
int i, j;
|
||||
uint64_t best_tot_mse = (uint64_t)1 << 63;
|
||||
int best_id0 = 0;
|
||||
int best_id1 = 0;
|
||||
#if CONFIG_CDEF_SINGLEPASS
|
||||
const int total_strengths = fast ? REDUCED_TOTAL_STRENGTHS : TOTAL_STRENGTHS;
|
||||
#endif
|
||||
memset(tot_mse, 0, sizeof(tot_mse));
|
||||
for (i = 0; i < sb_count; i++) {
|
||||
int gi;
|
||||
|
|
@ -232,13 +237,13 @@ static INLINE uint64_t mse_4x4_16bit(uint16_t *dst, int dstride, uint16_t *src,
|
|||
}
|
||||
|
||||
/* Compute MSE only on the blocks we filtered. */
|
||||
uint64_t compute_dering_dist(uint16_t *dst, int dstride, uint16_t *src,
|
||||
dering_list *dlist, int dering_count,
|
||||
BLOCK_SIZE bsize, int coeff_shift, int pli) {
|
||||
uint64_t compute_cdef_dist(uint16_t *dst, int dstride, uint16_t *src,
|
||||
cdef_list *dlist, int cdef_count, BLOCK_SIZE bsize,
|
||||
int coeff_shift, int pli) {
|
||||
uint64_t sum = 0;
|
||||
int bi, bx, by;
|
||||
if (bsize == BLOCK_8X8) {
|
||||
for (bi = 0; bi < dering_count; bi++) {
|
||||
for (bi = 0; bi < cdef_count; bi++) {
|
||||
by = dlist[bi].by;
|
||||
bx = dlist[bi].bx;
|
||||
if (pli == 0) {
|
||||
|
|
@ -250,7 +255,7 @@ uint64_t compute_dering_dist(uint16_t *dst, int dstride, uint16_t *src,
|
|||
}
|
||||
}
|
||||
} else if (bsize == BLOCK_4X8) {
|
||||
for (bi = 0; bi < dering_count; bi++) {
|
||||
for (bi = 0; bi < cdef_count; bi++) {
|
||||
by = dlist[bi].by;
|
||||
bx = dlist[bi].bx;
|
||||
sum += mse_4x4_16bit(&dst[(by << 3) * dstride + (bx << 2)], dstride,
|
||||
|
|
@ -259,7 +264,7 @@ uint64_t compute_dering_dist(uint16_t *dst, int dstride, uint16_t *src,
|
|||
&src[(bi << (3 + 2)) + 4 * 4], 4);
|
||||
}
|
||||
} else if (bsize == BLOCK_8X4) {
|
||||
for (bi = 0; bi < dering_count; bi++) {
|
||||
for (bi = 0; bi < cdef_count; bi++) {
|
||||
by = dlist[bi].by;
|
||||
bx = dlist[bi].bx;
|
||||
sum += mse_4x4_16bit(&dst[(by << 2) * dstride + (bx << 3)], dstride,
|
||||
|
|
@ -269,7 +274,7 @@ uint64_t compute_dering_dist(uint16_t *dst, int dstride, uint16_t *src,
|
|||
}
|
||||
} else {
|
||||
assert(bsize == BLOCK_4X4);
|
||||
for (bi = 0; bi < dering_count; bi++) {
|
||||
for (bi = 0; bi < cdef_count; bi++) {
|
||||
by = dlist[bi].by;
|
||||
bx = dlist[bi].bx;
|
||||
sum += mse_4x4_16bit(&dst[(by << 2) * dstride + (bx << 2)], dstride,
|
||||
|
|
@ -282,12 +287,12 @@ uint64_t compute_dering_dist(uint16_t *dst, int dstride, uint16_t *src,
|
|||
void av1_cdef_search(YV12_BUFFER_CONFIG *frame, const YV12_BUFFER_CONFIG *ref,
|
||||
AV1_COMMON *cm, MACROBLOCKD *xd, int fast) {
|
||||
int r, c;
|
||||
int sbr, sbc;
|
||||
int fbr, fbc;
|
||||
uint16_t *src[3];
|
||||
uint16_t *ref_coeff[3];
|
||||
dering_list dlist[MI_SIZE_64X64 * MI_SIZE_64X64];
|
||||
int dir[OD_DERING_NBLOCKS][OD_DERING_NBLOCKS] = { { 0 } };
|
||||
int var[OD_DERING_NBLOCKS][OD_DERING_NBLOCKS] = { { 0 } };
|
||||
cdef_list dlist[MI_SIZE_64X64 * MI_SIZE_64X64];
|
||||
int dir[CDEF_NBLOCKS][CDEF_NBLOCKS] = { { 0 } };
|
||||
int var[CDEF_NBLOCKS][CDEF_NBLOCKS] = { { 0 } };
|
||||
int stride[3];
|
||||
int bsize[3];
|
||||
int mi_wide_l2[3];
|
||||
|
|
@ -295,18 +300,22 @@ void av1_cdef_search(YV12_BUFFER_CONFIG *frame, const YV12_BUFFER_CONFIG *ref,
|
|||
int xdec[3];
|
||||
int ydec[3];
|
||||
int pli;
|
||||
int dering_count;
|
||||
int cdef_count;
|
||||
int coeff_shift = AOMMAX(cm->bit_depth - 8, 0);
|
||||
uint64_t best_tot_mse = (uint64_t)1 << 63;
|
||||
uint64_t tot_mse;
|
||||
int sb_count;
|
||||
int nvsb = (cm->mi_rows + MI_SIZE_64X64 - 1) / MI_SIZE_64X64;
|
||||
int nhsb = (cm->mi_cols + MI_SIZE_64X64 - 1) / MI_SIZE_64X64;
|
||||
int *sb_index = aom_malloc(nvsb * nhsb * sizeof(*sb_index));
|
||||
int *selected_strength = aom_malloc(nvsb * nhsb * sizeof(*sb_index));
|
||||
int nvfb = (cm->mi_rows + MI_SIZE_64X64 - 1) / MI_SIZE_64X64;
|
||||
int nhfb = (cm->mi_cols + MI_SIZE_64X64 - 1) / MI_SIZE_64X64;
|
||||
int *sb_index = aom_malloc(nvfb * nhfb * sizeof(*sb_index));
|
||||
int *selected_strength = aom_malloc(nvfb * nhfb * sizeof(*sb_index));
|
||||
uint64_t(*mse[2])[TOTAL_STRENGTHS];
|
||||
int clpf_damping = 3 + (cm->base_qindex >> 6);
|
||||
int dering_damping = 6;
|
||||
#if CONFIG_CDEF_SINGLEPASS
|
||||
int pri_damping = 3 + (cm->base_qindex >> 6);
|
||||
#else
|
||||
int pri_damping = 6;
|
||||
#endif
|
||||
int sec_damping = 3 + (cm->base_qindex >> 6);
|
||||
int i;
|
||||
int nb_strengths;
|
||||
int nb_strength_bits;
|
||||
|
|
@ -314,19 +323,18 @@ void av1_cdef_search(YV12_BUFFER_CONFIG *frame, const YV12_BUFFER_CONFIG *ref,
|
|||
double lambda;
|
||||
int nplanes = 3;
|
||||
const int total_strengths = fast ? REDUCED_TOTAL_STRENGTHS : TOTAL_STRENGTHS;
|
||||
DECLARE_ALIGNED(32, uint16_t, inbuf[OD_DERING_INBUF_SIZE]);
|
||||
DECLARE_ALIGNED(32, uint16_t, inbuf[CDEF_INBUF_SIZE]);
|
||||
uint16_t *in;
|
||||
DECLARE_ALIGNED(32, uint16_t, tmp_dst[MAX_SB_SQUARE]);
|
||||
int chroma_dering =
|
||||
xd->plane[1].subsampling_x == xd->plane[1].subsampling_y &&
|
||||
xd->plane[2].subsampling_x == xd->plane[2].subsampling_y;
|
||||
DECLARE_ALIGNED(32, uint16_t, tmp_dst[CDEF_BLOCKSIZE * CDEF_BLOCKSIZE]);
|
||||
int chroma_cdef = xd->plane[1].subsampling_x == xd->plane[1].subsampling_y &&
|
||||
xd->plane[2].subsampling_x == xd->plane[2].subsampling_y;
|
||||
quantizer =
|
||||
av1_ac_quant(cm->base_qindex, 0, cm->bit_depth) >> (cm->bit_depth - 8);
|
||||
lambda = .12 * quantizer * quantizer / 256.;
|
||||
|
||||
av1_setup_dst_planes(xd->plane, cm->sb_size, frame, 0, 0);
|
||||
mse[0] = aom_malloc(sizeof(**mse) * nvsb * nhsb);
|
||||
mse[1] = aom_malloc(sizeof(**mse) * nvsb * nhsb);
|
||||
mse[0] = aom_malloc(sizeof(**mse) * nvfb * nhfb);
|
||||
mse[1] = aom_malloc(sizeof(**mse) * nvfb * nhfb);
|
||||
for (pli = 0; pli < nplanes; pli++) {
|
||||
uint8_t *ref_buffer;
|
||||
int ref_stride;
|
||||
|
|
@ -380,65 +388,76 @@ void av1_cdef_search(YV12_BUFFER_CONFIG *frame, const YV12_BUFFER_CONFIG *ref,
|
|||
}
|
||||
}
|
||||
}
|
||||
in = inbuf + OD_FILT_VBORDER * OD_FILT_BSTRIDE + OD_FILT_HBORDER;
|
||||
in = inbuf + CDEF_VBORDER * CDEF_BSTRIDE + CDEF_HBORDER;
|
||||
sb_count = 0;
|
||||
for (sbr = 0; sbr < nvsb; ++sbr) {
|
||||
for (sbc = 0; sbc < nhsb; ++sbc) {
|
||||
for (fbr = 0; fbr < nvfb; ++fbr) {
|
||||
for (fbc = 0; fbc < nhfb; ++fbc) {
|
||||
int nvb, nhb;
|
||||
int gi;
|
||||
int dirinit = 0;
|
||||
nhb = AOMMIN(MI_SIZE_64X64, cm->mi_cols - MI_SIZE_64X64 * sbc);
|
||||
nvb = AOMMIN(MI_SIZE_64X64, cm->mi_rows - MI_SIZE_64X64 * sbr);
|
||||
cm->mi_grid_visible[MI_SIZE_64X64 * sbr * cm->mi_stride +
|
||||
MI_SIZE_64X64 * sbc]
|
||||
nhb = AOMMIN(MI_SIZE_64X64, cm->mi_cols - MI_SIZE_64X64 * fbc);
|
||||
nvb = AOMMIN(MI_SIZE_64X64, cm->mi_rows - MI_SIZE_64X64 * fbr);
|
||||
cm->mi_grid_visible[MI_SIZE_64X64 * fbr * cm->mi_stride +
|
||||
MI_SIZE_64X64 * fbc]
|
||||
->mbmi.cdef_strength = -1;
|
||||
if (sb_all_skip(cm, sbr * MI_SIZE_64X64, sbc * MI_SIZE_64X64)) continue;
|
||||
dering_count = sb_compute_dering_list(cm, sbr * MI_SIZE_64X64,
|
||||
sbc * MI_SIZE_64X64, dlist, 1);
|
||||
if (sb_all_skip(cm, fbr * MI_SIZE_64X64, fbc * MI_SIZE_64X64)) continue;
|
||||
cdef_count = sb_compute_cdef_list(cm, fbr * MI_SIZE_64X64,
|
||||
fbc * MI_SIZE_64X64, dlist, 1);
|
||||
for (pli = 0; pli < nplanes; pli++) {
|
||||
for (i = 0; i < OD_DERING_INBUF_SIZE; i++)
|
||||
inbuf[i] = OD_DERING_VERY_LARGE;
|
||||
for (i = 0; i < CDEF_INBUF_SIZE; i++) inbuf[i] = CDEF_VERY_LARGE;
|
||||
for (gi = 0; gi < total_strengths; gi++) {
|
||||
int threshold;
|
||||
uint64_t curr_mse;
|
||||
int clpf_strength;
|
||||
threshold = gi / CLPF_STRENGTHS;
|
||||
int sec_strength;
|
||||
threshold = gi / CDEF_SEC_STRENGTHS;
|
||||
if (fast) threshold = priconv[threshold];
|
||||
if (pli > 0 && !chroma_dering) threshold = 0;
|
||||
if (pli > 0 && !chroma_cdef) threshold = 0;
|
||||
/* We avoid filtering the pixels for which some of the pixels to
|
||||
average
|
||||
are outside the frame. We could change the filter instead, but it
|
||||
would add special cases for any future vectorization. */
|
||||
int yoff = OD_FILT_VBORDER * (sbr != 0);
|
||||
int xoff = OD_FILT_HBORDER * (sbc != 0);
|
||||
int yoff = CDEF_VBORDER * (fbr != 0);
|
||||
int xoff = CDEF_HBORDER * (fbc != 0);
|
||||
int ysize = (nvb << mi_high_l2[pli]) +
|
||||
OD_FILT_VBORDER * (sbr != nvsb - 1) + yoff;
|
||||
CDEF_VBORDER * (fbr != nvfb - 1) + yoff;
|
||||
int xsize = (nhb << mi_wide_l2[pli]) +
|
||||
OD_FILT_HBORDER * (sbc != nhsb - 1) + xoff;
|
||||
clpf_strength = gi % CLPF_STRENGTHS;
|
||||
if (clpf_strength == 0)
|
||||
copy_sb16_16(&in[(-yoff * OD_FILT_BSTRIDE - xoff)], OD_FILT_BSTRIDE,
|
||||
CDEF_HBORDER * (fbc != nhfb - 1) + xoff;
|
||||
sec_strength = gi % CDEF_SEC_STRENGTHS;
|
||||
#if CONFIG_CDEF_SINGLEPASS
|
||||
copy_sb16_16(&in[(-yoff * CDEF_BSTRIDE - xoff)], CDEF_BSTRIDE,
|
||||
src[pli],
|
||||
(fbr * MI_SIZE_64X64 << mi_high_l2[pli]) - yoff,
|
||||
(fbc * MI_SIZE_64X64 << mi_wide_l2[pli]) - xoff,
|
||||
stride[pli], ysize, xsize);
|
||||
cdef_filter_fb(NULL, tmp_dst, CDEF_BSTRIDE, in, xdec[pli], ydec[pli],
|
||||
dir, &dirinit, var, pli, dlist, cdef_count, threshold,
|
||||
sec_strength + (sec_strength == 3), pri_damping,
|
||||
sec_damping, coeff_shift);
|
||||
#else
|
||||
if (sec_strength == 0)
|
||||
copy_sb16_16(&in[(-yoff * CDEF_BSTRIDE - xoff)], CDEF_BSTRIDE,
|
||||
src[pli],
|
||||
(sbr * MI_SIZE_64X64 << mi_high_l2[pli]) - yoff,
|
||||
(sbc * MI_SIZE_64X64 << mi_wide_l2[pli]) - xoff,
|
||||
(fbr * MI_SIZE_64X64 << mi_high_l2[pli]) - yoff,
|
||||
(fbc * MI_SIZE_64X64 << mi_wide_l2[pli]) - xoff,
|
||||
stride[pli], ysize, xsize);
|
||||
od_dering(clpf_strength ? NULL : (uint8_t *)in, OD_FILT_BSTRIDE,
|
||||
tmp_dst, in, xdec[pli], ydec[pli], dir, &dirinit, var, pli,
|
||||
dlist, dering_count, threshold,
|
||||
clpf_strength + (clpf_strength == 3), clpf_damping,
|
||||
dering_damping, coeff_shift, clpf_strength != 0, 1);
|
||||
curr_mse = compute_dering_dist(
|
||||
cdef_filter_fb(sec_strength ? NULL : (uint8_t *)in, CDEF_BSTRIDE,
|
||||
tmp_dst, in, xdec[pli], ydec[pli], dir, &dirinit, var,
|
||||
pli, dlist, cdef_count, threshold,
|
||||
sec_strength + (sec_strength == 3), sec_damping,
|
||||
pri_damping, coeff_shift, sec_strength != 0, 1);
|
||||
#endif
|
||||
curr_mse = compute_cdef_dist(
|
||||
ref_coeff[pli] +
|
||||
(sbr * MI_SIZE_64X64 << mi_high_l2[pli]) * stride[pli] +
|
||||
(sbc * MI_SIZE_64X64 << mi_wide_l2[pli]),
|
||||
stride[pli], tmp_dst, dlist, dering_count, bsize[pli],
|
||||
coeff_shift, pli);
|
||||
(fbr * MI_SIZE_64X64 << mi_high_l2[pli]) * stride[pli] +
|
||||
(fbc * MI_SIZE_64X64 << mi_wide_l2[pli]),
|
||||
stride[pli], tmp_dst, dlist, cdef_count, bsize[pli], coeff_shift,
|
||||
pli);
|
||||
if (pli < 2)
|
||||
mse[pli][sb_count][gi] = curr_mse;
|
||||
else
|
||||
mse[1][sb_count][gi] += curr_mse;
|
||||
sb_index[sb_count] =
|
||||
MI_SIZE_64X64 * sbr * cm->mi_stride + MI_SIZE_64X64 * sbc;
|
||||
MI_SIZE_64X64 * fbr * cm->mi_stride + MI_SIZE_64X64 * fbc;
|
||||
}
|
||||
}
|
||||
sb_count++;
|
||||
|
|
@ -494,15 +513,17 @@ void av1_cdef_search(YV12_BUFFER_CONFIG *frame, const YV12_BUFFER_CONFIG *ref,
|
|||
if (fast) {
|
||||
for (int j = 0; j < nb_strengths; j++) {
|
||||
cm->cdef_strengths[j] =
|
||||
priconv[cm->cdef_strengths[j] / CLPF_STRENGTHS] * CLPF_STRENGTHS +
|
||||
(cm->cdef_strengths[j] % CLPF_STRENGTHS);
|
||||
priconv[cm->cdef_strengths[j] / CDEF_SEC_STRENGTHS] *
|
||||
CDEF_SEC_STRENGTHS +
|
||||
(cm->cdef_strengths[j] % CDEF_SEC_STRENGTHS);
|
||||
cm->cdef_uv_strengths[j] =
|
||||
priconv[cm->cdef_uv_strengths[j] / CLPF_STRENGTHS] * CLPF_STRENGTHS +
|
||||
(cm->cdef_uv_strengths[j] % CLPF_STRENGTHS);
|
||||
priconv[cm->cdef_uv_strengths[j] / CDEF_SEC_STRENGTHS] *
|
||||
CDEF_SEC_STRENGTHS +
|
||||
(cm->cdef_uv_strengths[j] % CDEF_SEC_STRENGTHS);
|
||||
}
|
||||
}
|
||||
cm->cdef_dering_damping = dering_damping;
|
||||
cm->cdef_clpf_damping = clpf_damping;
|
||||
cm->cdef_pri_damping = pri_damping;
|
||||
cm->cdef_sec_damping = sec_damping;
|
||||
aom_free(mse[0]);
|
||||
aom_free(mse[1]);
|
||||
for (pli = 0; pli < nplanes; pli++) {
|
||||
|
|
|
|||
404
third_party/aom/av1/encoder/picklpf.c
vendored
404
third_party/aom/av1/encoder/picklpf.c
vendored
|
|
@ -14,8 +14,8 @@
|
|||
|
||||
#include "./aom_scale_rtcd.h"
|
||||
|
||||
#include "aom_dsp/psnr.h"
|
||||
#include "aom_dsp/aom_dsp_common.h"
|
||||
#include "aom_dsp/psnr.h"
|
||||
#include "aom_mem/aom_mem.h"
|
||||
#include "aom_ports/mem.h"
|
||||
|
||||
|
|
@ -27,6 +27,85 @@
|
|||
#include "av1/encoder/encoder.h"
|
||||
#include "av1/encoder/picklpf.h"
|
||||
|
||||
#if CONFIG_LPF_SB
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
static int compute_sb_y_sse_highbd(const YV12_BUFFER_CONFIG *src,
|
||||
const YV12_BUFFER_CONFIG *frame,
|
||||
AV1_COMMON *const cm, int mi_row,
|
||||
int mi_col) {
|
||||
int sse = 0;
|
||||
const int mi_row_start = AOMMAX(0, mi_row - FILT_BOUNDARY_MI_OFFSET);
|
||||
const int mi_col_start = AOMMAX(0, mi_col - FILT_BOUNDARY_MI_OFFSET);
|
||||
const int mi_row_range = mi_row - FILT_BOUNDARY_MI_OFFSET + MAX_MIB_SIZE;
|
||||
const int mi_col_range = mi_col - FILT_BOUNDARY_MI_OFFSET + MAX_MIB_SIZE;
|
||||
const int mi_row_end = AOMMIN(mi_row_range, cm->mi_rows);
|
||||
const int mi_col_end = AOMMIN(mi_col_range, cm->mi_cols);
|
||||
|
||||
const int row = mi_row_start * MI_SIZE;
|
||||
const int col = mi_col_start * MI_SIZE;
|
||||
const uint16_t *src_y =
|
||||
CONVERT_TO_SHORTPTR(src->y_buffer) + row * src->y_stride + col;
|
||||
const uint16_t *frame_y =
|
||||
CONVERT_TO_SHORTPTR(frame->y_buffer) + row * frame->y_stride + col;
|
||||
const int row_end = (mi_row_end - mi_row_start) * MI_SIZE;
|
||||
const int col_end = (mi_col_end - mi_col_start) * MI_SIZE;
|
||||
|
||||
int x, y;
|
||||
for (y = 0; y < row_end; ++y) {
|
||||
for (x = 0; x < col_end; ++x) {
|
||||
const int diff = src_y[x] - frame_y[x];
|
||||
sse += diff * diff;
|
||||
}
|
||||
src_y += src->y_stride;
|
||||
frame_y += frame->y_stride;
|
||||
}
|
||||
return sse;
|
||||
}
|
||||
#endif
|
||||
|
||||
static int compute_sb_y_sse(const YV12_BUFFER_CONFIG *src,
|
||||
const YV12_BUFFER_CONFIG *frame,
|
||||
AV1_COMMON *const cm, int mi_row, int mi_col) {
|
||||
int sse = 0;
|
||||
const int mi_row_start = AOMMAX(0, mi_row - FILT_BOUNDARY_MI_OFFSET);
|
||||
const int mi_col_start = AOMMAX(0, mi_col - FILT_BOUNDARY_MI_OFFSET);
|
||||
const int mi_row_range = mi_row - FILT_BOUNDARY_MI_OFFSET + MAX_MIB_SIZE;
|
||||
const int mi_col_range = mi_col - FILT_BOUNDARY_MI_OFFSET + MAX_MIB_SIZE;
|
||||
const int mi_row_end = AOMMIN(mi_row_range, cm->mi_rows);
|
||||
const int mi_col_end = AOMMIN(mi_col_range, cm->mi_cols);
|
||||
|
||||
const int row = mi_row_start * MI_SIZE;
|
||||
const int col = mi_col_start * MI_SIZE;
|
||||
const uint8_t *src_y = src->y_buffer + row * src->y_stride + col;
|
||||
const uint8_t *frame_y = frame->y_buffer + row * frame->y_stride + col;
|
||||
const int row_end = (mi_row_end - mi_row_start) * MI_SIZE;
|
||||
const int col_end = (mi_col_end - mi_col_start) * MI_SIZE;
|
||||
|
||||
int x, y;
|
||||
for (y = 0; y < row_end; ++y) {
|
||||
for (x = 0; x < col_end; ++x) {
|
||||
const int diff = src_y[x] - frame_y[x];
|
||||
sse += diff * diff;
|
||||
}
|
||||
src_y += src->y_stride;
|
||||
frame_y += frame->y_stride;
|
||||
}
|
||||
return sse;
|
||||
}
|
||||
#endif // CONFIG_LPF_SB
|
||||
|
||||
#if !CONFIG_LPF_SB
|
||||
static void yv12_copy_plane(const YV12_BUFFER_CONFIG *src_bc,
|
||||
YV12_BUFFER_CONFIG *dst_bc, int plane) {
|
||||
switch (plane) {
|
||||
case 0: aom_yv12_copy_y(src_bc, dst_bc); break;
|
||||
case 1: aom_yv12_copy_u(src_bc, dst_bc); break;
|
||||
case 2: aom_yv12_copy_v(src_bc, dst_bc); break;
|
||||
default: assert(plane >= 0 && plane <= 2); break;
|
||||
}
|
||||
}
|
||||
#endif // CONFIG_LPF_SB
|
||||
|
||||
int av1_get_max_filter_level(const AV1_COMP *cpi) {
|
||||
if (cpi->oxcf.pass == 2) {
|
||||
return cpi->twopass.section_intra_rating > 8 ? MAX_LOOP_FILTER * 3 / 4
|
||||
|
|
@ -36,25 +115,17 @@ int av1_get_max_filter_level(const AV1_COMP *cpi) {
|
|||
}
|
||||
}
|
||||
|
||||
static int64_t try_filter_frame(const YV12_BUFFER_CONFIG *sd,
|
||||
AV1_COMP *const cpi, int filt_level,
|
||||
int partial_frame
|
||||
#if CONFIG_UV_LVL
|
||||
,
|
||||
int plane
|
||||
#endif
|
||||
) {
|
||||
#if CONFIG_LPF_SB
|
||||
// TODO(chengchen): reduce memory usage by copy superblock instead of frame
|
||||
static int try_filter_superblock(const YV12_BUFFER_CONFIG *sd,
|
||||
AV1_COMP *const cpi, int filt_level,
|
||||
int partial_frame, int mi_row, int mi_col) {
|
||||
AV1_COMMON *const cm = &cpi->common;
|
||||
int64_t filt_err;
|
||||
int filt_err;
|
||||
|
||||
#if CONFIG_VAR_TX || CONFIG_EXT_PARTITION || CONFIG_CB4X4
|
||||
#if CONFIG_UV_LVL
|
||||
av1_loop_filter_frame(cm->frame_to_show, cm, &cpi->td.mb.e_mbd, filt_level,
|
||||
plane, partial_frame);
|
||||
#else
|
||||
av1_loop_filter_frame(cm->frame_to_show, cm, &cpi->td.mb.e_mbd, filt_level, 1,
|
||||
partial_frame);
|
||||
#endif // CONFIG_UV_LVL
|
||||
partial_frame, mi_row, mi_col);
|
||||
#else
|
||||
if (cpi->num_workers > 1)
|
||||
av1_loop_filter_frame_mt(cm->frame_to_show, cm, cpi->td.mb.e_mbd.plane,
|
||||
|
|
@ -65,64 +136,172 @@ static int64_t try_filter_frame(const YV12_BUFFER_CONFIG *sd,
|
|||
1, partial_frame);
|
||||
#endif
|
||||
|
||||
#if CONFIG_UV_LVL
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
if (cm->use_highbitdepth) {
|
||||
if (plane == 0)
|
||||
filt_err = aom_highbd_get_y_sse(sd, cm->frame_to_show);
|
||||
else if (plane == 1)
|
||||
filt_err = aom_highbd_get_u_sse(sd, cm->frame_to_show);
|
||||
else
|
||||
filt_err = aom_highbd_get_v_sse(sd, cm->frame_to_show);
|
||||
filt_err =
|
||||
compute_sb_y_sse_highbd(sd, cm->frame_to_show, cm, mi_row, mi_col);
|
||||
} else {
|
||||
if (plane == 0)
|
||||
filt_err = aom_get_y_sse(sd, cm->frame_to_show);
|
||||
else if (plane == 1)
|
||||
filt_err = aom_get_u_sse(sd, cm->frame_to_show);
|
||||
else
|
||||
filt_err = aom_get_v_sse(sd, cm->frame_to_show);
|
||||
filt_err = compute_sb_y_sse(sd, cm->frame_to_show, cm, mi_row, mi_col);
|
||||
}
|
||||
#else
|
||||
if (plane == 0)
|
||||
filt_err = aom_get_y_sse(sd, cm->frame_to_show);
|
||||
else if (plane == 1)
|
||||
filt_err = aom_get_u_sse(sd, cm->frame_to_show);
|
||||
else
|
||||
filt_err = aom_get_v_sse(sd, cm->frame_to_show);
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
|
||||
// Re-instate the unfiltered frame
|
||||
if (plane == 0)
|
||||
aom_yv12_copy_y(&cpi->last_frame_uf, cm->frame_to_show);
|
||||
else if (plane == 1)
|
||||
aom_yv12_copy_u(&cpi->last_frame_uf, cm->frame_to_show);
|
||||
else
|
||||
aom_yv12_copy_v(&cpi->last_frame_uf, cm->frame_to_show);
|
||||
#else
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
if (cm->use_highbitdepth) {
|
||||
filt_err = aom_highbd_get_y_sse(sd, cm->frame_to_show);
|
||||
} else {
|
||||
filt_err = aom_get_y_sse(sd, cm->frame_to_show);
|
||||
}
|
||||
#else
|
||||
filt_err = aom_get_y_sse(sd, cm->frame_to_show);
|
||||
filt_err = compute_sb_y_sse(sd, cm->frame_to_show, cm, mi_row, mi_col);
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
|
||||
// TODO(chengchen): Copy the superblock only
|
||||
// Re-instate the unfiltered frame
|
||||
aom_yv12_copy_y(&cpi->last_frame_uf, cm->frame_to_show);
|
||||
#endif // CONFIG_UV_LVL
|
||||
|
||||
return filt_err;
|
||||
}
|
||||
|
||||
int av1_search_filter_level(const YV12_BUFFER_CONFIG *sd, AV1_COMP *cpi,
|
||||
int partial_frame, double *best_cost_ret
|
||||
#if CONFIG_UV_LVL
|
||||
,
|
||||
int plane
|
||||
static int search_filter_level(const YV12_BUFFER_CONFIG *sd, AV1_COMP *cpi,
|
||||
int partial_frame, double *best_cost_ret,
|
||||
int mi_row, int mi_col, int last_lvl) {
|
||||
assert(partial_frame == 1);
|
||||
assert(last_lvl >= 0);
|
||||
|
||||
const AV1_COMMON *const cm = &cpi->common;
|
||||
MACROBLOCK *x = &cpi->td.mb;
|
||||
|
||||
int min_filter_level = AOMMAX(0, last_lvl - MAX_LPF_OFFSET);
|
||||
int max_filter_level =
|
||||
AOMMIN(av1_get_max_filter_level(cpi), last_lvl + MAX_LPF_OFFSET);
|
||||
|
||||
// search a larger range for the start superblock
|
||||
if (mi_row == 0 && mi_col == 0) {
|
||||
min_filter_level = 0;
|
||||
max_filter_level = av1_get_max_filter_level(cpi);
|
||||
}
|
||||
|
||||
// TODO(chengchen): Copy for superblock only
|
||||
// Make a copy of the unfiltered / processed recon buffer
|
||||
aom_yv12_copy_y(cm->frame_to_show, &cpi->last_frame_uf);
|
||||
|
||||
int estimate_err =
|
||||
try_filter_superblock(sd, cpi, last_lvl, partial_frame, mi_row, mi_col);
|
||||
|
||||
int best_err = estimate_err;
|
||||
int filt_best = last_lvl;
|
||||
|
||||
int i;
|
||||
for (i = min_filter_level; i <= max_filter_level; i += LPF_STEP) {
|
||||
if (i == last_lvl) continue;
|
||||
|
||||
int filt_err =
|
||||
try_filter_superblock(sd, cpi, i, partial_frame, mi_row, mi_col);
|
||||
|
||||
if (filt_err < best_err) {
|
||||
best_err = filt_err;
|
||||
filt_best = i;
|
||||
}
|
||||
}
|
||||
|
||||
// If previous sb filter level has similar filtering performance as current
|
||||
// best filter level, use previous level such that we can only send one bit
|
||||
// to indicate current filter level is the same as the previous.
|
||||
int threshold = 400;
|
||||
|
||||
// ratio = the filtering area / a superblock size
|
||||
int ratio = 1;
|
||||
if (mi_row + MAX_MIB_SIZE > cm->mi_rows) {
|
||||
ratio *= (cm->mi_rows - mi_row);
|
||||
} else {
|
||||
if (mi_row == 0) {
|
||||
ratio *= (MAX_MIB_SIZE - FILT_BOUNDARY_MI_OFFSET);
|
||||
} else {
|
||||
ratio *= MAX_MIB_SIZE;
|
||||
}
|
||||
}
|
||||
if (mi_col + MAX_MIB_SIZE > cm->mi_cols) {
|
||||
ratio *= (cm->mi_cols - mi_col);
|
||||
} else {
|
||||
if (mi_col == 0) {
|
||||
ratio *= (MAX_MIB_SIZE - FILT_BOUNDARY_MI_OFFSET);
|
||||
} else {
|
||||
ratio *= MAX_MIB_SIZE;
|
||||
}
|
||||
}
|
||||
threshold = threshold * ratio / (MAX_MIB_SIZE * MAX_MIB_SIZE);
|
||||
|
||||
const int diff = abs(estimate_err - best_err);
|
||||
|
||||
const int percent_thresh = (int)((double)estimate_err * 0.01);
|
||||
threshold = AOMMAX(threshold, percent_thresh);
|
||||
if (diff < threshold) {
|
||||
best_err = estimate_err;
|
||||
filt_best = last_lvl;
|
||||
}
|
||||
|
||||
// Compute rdcost to determine whether to reuse previous filter lvl
|
||||
if (filt_best != last_lvl) {
|
||||
}
|
||||
|
||||
if (best_cost_ret) *best_cost_ret = RDCOST_DBL(x->rdmult, 0, best_err);
|
||||
return filt_best;
|
||||
}
|
||||
|
||||
#else // CONFIG_LPF_SB
|
||||
static int64_t try_filter_frame(const YV12_BUFFER_CONFIG *sd,
|
||||
AV1_COMP *const cpi, int filt_level,
|
||||
int partial_frame
|
||||
#if CONFIG_LOOPFILTER_LEVEL
|
||||
,
|
||||
int plane, int dir
|
||||
#endif
|
||||
) {
|
||||
) {
|
||||
AV1_COMMON *const cm = &cpi->common;
|
||||
int64_t filt_err;
|
||||
|
||||
#if CONFIG_VAR_TX || CONFIG_EXT_PARTITION || CONFIG_CB4X4
|
||||
#if CONFIG_LOOPFILTER_LEVEL
|
||||
assert(plane >= 0 && plane <= 2);
|
||||
int filter_level[2] = { filt_level, filt_level };
|
||||
if (plane == 0 && dir == 0) filter_level[1] = cm->lf.filter_level[1];
|
||||
if (plane == 0 && dir == 1) filter_level[0] = cm->lf.filter_level[0];
|
||||
|
||||
av1_loop_filter_frame(cm->frame_to_show, cm, &cpi->td.mb.e_mbd,
|
||||
filter_level[0], filter_level[1], plane, partial_frame);
|
||||
#else
|
||||
av1_loop_filter_frame(cm->frame_to_show, cm, &cpi->td.mb.e_mbd, filt_level, 1,
|
||||
partial_frame);
|
||||
#endif // CONFIG_LOOPFILTER_LEVEL
|
||||
#else
|
||||
if (cpi->num_workers > 1)
|
||||
av1_loop_filter_frame_mt(cm->frame_to_show, cm, cpi->td.mb.e_mbd.plane,
|
||||
filt_level, 1, partial_frame, cpi->workers,
|
||||
cpi->num_workers, &cpi->lf_row_sync);
|
||||
else
|
||||
av1_loop_filter_frame(cm->frame_to_show, cm, &cpi->td.mb.e_mbd, filt_level,
|
||||
1, partial_frame);
|
||||
#endif
|
||||
|
||||
int highbd = 0;
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
highbd = cm->use_highbitdepth;
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
|
||||
#if CONFIG_LOOPFILTER_LEVEL
|
||||
filt_err = aom_get_sse_plane(sd, cm->frame_to_show, plane, highbd);
|
||||
|
||||
// Re-instate the unfiltered frame
|
||||
yv12_copy_plane(&cpi->last_frame_uf, cm->frame_to_show, plane);
|
||||
#else
|
||||
filt_err = aom_get_sse_plane(sd, cm->frame_to_show, 0, highbd);
|
||||
|
||||
// Re-instate the unfiltered frame
|
||||
yv12_copy_plane(&cpi->last_frame_uf, cm->frame_to_show, 0);
|
||||
#endif // CONFIG_LOOPFILTER_LEVEL
|
||||
|
||||
return filt_err;
|
||||
}
|
||||
|
||||
static int search_filter_level(const YV12_BUFFER_CONFIG *sd, AV1_COMP *cpi,
|
||||
int partial_frame, double *best_cost_ret
|
||||
#if CONFIG_LOOPFILTER_LEVEL
|
||||
,
|
||||
int plane, int dir
|
||||
#endif
|
||||
) {
|
||||
const AV1_COMMON *const cm = &cpi->common;
|
||||
const struct loopfilter *const lf = &cm->lf;
|
||||
const int min_filter_level = 0;
|
||||
|
|
@ -134,18 +313,18 @@ int av1_search_filter_level(const YV12_BUFFER_CONFIG *sd, AV1_COMP *cpi,
|
|||
|
||||
// Start the search at the previous frame filter level unless it is now out of
|
||||
// range.
|
||||
#if CONFIG_UV_LVL
|
||||
#if CONFIG_LOOPFILTER_LEVEL
|
||||
int lvl;
|
||||
switch (plane) {
|
||||
case 0: lvl = lf->filter_level; break;
|
||||
case 0: lvl = (dir == 1) ? lf->filter_level[1] : lf->filter_level[0]; break;
|
||||
case 1: lvl = lf->filter_level_u; break;
|
||||
case 2: lvl = lf->filter_level_v; break;
|
||||
default: lvl = lf->filter_level; break;
|
||||
default: assert(plane >= 0 && plane <= 2); return 0;
|
||||
}
|
||||
int filt_mid = clamp(lvl, min_filter_level, max_filter_level);
|
||||
#else
|
||||
int filt_mid = clamp(lf->filter_level, min_filter_level, max_filter_level);
|
||||
#endif // CONFIG_UV_LVL
|
||||
#endif // CONFIG_LOOPFILTER_LEVEL
|
||||
int filter_step = filt_mid < 16 ? 4 : filt_mid / 4;
|
||||
// Sum squared error at each filter level
|
||||
int64_t ss_err[MAX_LOOP_FILTER + 1];
|
||||
|
|
@ -153,23 +332,18 @@ int av1_search_filter_level(const YV12_BUFFER_CONFIG *sd, AV1_COMP *cpi,
|
|||
// Set each entry to -1
|
||||
memset(ss_err, 0xFF, sizeof(ss_err));
|
||||
|
||||
#if CONFIG_UV_LVL
|
||||
if (plane == 0)
|
||||
aom_yv12_copy_y(cm->frame_to_show, &cpi->last_frame_uf);
|
||||
else if (plane == 1)
|
||||
aom_yv12_copy_u(cm->frame_to_show, &cpi->last_frame_uf);
|
||||
else if (plane == 2)
|
||||
aom_yv12_copy_v(cm->frame_to_show, &cpi->last_frame_uf);
|
||||
#if CONFIG_LOOPFILTER_LEVEL
|
||||
yv12_copy_plane(cm->frame_to_show, &cpi->last_frame_uf, plane);
|
||||
#else
|
||||
// Make a copy of the unfiltered / processed recon buffer
|
||||
aom_yv12_copy_y(cm->frame_to_show, &cpi->last_frame_uf);
|
||||
#endif // CONFIG_UV_LVL
|
||||
#endif // CONFIG_LOOPFILTER_LEVEL
|
||||
|
||||
#if CONFIG_UV_LVL
|
||||
best_err = try_filter_frame(sd, cpi, filt_mid, partial_frame, plane);
|
||||
#if CONFIG_LOOPFILTER_LEVEL
|
||||
best_err = try_filter_frame(sd, cpi, filt_mid, partial_frame, plane, dir);
|
||||
#else
|
||||
best_err = try_filter_frame(sd, cpi, filt_mid, partial_frame);
|
||||
#endif // CONFIG_UV_LVL
|
||||
#endif // CONFIG_LOOPFILTER_LEVEL
|
||||
filt_best = filt_mid;
|
||||
ss_err[filt_mid] = best_err;
|
||||
|
||||
|
|
@ -189,12 +363,12 @@ int av1_search_filter_level(const YV12_BUFFER_CONFIG *sd, AV1_COMP *cpi,
|
|||
if (filt_direction <= 0 && filt_low != filt_mid) {
|
||||
// Get Low filter error score
|
||||
if (ss_err[filt_low] < 0) {
|
||||
#if CONFIG_UV_LVL
|
||||
#if CONFIG_LOOPFILTER_LEVEL
|
||||
ss_err[filt_low] =
|
||||
try_filter_frame(sd, cpi, filt_low, partial_frame, plane);
|
||||
try_filter_frame(sd, cpi, filt_low, partial_frame, plane, dir);
|
||||
#else
|
||||
ss_err[filt_low] = try_filter_frame(sd, cpi, filt_low, partial_frame);
|
||||
#endif // CONFIG_UV_LVL
|
||||
#endif // CONFIG_LOOPFILTER_LEVEL
|
||||
}
|
||||
// If value is close to the best so far then bias towards a lower loop
|
||||
// filter value.
|
||||
|
|
@ -210,12 +384,12 @@ int av1_search_filter_level(const YV12_BUFFER_CONFIG *sd, AV1_COMP *cpi,
|
|||
// Now look at filt_high
|
||||
if (filt_direction >= 0 && filt_high != filt_mid) {
|
||||
if (ss_err[filt_high] < 0) {
|
||||
#if CONFIG_UV_LVL
|
||||
#if CONFIG_LOOPFILTER_LEVEL
|
||||
ss_err[filt_high] =
|
||||
try_filter_frame(sd, cpi, filt_high, partial_frame, plane);
|
||||
try_filter_frame(sd, cpi, filt_high, partial_frame, plane, dir);
|
||||
#else
|
||||
ss_err[filt_high] = try_filter_frame(sd, cpi, filt_high, partial_frame);
|
||||
#endif // CONFIG_UV_LVL
|
||||
#endif // CONFIG_LOOPFILTER_LEVEL
|
||||
}
|
||||
// If value is significantly better than previous best, bias added against
|
||||
// raising filter value
|
||||
|
|
@ -241,6 +415,7 @@ int av1_search_filter_level(const YV12_BUFFER_CONFIG *sd, AV1_COMP *cpi,
|
|||
if (best_cost_ret) *best_cost_ret = RDCOST_DBL(x->rdmult, 0, best_err);
|
||||
return filt_best;
|
||||
}
|
||||
#endif // CONFIG_LPF_SB
|
||||
|
||||
void av1_pick_filter_level(const YV12_BUFFER_CONFIG *sd, AV1_COMP *cpi,
|
||||
LPF_PICK_METHOD method) {
|
||||
|
|
@ -249,8 +424,13 @@ void av1_pick_filter_level(const YV12_BUFFER_CONFIG *sd, AV1_COMP *cpi,
|
|||
|
||||
lf->sharpness_level = cm->frame_type == KEY_FRAME ? 0 : cpi->oxcf.sharpness;
|
||||
|
||||
if (method == LPF_PICK_MINIMAL_LPF && lf->filter_level) {
|
||||
if (method == LPF_PICK_MINIMAL_LPF) {
|
||||
#if CONFIG_LOOPFILTER_LEVEL
|
||||
lf->filter_level[0] = 0;
|
||||
lf->filter_level[1] = 0;
|
||||
#else
|
||||
lf->filter_level = 0;
|
||||
#endif
|
||||
} else if (method >= LPF_PICK_FROM_Q) {
|
||||
const int min_filter_level = 0;
|
||||
const int max_filter_level = av1_get_max_filter_level(cpi);
|
||||
|
|
@ -279,18 +459,54 @@ void av1_pick_filter_level(const YV12_BUFFER_CONFIG *sd, AV1_COMP *cpi,
|
|||
int filt_guess = ROUND_POWER_OF_TWO(q * 20723 + 1015158, 18);
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
if (cm->frame_type == KEY_FRAME) filt_guess -= 4;
|
||||
lf->filter_level = clamp(filt_guess, min_filter_level, max_filter_level);
|
||||
} else {
|
||||
#if CONFIG_UV_LVL
|
||||
lf->filter_level = av1_search_filter_level(
|
||||
sd, cpi, method == LPF_PICK_FROM_SUBIMAGE, NULL, 0);
|
||||
lf->filter_level_u = av1_search_filter_level(
|
||||
sd, cpi, method == LPF_PICK_FROM_SUBIMAGE, NULL, 1);
|
||||
lf->filter_level_v = av1_search_filter_level(
|
||||
sd, cpi, method == LPF_PICK_FROM_SUBIMAGE, NULL, 2);
|
||||
#if CONFIG_LOOPFILTER_LEVEL
|
||||
lf->filter_level[0] = clamp(filt_guess, min_filter_level, max_filter_level);
|
||||
lf->filter_level[1] = clamp(filt_guess, min_filter_level, max_filter_level);
|
||||
#else
|
||||
lf->filter_level = av1_search_filter_level(
|
||||
sd, cpi, method == LPF_PICK_FROM_SUBIMAGE, NULL);
|
||||
#endif // CONFIG_UV_LVL
|
||||
lf->filter_level = clamp(filt_guess, min_filter_level, max_filter_level);
|
||||
#endif
|
||||
} else {
|
||||
#if CONFIG_LPF_SB
|
||||
int mi_row, mi_col;
|
||||
// TODO(chengchen): init last_lvl using previous frame's info?
|
||||
int last_lvl = 0;
|
||||
// TODO(chengchen): if the frame size makes the last superblock very small,
|
||||
// consider merge it to the previous superblock to save bits.
|
||||
// Example, if frame size 1080x720, then in the last row of superblock,
|
||||
// there're (FILT_BOUNDAR_OFFSET + 16) pixels.
|
||||
for (mi_row = 0; mi_row < cm->mi_rows; mi_row += MAX_MIB_SIZE) {
|
||||
for (mi_col = 0; mi_col < cm->mi_cols; mi_col += MAX_MIB_SIZE) {
|
||||
int lvl =
|
||||
search_filter_level(sd, cpi, 1, NULL, mi_row, mi_col, last_lvl);
|
||||
|
||||
av1_loop_filter_sb_level_init(cm, mi_row, mi_col, lvl);
|
||||
|
||||
// For the superblock at row start, its previous filter level should be
|
||||
// the one above it, not the one at the end of last row
|
||||
if (mi_col + MAX_MIB_SIZE >= cm->mi_cols) {
|
||||
last_lvl = cm->mi_grid_visible[mi_row * cm->mi_stride]->mbmi.filt_lvl;
|
||||
} else {
|
||||
last_lvl = lvl;
|
||||
}
|
||||
}
|
||||
}
|
||||
#else // CONFIG_LPF_SB
|
||||
#if CONFIG_LOOPFILTER_LEVEL
|
||||
lf->filter_level[0] = lf->filter_level[1] = search_filter_level(
|
||||
sd, cpi, method == LPF_PICK_FROM_SUBIMAGE, NULL, 0, 2);
|
||||
lf->filter_level[0] = search_filter_level(
|
||||
sd, cpi, method == LPF_PICK_FROM_SUBIMAGE, NULL, 0, 0);
|
||||
lf->filter_level[1] = search_filter_level(
|
||||
sd, cpi, method == LPF_PICK_FROM_SUBIMAGE, NULL, 0, 1);
|
||||
|
||||
lf->filter_level_u = search_filter_level(
|
||||
sd, cpi, method == LPF_PICK_FROM_SUBIMAGE, NULL, 1, 0);
|
||||
lf->filter_level_v = search_filter_level(
|
||||
sd, cpi, method == LPF_PICK_FROM_SUBIMAGE, NULL, 2, 0);
|
||||
#else
|
||||
lf->filter_level =
|
||||
search_filter_level(sd, cpi, method == LPF_PICK_FROM_SUBIMAGE, NULL);
|
||||
#endif // CONFIG_LOOPFILTER_LEVEL
|
||||
#endif // CONFIG_LPF_SB
|
||||
}
|
||||
}
|
||||
|
|
|
|||
7
third_party/aom/av1/encoder/picklpf.h
vendored
7
third_party/aom/av1/encoder/picklpf.h
vendored
|
|
@ -21,13 +21,6 @@ extern "C" {
|
|||
struct yv12_buffer_config;
|
||||
struct AV1_COMP;
|
||||
int av1_get_max_filter_level(const AV1_COMP *cpi);
|
||||
#if CONFIG_UV_LVL
|
||||
int av1_search_filter_level(const YV12_BUFFER_CONFIG *sd, AV1_COMP *cpi,
|
||||
int partial_frame, double *err, int plane);
|
||||
#else
|
||||
int av1_search_filter_level(const YV12_BUFFER_CONFIG *sd, AV1_COMP *cpi,
|
||||
int partial_frame, double *err);
|
||||
#endif
|
||||
void av1_pick_filter_level(const struct yv12_buffer_config *sd,
|
||||
struct AV1_COMP *cpi, LPF_PICK_METHOD method);
|
||||
#ifdef __cplusplus
|
||||
|
|
|
|||
1198
third_party/aom/av1/encoder/pickrst.c
vendored
1198
third_party/aom/av1/encoder/pickrst.c
vendored
File diff suppressed because it is too large
Load diff
29
third_party/aom/av1/encoder/random.h
vendored
Normal file
29
third_party/aom/av1/encoder/random.h
vendored
Normal file
|
|
@ -0,0 +1,29 @@
|
|||
/*
|
||||
* Copyright (c) 2017, Alliance for Open Media. All rights reserved
|
||||
*
|
||||
* This source code is subject to the terms of the BSD 2 Clause License and
|
||||
* the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
|
||||
* was not distributed with this source code in the LICENSE file, you can
|
||||
* obtain it at www.aomedia.org/license/software. If the Alliance for Open
|
||||
* Media Patent License 1.0 was not distributed with this source code in the
|
||||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#ifndef AV1_ENCODER_RANDOM_H_
|
||||
#define AV1_ENCODER_RANDOM_H_
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
|
||||
// Generate a random number in the range [0, 32768).
|
||||
static INLINE unsigned int lcg_rand16(unsigned int *state) {
|
||||
*state = (unsigned int)(*state * 1103515245ULL + 12345);
|
||||
return *state / 65536 % 32768;
|
||||
}
|
||||
|
||||
#ifdef __cplusplus
|
||||
} // extern "C"
|
||||
#endif
|
||||
|
||||
#endif // AV1_ENCODER_RANDOM_H_
|
||||
7
third_party/aom/av1/encoder/ransac.c
vendored
7
third_party/aom/av1/encoder/ransac.c
vendored
|
|
@ -17,6 +17,7 @@
|
|||
|
||||
#include "av1/encoder/ransac.h"
|
||||
#include "av1/encoder/mathutils.h"
|
||||
#include "av1/encoder/random.h"
|
||||
|
||||
#define MAX_MINPTS 4
|
||||
#define MAX_DEGENERATE_ITER 10
|
||||
|
|
@ -587,12 +588,6 @@ static int find_homography(int np, double *pts1, double *pts2, double *mat) {
|
|||
return 0;
|
||||
}
|
||||
|
||||
// Generate a random number in the range [0, 32768).
|
||||
static unsigned int lcg_rand16(unsigned int *state) {
|
||||
*state = (unsigned int)(*state * 1103515245ULL + 12345);
|
||||
return *state / 65536 % 32768;
|
||||
}
|
||||
|
||||
static int get_rand_indices(int npoints, int minpts, int *indices,
|
||||
unsigned int *seed) {
|
||||
int i, j;
|
||||
|
|
|
|||
215
third_party/aom/av1/encoder/ratectrl.c
vendored
215
third_party/aom/av1/encoder/ratectrl.c
vendored
|
|
@ -29,6 +29,7 @@
|
|||
#include "av1/common/seg_common.h"
|
||||
|
||||
#include "av1/encoder/encodemv.h"
|
||||
#include "av1/encoder/random.h"
|
||||
#include "av1/encoder/ratectrl.h"
|
||||
|
||||
// Max rate target for 1080P and below encodes under normal circumstances
|
||||
|
|
@ -93,9 +94,11 @@ static int gf_low = 400;
|
|||
static int kf_high = 5000;
|
||||
static int kf_low = 400;
|
||||
|
||||
double av1_resize_rate_factor(const AV1_COMP *cpi) {
|
||||
return (double)(cpi->oxcf.width * cpi->oxcf.height) /
|
||||
(cpi->common.width * cpi->common.height);
|
||||
// How many times less pixels there are to encode given the current scaling.
|
||||
// Temporary replacement for rcf_mult and rate_thresh_mult.
|
||||
static double resize_rate_factor(const AV1_COMP *cpi, int width, int height) {
|
||||
(void)cpi;
|
||||
return (double)(cpi->oxcf.width * cpi->oxcf.height) / (width * height);
|
||||
}
|
||||
|
||||
// Functions to compute the active minq lookup table entries based on a
|
||||
|
|
@ -371,7 +374,8 @@ int av1_rc_drop_frame(AV1_COMP *cpi) {
|
|||
}
|
||||
}
|
||||
|
||||
static double get_rate_correction_factor(const AV1_COMP *cpi) {
|
||||
static double get_rate_correction_factor(const AV1_COMP *cpi, int width,
|
||||
int height) {
|
||||
const RATE_CONTROL *const rc = &cpi->rc;
|
||||
double rcf;
|
||||
|
||||
|
|
@ -389,15 +393,16 @@ static double get_rate_correction_factor(const AV1_COMP *cpi) {
|
|||
else
|
||||
rcf = rc->rate_correction_factors[INTER_NORMAL];
|
||||
}
|
||||
rcf *= av1_resize_rate_factor(cpi);
|
||||
rcf *= resize_rate_factor(cpi, width, height);
|
||||
return fclamp(rcf, MIN_BPB_FACTOR, MAX_BPB_FACTOR);
|
||||
}
|
||||
|
||||
static void set_rate_correction_factor(AV1_COMP *cpi, double factor) {
|
||||
static void set_rate_correction_factor(AV1_COMP *cpi, double factor, int width,
|
||||
int height) {
|
||||
RATE_CONTROL *const rc = &cpi->rc;
|
||||
|
||||
// Normalize RCF to account for the size-dependent scaling factor.
|
||||
factor /= av1_resize_rate_factor(cpi);
|
||||
factor /= resize_rate_factor(cpi, width, height);
|
||||
|
||||
factor = fclamp(factor, MIN_BPB_FACTOR, MAX_BPB_FACTOR);
|
||||
|
||||
|
|
@ -417,11 +422,14 @@ static void set_rate_correction_factor(AV1_COMP *cpi, double factor) {
|
|||
}
|
||||
}
|
||||
|
||||
void av1_rc_update_rate_correction_factors(AV1_COMP *cpi) {
|
||||
void av1_rc_update_rate_correction_factors(AV1_COMP *cpi, int width,
|
||||
int height) {
|
||||
const AV1_COMMON *const cm = &cpi->common;
|
||||
int correction_factor = 100;
|
||||
double rate_correction_factor = get_rate_correction_factor(cpi);
|
||||
double rate_correction_factor =
|
||||
get_rate_correction_factor(cpi, width, height);
|
||||
double adjustment_limit;
|
||||
const int MBs = av1_get_MBs(width, height);
|
||||
|
||||
int projected_size_based_on_q = 0;
|
||||
|
||||
|
|
@ -439,7 +447,7 @@ void av1_rc_update_rate_correction_factors(AV1_COMP *cpi) {
|
|||
av1_cyclic_refresh_estimate_bits_at_q(cpi, rate_correction_factor);
|
||||
} else {
|
||||
projected_size_based_on_q =
|
||||
av1_estimate_bits_at_q(cpi->common.frame_type, cm->base_qindex, cm->MBs,
|
||||
av1_estimate_bits_at_q(cpi->common.frame_type, cm->base_qindex, MBs,
|
||||
rate_correction_factor, cm->bit_depth);
|
||||
}
|
||||
// Work out a size correction factor.
|
||||
|
|
@ -485,21 +493,24 @@ void av1_rc_update_rate_correction_factors(AV1_COMP *cpi) {
|
|||
rate_correction_factor = MIN_BPB_FACTOR;
|
||||
}
|
||||
|
||||
set_rate_correction_factor(cpi, rate_correction_factor);
|
||||
set_rate_correction_factor(cpi, rate_correction_factor, width, height);
|
||||
}
|
||||
|
||||
int av1_rc_regulate_q(const AV1_COMP *cpi, int target_bits_per_frame,
|
||||
int active_best_quality, int active_worst_quality) {
|
||||
int active_best_quality, int active_worst_quality,
|
||||
int width, int height) {
|
||||
const AV1_COMMON *const cm = &cpi->common;
|
||||
int q = active_worst_quality;
|
||||
int last_error = INT_MAX;
|
||||
int i, target_bits_per_mb, bits_per_mb_at_this_q;
|
||||
const double correction_factor = get_rate_correction_factor(cpi);
|
||||
const int MBs = av1_get_MBs(width, height);
|
||||
const double correction_factor =
|
||||
get_rate_correction_factor(cpi, width, height);
|
||||
|
||||
// Calculate required scaling factor based on target frame size and size of
|
||||
// frame produced using previous Q.
|
||||
target_bits_per_mb =
|
||||
(int)((uint64_t)target_bits_per_frame << BPER_MB_NORMBITS) / cm->MBs;
|
||||
(int)((uint64_t)(target_bits_per_frame) << BPER_MB_NORMBITS) / MBs;
|
||||
|
||||
i = active_best_quality;
|
||||
|
||||
|
|
@ -579,8 +590,11 @@ static int calc_active_worst_quality_one_pass_vbr(const AV1_COMP *cpi) {
|
|||
active_worst_quality =
|
||||
curr_frame == 0 ? rc->worst_quality : rc->last_q[KEY_FRAME] * 2;
|
||||
} else {
|
||||
if (!rc->is_src_frame_alt_ref &&
|
||||
(cpi->refresh_golden_frame || cpi->refresh_alt_ref_frame)) {
|
||||
if (!rc->is_src_frame_alt_ref && (cpi->refresh_golden_frame ||
|
||||
#if CONFIG_EXT_REFS
|
||||
cpi->refresh_alt2_ref_frame ||
|
||||
#endif // CONFIG_EXT_REFS
|
||||
cpi->refresh_alt_ref_frame)) {
|
||||
active_worst_quality = curr_frame == 1 ? rc->last_q[KEY_FRAME] * 5 / 4
|
||||
: rc->last_q[INTER_FRAME];
|
||||
} else {
|
||||
|
|
@ -647,8 +661,8 @@ static int calc_active_worst_quality_one_pass_cbr(const AV1_COMP *cpi) {
|
|||
return active_worst_quality;
|
||||
}
|
||||
|
||||
static int rc_pick_q_and_bounds_one_pass_cbr(const AV1_COMP *cpi,
|
||||
int *bottom_index,
|
||||
static int rc_pick_q_and_bounds_one_pass_cbr(const AV1_COMP *cpi, int width,
|
||||
int height, int *bottom_index,
|
||||
int *top_index) {
|
||||
const AV1_COMMON *const cm = &cpi->common;
|
||||
const RATE_CONTROL *const rc = &cpi->rc;
|
||||
|
|
@ -678,7 +692,7 @@ static int rc_pick_q_and_bounds_one_pass_cbr(const AV1_COMP *cpi,
|
|||
rc, rc->avg_frame_qindex[KEY_FRAME], cm->bit_depth);
|
||||
|
||||
// Allow somewhat lower kf minq with small image formats.
|
||||
if ((cm->width * cm->height) <= (352 * 288)) {
|
||||
if ((width * height) <= (352 * 288)) {
|
||||
q_adj_factor -= 0.25;
|
||||
}
|
||||
|
||||
|
|
@ -740,7 +754,7 @@ static int rc_pick_q_and_bounds_one_pass_cbr(const AV1_COMP *cpi,
|
|||
q = rc->last_boosted_qindex;
|
||||
} else {
|
||||
q = av1_rc_regulate_q(cpi, rc->this_frame_target, active_best_quality,
|
||||
active_worst_quality);
|
||||
active_worst_quality, width, height);
|
||||
if (q > *top_index) {
|
||||
// Special case when we are targeting the max allowed rate
|
||||
if (rc->this_frame_target >= rc->max_frame_bandwidth)
|
||||
|
|
@ -770,8 +784,8 @@ static int get_active_cq_level(const RATE_CONTROL *rc,
|
|||
return active_cq_level;
|
||||
}
|
||||
|
||||
static int rc_pick_q_and_bounds_one_pass_vbr(const AV1_COMP *cpi,
|
||||
int *bottom_index,
|
||||
static int rc_pick_q_and_bounds_one_pass_vbr(const AV1_COMP *cpi, int width,
|
||||
int height, int *bottom_index,
|
||||
int *top_index) {
|
||||
const AV1_COMMON *const cm = &cpi->common;
|
||||
const RATE_CONTROL *const rc = &cpi->rc;
|
||||
|
|
@ -804,7 +818,7 @@ static int rc_pick_q_and_bounds_one_pass_vbr(const AV1_COMP *cpi,
|
|||
rc, rc->avg_frame_qindex[KEY_FRAME], cm->bit_depth);
|
||||
|
||||
// Allow somewhat lower kf minq with small image formats.
|
||||
if ((cm->width * cm->height) <= (352 * 288)) {
|
||||
if ((width * height) <= (352 * 288)) {
|
||||
q_adj_factor -= 0.25;
|
||||
}
|
||||
|
||||
|
|
@ -899,7 +913,7 @@ static int rc_pick_q_and_bounds_one_pass_vbr(const AV1_COMP *cpi,
|
|||
q = rc->last_boosted_qindex;
|
||||
} else {
|
||||
q = av1_rc_regulate_q(cpi, rc->this_frame_target, active_best_quality,
|
||||
active_worst_quality);
|
||||
active_worst_quality, width, height);
|
||||
if (q > *top_index) {
|
||||
// Special case when we are targeting the max allowed rate
|
||||
if (rc->this_frame_target >= rc->max_frame_bandwidth)
|
||||
|
|
@ -945,7 +959,8 @@ int av1_frame_type_qdelta(const AV1_COMP *cpi, int rf_level, int q) {
|
|||
}
|
||||
|
||||
#define STATIC_MOTION_THRESH 95
|
||||
static int rc_pick_q_and_bounds_two_pass(const AV1_COMP *cpi, int *bottom_index,
|
||||
static int rc_pick_q_and_bounds_two_pass(const AV1_COMP *cpi, int width,
|
||||
int height, int *bottom_index,
|
||||
int *top_index) {
|
||||
const AV1_COMMON *const cm = &cpi->common;
|
||||
const RATE_CONTROL *const rc = &cpi->rc;
|
||||
|
|
@ -992,7 +1007,7 @@ static int rc_pick_q_and_bounds_two_pass(const AV1_COMP *cpi, int *bottom_index,
|
|||
get_kf_active_quality(rc, active_worst_quality, cm->bit_depth);
|
||||
|
||||
// Allow somewhat lower kf minq with small image formats.
|
||||
if ((cm->width * cm->height) <= (352 * 288)) {
|
||||
if ((width * height) <= (352 * 288)) {
|
||||
q_adj_factor -= 0.25;
|
||||
}
|
||||
|
||||
|
|
@ -1005,8 +1020,11 @@ static int rc_pick_q_and_bounds_two_pass(const AV1_COMP *cpi, int *bottom_index,
|
|||
active_best_quality +=
|
||||
av1_compute_qdelta(rc, q_val, q_val * q_adj_factor, cm->bit_depth);
|
||||
}
|
||||
} else if (!rc->is_src_frame_alt_ref &&
|
||||
(cpi->refresh_golden_frame || cpi->refresh_alt_ref_frame)) {
|
||||
} else if (!rc->is_src_frame_alt_ref && (cpi->refresh_golden_frame ||
|
||||
#if CONFIG_EXT_REFS
|
||||
cpi->refresh_alt2_ref_frame ||
|
||||
#endif // CONFIG_EXT_REFS
|
||||
cpi->refresh_alt_ref_frame)) {
|
||||
// Use the lower of active_worst_quality and recent
|
||||
// average Q as basis for GF/ARF best Q limit unless last frame was
|
||||
// a key frame.
|
||||
|
|
@ -1026,7 +1044,11 @@ static int rc_pick_q_and_bounds_two_pass(const AV1_COMP *cpi, int *bottom_index,
|
|||
active_best_quality = active_best_quality * 15 / 16;
|
||||
|
||||
} else if (oxcf->rc_mode == AOM_Q) {
|
||||
#if CONFIG_EXT_REFS
|
||||
if (!cpi->refresh_alt_ref_frame && !cpi->refresh_alt2_ref_frame) {
|
||||
#else
|
||||
if (!cpi->refresh_alt_ref_frame) {
|
||||
#endif // CONFIG_EXT_REFS
|
||||
active_best_quality = cq_level;
|
||||
} else {
|
||||
active_best_quality = get_gf_active_quality(rc, q, cm->bit_depth);
|
||||
|
|
@ -1058,8 +1080,11 @@ static int rc_pick_q_and_bounds_two_pass(const AV1_COMP *cpi, int *bottom_index,
|
|||
if ((cpi->oxcf.rc_mode != AOM_Q) &&
|
||||
(cpi->twopass.gf_zeromotion_pct < VLOW_MOTION_THRESHOLD)) {
|
||||
if (frame_is_intra_only(cm) ||
|
||||
(!rc->is_src_frame_alt_ref &&
|
||||
(cpi->refresh_golden_frame || cpi->refresh_alt_ref_frame))) {
|
||||
(!rc->is_src_frame_alt_ref && (cpi->refresh_golden_frame ||
|
||||
#if CONFIG_EXT_REFS
|
||||
cpi->refresh_alt2_ref_frame ||
|
||||
#endif // CONFIG_EXT_REFS
|
||||
cpi->refresh_alt_ref_frame))) {
|
||||
active_best_quality -=
|
||||
(cpi->twopass.extend_minq + cpi->twopass.extend_minq_fast);
|
||||
active_worst_quality += (cpi->twopass.extend_maxq / 2);
|
||||
|
|
@ -1105,7 +1130,7 @@ static int rc_pick_q_and_bounds_two_pass(const AV1_COMP *cpi, int *bottom_index,
|
|||
}
|
||||
} else {
|
||||
q = av1_rc_regulate_q(cpi, rc->this_frame_target, active_best_quality,
|
||||
active_worst_quality);
|
||||
active_worst_quality, width, height);
|
||||
if (q > active_worst_quality) {
|
||||
// Special case when we are targeting the max allowed rate.
|
||||
if (rc->this_frame_target >= rc->max_frame_bandwidth)
|
||||
|
|
@ -1126,16 +1151,19 @@ static int rc_pick_q_and_bounds_two_pass(const AV1_COMP *cpi, int *bottom_index,
|
|||
return q;
|
||||
}
|
||||
|
||||
int av1_rc_pick_q_and_bounds(const AV1_COMP *cpi, int *bottom_index,
|
||||
int *top_index) {
|
||||
int av1_rc_pick_q_and_bounds(const AV1_COMP *cpi, int width, int height,
|
||||
int *bottom_index, int *top_index) {
|
||||
int q;
|
||||
if (cpi->oxcf.pass == 0) {
|
||||
if (cpi->oxcf.rc_mode == AOM_CBR)
|
||||
q = rc_pick_q_and_bounds_one_pass_cbr(cpi, bottom_index, top_index);
|
||||
q = rc_pick_q_and_bounds_one_pass_cbr(cpi, width, height, bottom_index,
|
||||
top_index);
|
||||
else
|
||||
q = rc_pick_q_and_bounds_one_pass_vbr(cpi, bottom_index, top_index);
|
||||
q = rc_pick_q_and_bounds_one_pass_vbr(cpi, width, height, bottom_index,
|
||||
top_index);
|
||||
} else {
|
||||
q = rc_pick_q_and_bounds_two_pass(cpi, bottom_index, top_index);
|
||||
q = rc_pick_q_and_bounds_two_pass(cpi, width, height, bottom_index,
|
||||
top_index);
|
||||
}
|
||||
|
||||
return q;
|
||||
|
|
@ -1157,7 +1185,8 @@ void av1_rc_compute_frame_size_bounds(const AV1_COMP *cpi, int frame_target,
|
|||
}
|
||||
}
|
||||
|
||||
void av1_rc_set_frame_target(AV1_COMP *cpi, int target) {
|
||||
static void rc_set_frame_target(AV1_COMP *cpi, int target, int width,
|
||||
int height) {
|
||||
const AV1_COMMON *const cm = &cpi->common;
|
||||
RATE_CONTROL *const rc = &cpi->rc;
|
||||
|
||||
|
|
@ -1166,11 +1195,11 @@ void av1_rc_set_frame_target(AV1_COMP *cpi, int target) {
|
|||
// Modify frame size target when down-scaled.
|
||||
if (!av1_frame_unscaled(cm))
|
||||
rc->this_frame_target =
|
||||
(int)(rc->this_frame_target * av1_resize_rate_factor(cpi));
|
||||
(int)(rc->this_frame_target * resize_rate_factor(cpi, width, height));
|
||||
|
||||
// Target rate per SB64 (including partial SB64s.
|
||||
rc->sb64_target_rate = (int)((int64_t)rc->this_frame_target * 64 * 64) /
|
||||
(cm->width * cm->height);
|
||||
rc->sb64_target_rate =
|
||||
(int)((int64_t)rc->this_frame_target * 64 * 64) / (width * height);
|
||||
}
|
||||
|
||||
static void update_alt_ref_frame_stats(AV1_COMP *cpi) {
|
||||
|
|
@ -1194,7 +1223,7 @@ static void update_golden_frame_stats(AV1_COMP *cpi) {
|
|||
// only the virtual indices for the reference frame will be
|
||||
// updated and cpi->refresh_golden_frame will still be zero.
|
||||
if (cpi->refresh_golden_frame || rc->is_src_frame_alt_ref) {
|
||||
#else
|
||||
#else // !CONFIG_EXT_REFS
|
||||
// Update the Golden frame usage counts.
|
||||
if (cpi->refresh_golden_frame) {
|
||||
#endif // CONFIG_EXT_REFS
|
||||
|
|
@ -1219,7 +1248,11 @@ static void update_golden_frame_stats(AV1_COMP *cpi) {
|
|||
// Decrement count down till next gf
|
||||
if (rc->frames_till_gf_update_due > 0) rc->frames_till_gf_update_due--;
|
||||
|
||||
#if CONFIG_EXT_REFS
|
||||
} else if (!cpi->refresh_alt_ref_frame && !cpi->refresh_alt2_ref_frame) {
|
||||
#else
|
||||
} else if (!cpi->refresh_alt_ref_frame) {
|
||||
#endif // CONFIG_EXT_REFS
|
||||
// Decrement count down till next gf
|
||||
if (rc->frames_till_gf_update_due > 0) rc->frames_till_gf_update_due--;
|
||||
|
||||
|
|
@ -1240,7 +1273,7 @@ void av1_rc_postencode_update(AV1_COMP *cpi, uint64_t bytes_used) {
|
|||
rc->projected_frame_size = (int)(bytes_used << 3);
|
||||
|
||||
// Post encode loop adjustment of Q prediction.
|
||||
av1_rc_update_rate_correction_factors(cpi);
|
||||
av1_rc_update_rate_correction_factors(cpi, cm->width, cm->height);
|
||||
|
||||
// Keep a record of last Q and ambient average Q.
|
||||
if (cm->frame_type == KEY_FRAME) {
|
||||
|
|
@ -1249,7 +1282,11 @@ void av1_rc_postencode_update(AV1_COMP *cpi, uint64_t bytes_used) {
|
|||
ROUND_POWER_OF_TWO(3 * rc->avg_frame_qindex[KEY_FRAME] + qindex, 2);
|
||||
} else {
|
||||
if (!rc->is_src_frame_alt_ref &&
|
||||
!(cpi->refresh_golden_frame || cpi->refresh_alt_ref_frame)) {
|
||||
!(cpi->refresh_golden_frame ||
|
||||
#if CONFIG_EXT_REFS
|
||||
cpi->refresh_alt2_ref_frame ||
|
||||
#endif // CONFIG_EXT_REFS
|
||||
cpi->refresh_alt_ref_frame)) {
|
||||
rc->last_q[INTER_FRAME] = qindex;
|
||||
rc->avg_frame_qindex[INTER_FRAME] =
|
||||
ROUND_POWER_OF_TWO(3 * rc->avg_frame_qindex[INTER_FRAME] + qindex, 2);
|
||||
|
|
@ -1271,6 +1308,9 @@ void av1_rc_postencode_update(AV1_COMP *cpi, uint64_t bytes_used) {
|
|||
if ((qindex < rc->last_boosted_qindex) || (cm->frame_type == KEY_FRAME) ||
|
||||
(!rc->constrained_gf_group &&
|
||||
(cpi->refresh_alt_ref_frame ||
|
||||
#if CONFIG_EXT_REFS
|
||||
cpi->refresh_alt2_ref_frame ||
|
||||
#endif // CONFIG_EXT_REFS
|
||||
(cpi->refresh_golden_frame && !rc->is_src_frame_alt_ref)))) {
|
||||
rc->last_boosted_qindex = qindex;
|
||||
}
|
||||
|
|
@ -1280,6 +1320,10 @@ void av1_rc_postencode_update(AV1_COMP *cpi, uint64_t bytes_used) {
|
|||
|
||||
// Rolling monitors of whether we are over or underspending used to help
|
||||
// regulate min and Max Q in two pass.
|
||||
if (!av1_frame_unscaled(cm))
|
||||
rc->this_frame_target =
|
||||
(int)(rc->this_frame_target /
|
||||
resize_rate_factor(cpi, cm->width, cm->height));
|
||||
if (cm->frame_type != KEY_FRAME) {
|
||||
rc->rolling_target_bits = ROUND_POWER_OF_TWO(
|
||||
rc->rolling_target_bits * 3 + rc->this_frame_target, 2);
|
||||
|
|
@ -1294,6 +1338,8 @@ void av1_rc_postencode_update(AV1_COMP *cpi, uint64_t bytes_used) {
|
|||
// Actual bits spent
|
||||
rc->total_actual_bits += rc->projected_frame_size;
|
||||
#if CONFIG_EXT_REFS
|
||||
// TODO(zoeliu): To investigate whether we should treat BWDREF_FRAME
|
||||
// differently here for rc->avg_frame_bandwidth.
|
||||
rc->total_target_bits +=
|
||||
(cm->show_frame || rc->is_bwd_ref_frame) ? rc->avg_frame_bandwidth : 0;
|
||||
#else
|
||||
|
|
@ -1313,6 +1359,8 @@ void av1_rc_postencode_update(AV1_COMP *cpi, uint64_t bytes_used) {
|
|||
if (cm->frame_type == KEY_FRAME) rc->frames_since_key = 0;
|
||||
|
||||
#if CONFIG_EXT_REFS
|
||||
// TODO(zoeliu): To investigate whether we should treat BWDREF_FRAME
|
||||
// differently here for rc->avg_frame_bandwidth.
|
||||
if (cm->show_frame || rc->is_bwd_ref_frame) {
|
||||
#else
|
||||
if (cm->show_frame) {
|
||||
|
|
@ -1320,6 +1368,12 @@ void av1_rc_postencode_update(AV1_COMP *cpi, uint64_t bytes_used) {
|
|||
rc->frames_since_key++;
|
||||
rc->frames_to_key--;
|
||||
}
|
||||
// if (cm->current_video_frame == 1 && cm->show_frame)
|
||||
/*
|
||||
rc->this_frame_target =
|
||||
(int)(rc->this_frame_target / resize_rate_factor(cpi, cm->width,
|
||||
cm->height));
|
||||
*/
|
||||
}
|
||||
|
||||
void av1_rc_postencode_update_drop_frame(AV1_COMP *cpi) {
|
||||
|
|
@ -1394,7 +1448,7 @@ void av1_rc_get_one_pass_vbr_params(AV1_COMP *cpi) {
|
|||
target = calc_iframe_target_size_one_pass_vbr(cpi);
|
||||
else
|
||||
target = calc_pframe_target_size_one_pass_vbr(cpi);
|
||||
av1_rc_set_frame_target(cpi, target);
|
||||
rc_set_frame_target(cpi, target, cm->width, cm->height);
|
||||
}
|
||||
|
||||
static int calc_pframe_target_size_one_pass_cbr(const AV1_COMP *cpi) {
|
||||
|
|
@ -1496,7 +1550,7 @@ void av1_rc_get_one_pass_cbr_params(AV1_COMP *cpi) {
|
|||
else
|
||||
target = calc_pframe_target_size_one_pass_cbr(cpi);
|
||||
|
||||
av1_rc_set_frame_target(cpi, target);
|
||||
rc_set_frame_target(cpi, target, cm->width, cm->height);
|
||||
// TODO(afergs): Decide whether to scale up, down, or not at all
|
||||
}
|
||||
|
||||
|
|
@ -1581,11 +1635,11 @@ void av1_rc_set_gf_interval_range(const AV1_COMP *const cpi,
|
|||
}
|
||||
}
|
||||
|
||||
void av1_rc_update_framerate(AV1_COMP *cpi) {
|
||||
const AV1_COMMON *const cm = &cpi->common;
|
||||
void av1_rc_update_framerate(AV1_COMP *cpi, int width, int height) {
|
||||
const AV1EncoderConfig *const oxcf = &cpi->oxcf;
|
||||
RATE_CONTROL *const rc = &cpi->rc;
|
||||
int vbr_max_bits;
|
||||
const int MBs = av1_get_MBs(width, height);
|
||||
|
||||
rc->avg_frame_bandwidth = (int)(oxcf->target_bandwidth / cpi->framerate);
|
||||
rc->min_frame_bandwidth =
|
||||
|
|
@ -1605,7 +1659,7 @@ void av1_rc_update_framerate(AV1_COMP *cpi) {
|
|||
(int)(((int64_t)rc->avg_frame_bandwidth * oxcf->two_pass_vbrmax_section) /
|
||||
100);
|
||||
rc->max_frame_bandwidth =
|
||||
AOMMAX(AOMMAX((cm->MBs * MAX_MB_RATE), MAXRATE_1080P), vbr_max_bits);
|
||||
AOMMAX(AOMMAX((MBs * MAX_MB_RATE), MAXRATE_1080P), vbr_max_bits);
|
||||
|
||||
av1_rc_set_gf_interval_range(cpi, rc);
|
||||
}
|
||||
|
|
@ -1654,73 +1708,12 @@ static void vbr_rate_correction(AV1_COMP *cpi, int *this_frame_target) {
|
|||
}
|
||||
}
|
||||
|
||||
void av1_set_target_rate(AV1_COMP *cpi) {
|
||||
void av1_set_target_rate(AV1_COMP *cpi, int width, int height) {
|
||||
RATE_CONTROL *const rc = &cpi->rc;
|
||||
int target_rate = rc->base_frame_target;
|
||||
|
||||
// Correction to rate target based on prior over or under shoot.
|
||||
if (cpi->oxcf.rc_mode == AOM_VBR || cpi->oxcf.rc_mode == AOM_CQ)
|
||||
vbr_rate_correction(cpi, &target_rate);
|
||||
av1_rc_set_frame_target(cpi, target_rate);
|
||||
rc_set_frame_target(cpi, target_rate, width, height);
|
||||
}
|
||||
|
||||
static unsigned int lcg_rand16(unsigned int *state) {
|
||||
*state = (unsigned int)(*state * 1103515245ULL + 12345);
|
||||
return *state / 65536 % 32768;
|
||||
}
|
||||
|
||||
uint8_t av1_calculate_next_resize_scale(const AV1_COMP *cpi) {
|
||||
static unsigned int seed = 56789;
|
||||
const AV1EncoderConfig *oxcf = &cpi->oxcf;
|
||||
if (oxcf->pass == 1) return SCALE_DENOMINATOR;
|
||||
uint8_t new_num = SCALE_DENOMINATOR;
|
||||
|
||||
switch (oxcf->resize_mode) {
|
||||
case RESIZE_NONE: new_num = SCALE_DENOMINATOR; break;
|
||||
case RESIZE_FIXED:
|
||||
if (cpi->common.frame_type == KEY_FRAME)
|
||||
new_num = oxcf->resize_kf_scale_numerator;
|
||||
else
|
||||
new_num = oxcf->resize_scale_numerator;
|
||||
break;
|
||||
case RESIZE_DYNAMIC:
|
||||
// RESIZE_DYNAMIC: Just random for now.
|
||||
new_num = lcg_rand16(&seed) % 4 + 13;
|
||||
break;
|
||||
default: assert(0);
|
||||
}
|
||||
return new_num;
|
||||
}
|
||||
|
||||
#if CONFIG_FRAME_SUPERRES
|
||||
// TODO(afergs): Rename av1_rc_update_superres_scale(...)?
|
||||
uint8_t av1_calculate_next_superres_scale(const AV1_COMP *cpi, int width,
|
||||
int height) {
|
||||
static unsigned int seed = 34567;
|
||||
const AV1EncoderConfig *oxcf = &cpi->oxcf;
|
||||
if (oxcf->pass == 1) return SCALE_DENOMINATOR;
|
||||
uint8_t new_num = SCALE_DENOMINATOR;
|
||||
|
||||
switch (oxcf->superres_mode) {
|
||||
case SUPERRES_NONE: new_num = SCALE_DENOMINATOR; break;
|
||||
case SUPERRES_FIXED:
|
||||
if (cpi->common.frame_type == KEY_FRAME)
|
||||
new_num = oxcf->superres_kf_scale_numerator;
|
||||
else
|
||||
new_num = oxcf->superres_scale_numerator;
|
||||
break;
|
||||
case SUPERRES_DYNAMIC:
|
||||
// SUPERRES_DYNAMIC: Just random for now.
|
||||
new_num = lcg_rand16(&seed) % 9 + 8;
|
||||
break;
|
||||
default: assert(0);
|
||||
}
|
||||
|
||||
// Make sure overall reduction is no more than 1/2 of the source size.
|
||||
av1_calculate_scaled_size(&width, &height, new_num);
|
||||
if (width * 2 < oxcf->width || height * 2 < oxcf->height)
|
||||
new_num = SCALE_DENOMINATOR;
|
||||
|
||||
return new_num;
|
||||
}
|
||||
#endif // CONFIG_FRAME_SUPERRES
|
||||
|
|
|
|||
31
third_party/aom/av1/encoder/ratectrl.h
vendored
31
third_party/aom/av1/encoder/ratectrl.h
vendored
|
|
@ -49,6 +49,14 @@ typedef enum {
|
|||
} RATE_FACTOR_LEVEL;
|
||||
#endif // CONFIG_EXT_REFS
|
||||
|
||||
typedef struct {
|
||||
int resize_width;
|
||||
int resize_height;
|
||||
#if CONFIG_FRAME_SUPERRES
|
||||
uint8_t superres_denom;
|
||||
#endif // CONFIG_FRAME_SUPERRES
|
||||
} size_params_type;
|
||||
|
||||
typedef struct {
|
||||
// Rate targetting variables
|
||||
int base_frame_target; // A baseline frame target before adjustment
|
||||
|
|
@ -189,10 +197,6 @@ int av1_rc_get_default_max_gf_interval(double framerate, int min_frame_rate);
|
|||
void av1_rc_get_one_pass_vbr_params(struct AV1_COMP *cpi);
|
||||
void av1_rc_get_one_pass_cbr_params(struct AV1_COMP *cpi);
|
||||
|
||||
// How many times less pixels there are to encode given the current scaling.
|
||||
// Temporary replacement for rcf_mult and rate_thresh_mult.
|
||||
double av1_resize_rate_factor(const struct AV1_COMP *cpi);
|
||||
|
||||
// Post encode update of the rate control parameters based
|
||||
// on bytes used
|
||||
void av1_rc_postencode_update(struct AV1_COMP *cpi, uint64_t bytes_used);
|
||||
|
|
@ -201,7 +205,8 @@ void av1_rc_postencode_update_drop_frame(struct AV1_COMP *cpi);
|
|||
|
||||
// Updates rate correction factors
|
||||
// Changes only the rate correction factors in the rate control structure.
|
||||
void av1_rc_update_rate_correction_factors(struct AV1_COMP *cpi);
|
||||
void av1_rc_update_rate_correction_factors(struct AV1_COMP *cpi, int width,
|
||||
int height);
|
||||
|
||||
// Decide if we should drop this frame: For 1-pass CBR.
|
||||
// Changes only the decimation count in the rate control structure
|
||||
|
|
@ -214,12 +219,13 @@ void av1_rc_compute_frame_size_bounds(const struct AV1_COMP *cpi,
|
|||
int *frame_over_shoot_limit);
|
||||
|
||||
// Picks q and q bounds given the target for bits
|
||||
int av1_rc_pick_q_and_bounds(const struct AV1_COMP *cpi, int *bottom_index,
|
||||
int *top_index);
|
||||
int av1_rc_pick_q_and_bounds(const struct AV1_COMP *cpi, int width, int height,
|
||||
int *bottom_index, int *top_index);
|
||||
|
||||
// Estimates q to achieve a target bits per frame
|
||||
int av1_rc_regulate_q(const struct AV1_COMP *cpi, int target_bits_per_frame,
|
||||
int active_best_quality, int active_worst_quality);
|
||||
int active_best_quality, int active_worst_quality,
|
||||
int width, int height);
|
||||
|
||||
// Estimates bits per mb for a given qindex and correction factor.
|
||||
int av1_rc_bits_per_mb(FRAME_TYPE frame_type, int qindex,
|
||||
|
|
@ -247,20 +253,15 @@ int av1_compute_qdelta_by_rate(const RATE_CONTROL *rc, FRAME_TYPE frame_type,
|
|||
|
||||
int av1_frame_type_qdelta(const struct AV1_COMP *cpi, int rf_level, int q);
|
||||
|
||||
void av1_rc_update_framerate(struct AV1_COMP *cpi);
|
||||
void av1_rc_update_framerate(struct AV1_COMP *cpi, int width, int height);
|
||||
|
||||
void av1_rc_set_gf_interval_range(const struct AV1_COMP *const cpi,
|
||||
RATE_CONTROL *const rc);
|
||||
|
||||
void av1_set_target_rate(struct AV1_COMP *cpi);
|
||||
void av1_set_target_rate(struct AV1_COMP *cpi, int width, int height);
|
||||
|
||||
int av1_resize_one_pass_cbr(struct AV1_COMP *cpi);
|
||||
|
||||
uint8_t av1_calculate_next_resize_scale(const struct AV1_COMP *cpi);
|
||||
#if CONFIG_FRAME_SUPERRES
|
||||
uint8_t av1_calculate_next_superres_scale(const struct AV1_COMP *cpi, int width,
|
||||
int height);
|
||||
#endif // CONFIG_FRAME_SUPERRES
|
||||
#ifdef __cplusplus
|
||||
} // extern "C"
|
||||
#endif
|
||||
|
|
|
|||
779
third_party/aom/av1/encoder/rd.c
vendored
779
third_party/aom/av1/encoder/rd.c
vendored
File diff suppressed because it is too large
Load diff
190
third_party/aom/av1/encoder/rd.h
vendored
190
third_party/aom/av1/encoder/rd.h
vendored
|
|
@ -43,14 +43,6 @@ extern "C" {
|
|||
#define MV_COST_WEIGHT 108
|
||||
#define MV_COST_WEIGHT_SUB 120
|
||||
|
||||
#define INVALID_MV 0x80008000
|
||||
|
||||
#if CONFIG_EXT_REFS
|
||||
#define MAX_REFS 15
|
||||
#else
|
||||
#define MAX_REFS 6
|
||||
#endif // CONFIG_EXT_REFS
|
||||
|
||||
#define RD_THRESH_MAX_FACT 64
|
||||
#define RD_THRESH_INC 1
|
||||
|
||||
|
|
@ -62,6 +54,7 @@ typedef enum {
|
|||
THR_NEARESTL2,
|
||||
THR_NEARESTL3,
|
||||
THR_NEARESTB,
|
||||
THR_NEARESTA2,
|
||||
#endif // CONFIG_EXT_REFS
|
||||
THR_NEARESTA,
|
||||
THR_NEARESTG,
|
||||
|
|
@ -73,6 +66,7 @@ typedef enum {
|
|||
THR_NEWL2,
|
||||
THR_NEWL3,
|
||||
THR_NEWB,
|
||||
THR_NEWA2,
|
||||
#endif // CONFIG_EXT_REFS
|
||||
THR_NEWA,
|
||||
THR_NEWG,
|
||||
|
|
@ -82,6 +76,7 @@ typedef enum {
|
|||
THR_NEARL2,
|
||||
THR_NEARL3,
|
||||
THR_NEARB,
|
||||
THR_NEARA2,
|
||||
#endif // CONFIG_EXT_REFS
|
||||
THR_NEARA,
|
||||
THR_NEARG,
|
||||
|
|
@ -91,11 +86,10 @@ typedef enum {
|
|||
THR_ZEROL2,
|
||||
THR_ZEROL3,
|
||||
THR_ZEROB,
|
||||
THR_ZEROA2,
|
||||
#endif // CONFIG_EXT_REFS
|
||||
THR_ZEROG,
|
||||
THR_ZEROA,
|
||||
|
||||
#if CONFIG_EXT_INTER
|
||||
THR_ZEROG,
|
||||
|
||||
#if CONFIG_COMPOUND_SINGLEREF
|
||||
THR_SR_NEAREST_NEARMV,
|
||||
|
|
@ -156,6 +150,10 @@ typedef enum {
|
|||
THR_COMP_NEAREST_NEARESTL2B,
|
||||
THR_COMP_NEAREST_NEARESTL3B,
|
||||
THR_COMP_NEAREST_NEARESTGB,
|
||||
THR_COMP_NEAREST_NEARESTLA2,
|
||||
THR_COMP_NEAREST_NEARESTL2A2,
|
||||
THR_COMP_NEAREST_NEARESTL3A2,
|
||||
THR_COMP_NEAREST_NEARESTGA2,
|
||||
#if CONFIG_EXT_COMP_REFS
|
||||
THR_COMP_NEAREST_NEARESTLL2,
|
||||
THR_COMP_NEAREST_NEARESTLL3,
|
||||
|
|
@ -164,40 +162,13 @@ typedef enum {
|
|||
#endif // CONFIG_EXT_COMP_REFS
|
||||
#endif // CONFIG_EXT_REFS
|
||||
|
||||
#else // CONFIG_EXT_INTER
|
||||
|
||||
THR_COMP_NEARESTLA,
|
||||
#if CONFIG_EXT_REFS
|
||||
THR_COMP_NEARESTL2A,
|
||||
THR_COMP_NEARESTL3A,
|
||||
#endif // CONFIG_EXT_REFS
|
||||
THR_COMP_NEARESTGA,
|
||||
#if CONFIG_EXT_REFS
|
||||
THR_COMP_NEARESTLB,
|
||||
THR_COMP_NEARESTL2B,
|
||||
THR_COMP_NEARESTL3B,
|
||||
THR_COMP_NEARESTGB,
|
||||
#if CONFIG_EXT_COMP_REFS
|
||||
THR_COMP_NEARESTLL2,
|
||||
THR_COMP_NEARESTLL3,
|
||||
THR_COMP_NEARESTLG,
|
||||
THR_COMP_NEARESTBA,
|
||||
#endif // CONFIG_EXT_COMP_REFS
|
||||
#endif // CONFIG_EXT_REFS
|
||||
|
||||
#endif // CONFIG_EXT_INTER
|
||||
|
||||
THR_TM,
|
||||
|
||||
#if CONFIG_ALT_INTRA
|
||||
THR_SMOOTH,
|
||||
#if CONFIG_SMOOTH_HV
|
||||
THR_SMOOTH_V,
|
||||
THR_SMOOTH_H,
|
||||
#endif // CONFIG_SMOOTH_HV
|
||||
#endif // CONFIG_ALT_INTRA
|
||||
|
||||
#if CONFIG_EXT_INTER
|
||||
|
||||
THR_COMP_NEAR_NEARLA,
|
||||
THR_COMP_NEW_NEARESTLA,
|
||||
|
|
@ -266,6 +237,38 @@ typedef enum {
|
|||
THR_COMP_NEW_NEWGB,
|
||||
THR_COMP_ZERO_ZEROGB,
|
||||
|
||||
THR_COMP_NEAR_NEARLA2,
|
||||
THR_COMP_NEW_NEARESTLA2,
|
||||
THR_COMP_NEAREST_NEWLA2,
|
||||
THR_COMP_NEW_NEARLA2,
|
||||
THR_COMP_NEAR_NEWLA2,
|
||||
THR_COMP_NEW_NEWLA2,
|
||||
THR_COMP_ZERO_ZEROLA2,
|
||||
|
||||
THR_COMP_NEAR_NEARL2A2,
|
||||
THR_COMP_NEW_NEARESTL2A2,
|
||||
THR_COMP_NEAREST_NEWL2A2,
|
||||
THR_COMP_NEW_NEARL2A2,
|
||||
THR_COMP_NEAR_NEWL2A2,
|
||||
THR_COMP_NEW_NEWL2A2,
|
||||
THR_COMP_ZERO_ZEROL2A2,
|
||||
|
||||
THR_COMP_NEAR_NEARL3A2,
|
||||
THR_COMP_NEW_NEARESTL3A2,
|
||||
THR_COMP_NEAREST_NEWL3A2,
|
||||
THR_COMP_NEW_NEARL3A2,
|
||||
THR_COMP_NEAR_NEWL3A2,
|
||||
THR_COMP_NEW_NEWL3A2,
|
||||
THR_COMP_ZERO_ZEROL3A2,
|
||||
|
||||
THR_COMP_NEAR_NEARGA2,
|
||||
THR_COMP_NEW_NEARESTGA2,
|
||||
THR_COMP_NEAREST_NEWGA2,
|
||||
THR_COMP_NEW_NEARGA2,
|
||||
THR_COMP_NEAR_NEWGA2,
|
||||
THR_COMP_NEW_NEWGA2,
|
||||
THR_COMP_ZERO_ZEROGA2,
|
||||
|
||||
#if CONFIG_EXT_COMP_REFS
|
||||
THR_COMP_NEAR_NEARLL2,
|
||||
THR_COMP_NEW_NEARESTLL2,
|
||||
|
|
@ -301,64 +304,6 @@ typedef enum {
|
|||
#endif // CONFIG_EXT_COMP_REFS
|
||||
#endif // CONFIG_EXT_REFS
|
||||
|
||||
#else // CONFIG_EXT_INTER
|
||||
|
||||
THR_COMP_NEARLA,
|
||||
THR_COMP_NEWLA,
|
||||
#if CONFIG_EXT_REFS
|
||||
THR_COMP_NEARL2A,
|
||||
THR_COMP_NEWL2A,
|
||||
THR_COMP_NEARL3A,
|
||||
THR_COMP_NEWL3A,
|
||||
#endif // CONFIG_EXT_REFS
|
||||
THR_COMP_NEARGA,
|
||||
THR_COMP_NEWGA,
|
||||
|
||||
#if CONFIG_EXT_REFS
|
||||
THR_COMP_NEARLB,
|
||||
THR_COMP_NEWLB,
|
||||
THR_COMP_NEARL2B,
|
||||
THR_COMP_NEWL2B,
|
||||
THR_COMP_NEARL3B,
|
||||
THR_COMP_NEWL3B,
|
||||
THR_COMP_NEARGB,
|
||||
THR_COMP_NEWGB,
|
||||
|
||||
#if CONFIG_EXT_COMP_REFS
|
||||
THR_COMP_NEARLL2,
|
||||
THR_COMP_NEWLL2,
|
||||
THR_COMP_NEARLL3,
|
||||
THR_COMP_NEWLL3,
|
||||
THR_COMP_NEARLG,
|
||||
THR_COMP_NEWLG,
|
||||
THR_COMP_NEARBA,
|
||||
THR_COMP_NEWBA,
|
||||
#endif // CONFIG_EXT_COMP_REFS
|
||||
#endif // CONFIG_EXT_REFS
|
||||
|
||||
THR_COMP_ZEROLA,
|
||||
#if CONFIG_EXT_REFS
|
||||
THR_COMP_ZEROL2A,
|
||||
THR_COMP_ZEROL3A,
|
||||
#endif // CONFIG_EXT_REFS
|
||||
THR_COMP_ZEROGA,
|
||||
|
||||
#if CONFIG_EXT_REFS
|
||||
THR_COMP_ZEROLB,
|
||||
THR_COMP_ZEROL2B,
|
||||
THR_COMP_ZEROL3B,
|
||||
THR_COMP_ZEROGB,
|
||||
|
||||
#if CONFIG_EXT_COMP_REFS
|
||||
THR_COMP_ZEROLL2,
|
||||
THR_COMP_ZEROLL3,
|
||||
THR_COMP_ZEROLG,
|
||||
THR_COMP_ZEROBA,
|
||||
#endif // CONFIG_EXT_COMP_REFS
|
||||
#endif // CONFIG_EXT_REFS
|
||||
|
||||
#endif // CONFIG_EXT_INTER
|
||||
|
||||
THR_H_PRED,
|
||||
THR_V_PRED,
|
||||
THR_D135_PRED,
|
||||
|
|
@ -368,7 +313,6 @@ typedef enum {
|
|||
THR_D117_PRED,
|
||||
THR_D45_PRED,
|
||||
|
||||
#if CONFIG_EXT_INTER
|
||||
THR_COMP_INTERINTRA_ZEROL,
|
||||
THR_COMP_INTERINTRA_NEARESTL,
|
||||
THR_COMP_INTERINTRA_NEARL,
|
||||
|
|
@ -396,13 +340,17 @@ typedef enum {
|
|||
THR_COMP_INTERINTRA_NEARESTB,
|
||||
THR_COMP_INTERINTRA_NEARB,
|
||||
THR_COMP_INTERINTRA_NEWB,
|
||||
|
||||
THR_COMP_INTERINTRA_ZEROA2,
|
||||
THR_COMP_INTERINTRA_NEARESTA2,
|
||||
THR_COMP_INTERINTRA_NEARA2,
|
||||
THR_COMP_INTERINTRA_NEWA2,
|
||||
#endif // CONFIG_EXT_REFS
|
||||
|
||||
THR_COMP_INTERINTRA_ZEROA,
|
||||
THR_COMP_INTERINTRA_NEARESTA,
|
||||
THR_COMP_INTERINTRA_NEARA,
|
||||
THR_COMP_INTERINTRA_NEWA,
|
||||
#endif // CONFIG_EXT_INTER
|
||||
MAX_MODES
|
||||
} THR_MODES;
|
||||
|
||||
|
|
@ -412,6 +360,7 @@ typedef enum {
|
|||
THR_LAST2,
|
||||
THR_LAST3,
|
||||
THR_BWDR,
|
||||
THR_ALTR2,
|
||||
#endif // CONFIG_EXT_REFS
|
||||
THR_GOLD,
|
||||
THR_ALTR,
|
||||
|
|
@ -428,9 +377,16 @@ typedef enum {
|
|||
THR_COMP_L2B,
|
||||
THR_COMP_L3B,
|
||||
THR_COMP_GB,
|
||||
|
||||
THR_COMP_LA2,
|
||||
THR_COMP_L2A2,
|
||||
THR_COMP_L3A2,
|
||||
THR_COMP_GA2,
|
||||
#endif // CONFIG_EXT_REFS
|
||||
|
||||
THR_INTRA,
|
||||
|
||||
MAX_REFS
|
||||
} THR_MODES_SUB8X8;
|
||||
|
||||
typedef struct RD_OPT {
|
||||
|
|
@ -458,10 +414,8 @@ static INLINE void av1_init_rd_stats(RD_STATS *rd_stats) {
|
|||
rd_stats->sse = 0;
|
||||
rd_stats->skip = 1;
|
||||
rd_stats->zero_rate = 0;
|
||||
rd_stats->invalid_rate = 0;
|
||||
rd_stats->ref_rdcost = INT64_MAX;
|
||||
#if CONFIG_DIST_8X8 && CONFIG_CB4X4
|
||||
rd_stats->dist_y = 0;
|
||||
#endif
|
||||
#if CONFIG_RD_DEBUG
|
||||
for (plane = 0; plane < MAX_MB_PLANE; ++plane) {
|
||||
rd_stats->txb_coeff_cost[plane] = 0;
|
||||
|
|
@ -487,10 +441,8 @@ static INLINE void av1_invalid_rd_stats(RD_STATS *rd_stats) {
|
|||
rd_stats->sse = INT64_MAX;
|
||||
rd_stats->skip = 0;
|
||||
rd_stats->zero_rate = 0;
|
||||
rd_stats->invalid_rate = 1;
|
||||
rd_stats->ref_rdcost = INT64_MAX;
|
||||
#if CONFIG_DIST_8X8 && CONFIG_CB4X4
|
||||
rd_stats->dist_y = INT64_MAX;
|
||||
#endif
|
||||
#if CONFIG_RD_DEBUG
|
||||
for (plane = 0; plane < MAX_MB_PLANE; ++plane) {
|
||||
rd_stats->txb_coeff_cost[plane] = INT_MAX;
|
||||
|
|
@ -515,9 +467,7 @@ static INLINE void av1_merge_rd_stats(RD_STATS *rd_stats_dst,
|
|||
rd_stats_dst->dist += rd_stats_src->dist;
|
||||
rd_stats_dst->sse += rd_stats_src->sse;
|
||||
rd_stats_dst->skip &= rd_stats_src->skip;
|
||||
#if CONFIG_DIST_8X8 && CONFIG_CB4X4
|
||||
rd_stats_dst->dist_y += rd_stats_src->dist_y;
|
||||
#endif
|
||||
rd_stats_dst->invalid_rate &= rd_stats_src->invalid_rate;
|
||||
#if CONFIG_RD_DEBUG
|
||||
for (plane = 0; plane < MAX_MB_PLANE; ++plane) {
|
||||
rd_stats_dst->txb_coeff_cost[plane] += rd_stats_src->txb_coeff_cost[plane];
|
||||
|
|
@ -539,6 +489,16 @@ static INLINE void av1_merge_rd_stats(RD_STATS *rd_stats_dst,
|
|||
#endif
|
||||
}
|
||||
|
||||
static INLINE int av1_get_coeff_token_cost(int token, int eob_val, int is_first,
|
||||
const int *head_cost_table,
|
||||
const int *tail_cost_table) {
|
||||
if (eob_val == LAST_EOB) return av1_cost_zero(128);
|
||||
const int comb_symb = 2 * AOMMIN(token, TWO_TOKEN) - eob_val + is_first;
|
||||
int cost = head_cost_table[comb_symb];
|
||||
if (token > ONE_TOKEN) cost += tail_cost_table[token - TWO_TOKEN];
|
||||
return cost;
|
||||
}
|
||||
|
||||
struct TileInfo;
|
||||
struct TileDataEnc;
|
||||
struct AV1_COMP;
|
||||
|
|
@ -554,7 +514,8 @@ void av1_initialize_me_consts(const struct AV1_COMP *cpi, MACROBLOCK *x,
|
|||
void av1_model_rd_from_var_lapndz(int64_t var, unsigned int n,
|
||||
unsigned int qstep, int *rate, int64_t *dist);
|
||||
|
||||
int av1_get_switchable_rate(const struct AV1_COMP *cpi, const MACROBLOCKD *xd);
|
||||
int av1_get_switchable_rate(const AV1_COMMON *const cm, MACROBLOCK *x,
|
||||
const MACROBLOCKD *xd);
|
||||
|
||||
int av1_raster_block_offset(BLOCK_SIZE plane_bsize, int raster_block,
|
||||
int stride);
|
||||
|
|
@ -583,9 +544,6 @@ void av1_update_rd_thresh_fact(const AV1_COMMON *const cm,
|
|||
int (*fact)[MAX_MODES], int rd_thresh, int bsize,
|
||||
int best_mode_index);
|
||||
|
||||
void av1_fill_token_costs(av1_coeff_cost *c,
|
||||
av1_coeff_probs_model (*p)[PLANE_TYPES]);
|
||||
|
||||
static INLINE int rd_less_than_thresh(int64_t best_rd, int thresh,
|
||||
int thresh_fact) {
|
||||
return best_rd < ((int64_t)thresh * thresh_fact >> 5) || thresh == INT_MAX;
|
||||
|
|
@ -609,6 +567,16 @@ void av1_setup_pred_block(const MACROBLOCKD *xd,
|
|||
int av1_get_intra_cost_penalty(int qindex, int qdelta,
|
||||
aom_bit_depth_t bit_depth);
|
||||
|
||||
void av1_fill_mode_rates(AV1_COMMON *const cm, MACROBLOCK *x,
|
||||
FRAME_CONTEXT *fc);
|
||||
|
||||
#if CONFIG_LV_MAP
|
||||
void av1_fill_coeff_costs(MACROBLOCK *x, FRAME_CONTEXT *fc);
|
||||
#endif
|
||||
|
||||
void av1_fill_token_costs_from_cdf(av1_coeff_cost *cost,
|
||||
coeff_cdf_model (*cdf)[PLANE_TYPES]);
|
||||
|
||||
#ifdef __cplusplus
|
||||
} // extern "C"
|
||||
#endif
|
||||
|
|
|
|||
4229
third_party/aom/av1/encoder/rdopt.c
vendored
4229
third_party/aom/av1/encoder/rdopt.c
vendored
File diff suppressed because it is too large
Load diff
23
third_party/aom/av1/encoder/rdopt.h
vendored
23
third_party/aom/av1/encoder/rdopt.h
vendored
|
|
@ -57,7 +57,6 @@ typedef enum OUTPUT_STATUS {
|
|||
OUTPUT_HAS_DECODED_PIXELS
|
||||
} OUTPUT_STATUS;
|
||||
|
||||
#if CONFIG_PALETTE || CONFIG_INTRABC
|
||||
// Returns the number of colors in 'src'.
|
||||
int av1_count_colors(const uint8_t *src, int stride, int rows, int cols);
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
|
|
@ -65,7 +64,6 @@ int av1_count_colors(const uint8_t *src, int stride, int rows, int cols);
|
|||
int av1_count_colors_highbd(const uint8_t *src8, int stride, int rows, int cols,
|
||||
int bit_depth);
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
#endif // CONFIG_PALETTE || CONFIG_INTRABC
|
||||
|
||||
void av1_dist_block(const AV1_COMP *cpi, MACROBLOCK *x, int plane,
|
||||
BLOCK_SIZE plane_bsize, int block, int blk_row, int blk_col,
|
||||
|
|
@ -73,7 +71,7 @@ void av1_dist_block(const AV1_COMP *cpi, MACROBLOCK *x, int plane,
|
|||
OUTPUT_STATUS output_status);
|
||||
|
||||
#if CONFIG_DIST_8X8
|
||||
int64_t av1_dist_8x8(const AV1_COMP *const cpi, const MACROBLOCKD *xd,
|
||||
int64_t av1_dist_8x8(const AV1_COMP *const cpi, const MACROBLOCK *x,
|
||||
const uint8_t *src, int src_stride, const uint8_t *dst,
|
||||
int dst_stride, const BLOCK_SIZE tx_bsize, int bsw,
|
||||
int bsh, int visible_w, int visible_h, int qindex);
|
||||
|
|
@ -142,8 +140,21 @@ void av1_txfm_rd_in_plane_supertx(MACROBLOCK *x, const AV1_COMP *cpi, int *rate,
|
|||
} // extern "C"
|
||||
#endif
|
||||
|
||||
int av1_tx_type_cost(const AV1_COMP *cpi, const MACROBLOCKD *xd,
|
||||
BLOCK_SIZE bsize, int plane, TX_SIZE tx_size,
|
||||
TX_TYPE tx_type);
|
||||
int av1_tx_type_cost(const AV1_COMMON *cm, const MACROBLOCK *x,
|
||||
const MACROBLOCKD *xd, BLOCK_SIZE bsize, int plane,
|
||||
TX_SIZE tx_size, TX_TYPE tx_type);
|
||||
|
||||
int64_t get_prediction_rd_cost(const struct AV1_COMP *cpi, struct macroblock *x,
|
||||
int mi_row, int mi_col, int *skip_blk,
|
||||
MB_MODE_INFO *backup_mbmi);
|
||||
|
||||
#if CONFIG_NCOBMC_ADAPT_WEIGHT
|
||||
void av1_check_ncobmc_adapt_weight_rd(const struct AV1_COMP *cpi,
|
||||
struct macroblock *x, int mi_row,
|
||||
int mi_col);
|
||||
int get_ncobmc_mode(const AV1_COMP *const cpi, MACROBLOCK *const x,
|
||||
MACROBLOCKD *xd, int mi_row, int mi_col, int bsize);
|
||||
|
||||
#endif
|
||||
|
||||
#endif // AV1_ENCODER_RDOPT_H_
|
||||
|
|
|
|||
115
third_party/aom/av1/encoder/segmentation.c
vendored
115
third_party/aom/av1/encoder/segmentation.c
vendored
|
|
@ -32,7 +32,7 @@ void av1_disable_segmentation(struct segmentation *seg) {
|
|||
seg->update_data = 0;
|
||||
}
|
||||
|
||||
void av1_set_segment_data(struct segmentation *seg, signed char *feature_data,
|
||||
void av1_set_segment_data(struct segmentation *seg, int8_t *feature_data,
|
||||
unsigned char abs_delta) {
|
||||
seg->abs_delta = abs_delta;
|
||||
|
||||
|
|
@ -167,76 +167,78 @@ static void count_segs_sb(const AV1_COMMON *cm, MACROBLOCKD *xd,
|
|||
const int bs = mi_size_wide[bsize], hbs = bs / 2;
|
||||
#if CONFIG_EXT_PARTITION_TYPES
|
||||
PARTITION_TYPE partition;
|
||||
#if CONFIG_EXT_PARTITION_TYPES_AB
|
||||
const int qbs = bs / 4;
|
||||
#endif // CONFIG_EXT_PARTITION_TYPES_AB
|
||||
#else
|
||||
int bw, bh;
|
||||
#endif // CONFIG_EXT_PARTITION_TYPES
|
||||
|
||||
if (mi_row >= cm->mi_rows || mi_col >= cm->mi_cols) return;
|
||||
|
||||
#define CSEGS(cs_bw, cs_bh, cs_rowoff, cs_coloff) \
|
||||
count_segs(cm, xd, tile, mi + mis * (cs_rowoff) + (cs_coloff), \
|
||||
no_pred_segcounts, temporal_predictor_count, t_unpred_seg_counts, \
|
||||
(cs_bw), (cs_bh), mi_row + (cs_rowoff), mi_col + (cs_coloff));
|
||||
|
||||
#if CONFIG_EXT_PARTITION_TYPES
|
||||
if (bsize == BLOCK_8X8)
|
||||
partition = PARTITION_NONE;
|
||||
else
|
||||
partition = get_partition(cm, mi_row, mi_col, bsize);
|
||||
switch (partition) {
|
||||
case PARTITION_NONE:
|
||||
count_segs(cm, xd, tile, mi, no_pred_segcounts, temporal_predictor_count,
|
||||
t_unpred_seg_counts, bs, bs, mi_row, mi_col);
|
||||
break;
|
||||
case PARTITION_NONE: CSEGS(bs, bs, 0, 0); break;
|
||||
case PARTITION_HORZ:
|
||||
count_segs(cm, xd, tile, mi, no_pred_segcounts, temporal_predictor_count,
|
||||
t_unpred_seg_counts, bs, hbs, mi_row, mi_col);
|
||||
count_segs(cm, xd, tile, mi + hbs * mis, no_pred_segcounts,
|
||||
temporal_predictor_count, t_unpred_seg_counts, bs, hbs,
|
||||
mi_row + hbs, mi_col);
|
||||
CSEGS(bs, hbs, 0, 0);
|
||||
CSEGS(bs, hbs, hbs, 0);
|
||||
break;
|
||||
case PARTITION_VERT:
|
||||
count_segs(cm, xd, tile, mi, no_pred_segcounts, temporal_predictor_count,
|
||||
t_unpred_seg_counts, hbs, bs, mi_row, mi_col);
|
||||
count_segs(cm, xd, tile, mi + hbs, no_pred_segcounts,
|
||||
temporal_predictor_count, t_unpred_seg_counts, hbs, bs, mi_row,
|
||||
mi_col + hbs);
|
||||
CSEGS(hbs, bs, 0, 0);
|
||||
CSEGS(hbs, bs, 0, hbs);
|
||||
break;
|
||||
#if CONFIG_EXT_PARTITION_TYPES_AB
|
||||
case PARTITION_HORZ_A:
|
||||
count_segs(cm, xd, tile, mi, no_pred_segcounts, temporal_predictor_count,
|
||||
t_unpred_seg_counts, hbs, hbs, mi_row, mi_col);
|
||||
count_segs(cm, xd, tile, mi + hbs, no_pred_segcounts,
|
||||
temporal_predictor_count, t_unpred_seg_counts, hbs, hbs,
|
||||
mi_row, mi_col + hbs);
|
||||
count_segs(cm, xd, tile, mi + hbs * mis, no_pred_segcounts,
|
||||
temporal_predictor_count, t_unpred_seg_counts, bs, hbs,
|
||||
mi_row + hbs, mi_col);
|
||||
CSEGS(bs, qbs, 0, 0);
|
||||
CSEGS(bs, qbs, qbs, 0);
|
||||
CSEGS(bs, hbs, hbs, 0);
|
||||
break;
|
||||
case PARTITION_HORZ_B:
|
||||
count_segs(cm, xd, tile, mi, no_pred_segcounts, temporal_predictor_count,
|
||||
t_unpred_seg_counts, bs, hbs, mi_row, mi_col);
|
||||
count_segs(cm, xd, tile, mi + hbs * mis, no_pred_segcounts,
|
||||
temporal_predictor_count, t_unpred_seg_counts, hbs, hbs,
|
||||
mi_row + hbs, mi_col);
|
||||
count_segs(cm, xd, tile, mi + hbs + hbs * mis, no_pred_segcounts,
|
||||
temporal_predictor_count, t_unpred_seg_counts, hbs, hbs,
|
||||
mi_row + hbs, mi_col + hbs);
|
||||
CSEGS(bs, hbs, 0, 0);
|
||||
CSEGS(bs, qbs, hbs, 0);
|
||||
if (mi_row + 3 * qbs < cm->mi_rows) CSEGS(bs, qbs, 3 * qbs, 0);
|
||||
break;
|
||||
case PARTITION_VERT_A:
|
||||
count_segs(cm, xd, tile, mi, no_pred_segcounts, temporal_predictor_count,
|
||||
t_unpred_seg_counts, hbs, hbs, mi_row, mi_col);
|
||||
count_segs(cm, xd, tile, mi + hbs * mis, no_pred_segcounts,
|
||||
temporal_predictor_count, t_unpred_seg_counts, hbs, hbs,
|
||||
mi_row + hbs, mi_col);
|
||||
count_segs(cm, xd, tile, mi + hbs, no_pred_segcounts,
|
||||
temporal_predictor_count, t_unpred_seg_counts, hbs, bs, mi_row,
|
||||
mi_col + hbs);
|
||||
CSEGS(qbs, bs, 0, 0);
|
||||
CSEGS(qbs, bs, 0, qbs);
|
||||
CSEGS(hbs, bs, 0, hbs);
|
||||
break;
|
||||
case PARTITION_VERT_B:
|
||||
count_segs(cm, xd, tile, mi, no_pred_segcounts, temporal_predictor_count,
|
||||
t_unpred_seg_counts, hbs, bs, mi_row, mi_col);
|
||||
count_segs(cm, xd, tile, mi + hbs, no_pred_segcounts,
|
||||
temporal_predictor_count, t_unpred_seg_counts, hbs, hbs,
|
||||
mi_row, mi_col + hbs);
|
||||
count_segs(cm, xd, tile, mi + hbs + hbs * mis, no_pred_segcounts,
|
||||
temporal_predictor_count, t_unpred_seg_counts, hbs, hbs,
|
||||
mi_row + hbs, mi_col + hbs);
|
||||
CSEGS(hbs, bs, 0, 0);
|
||||
CSEGS(qbs, bs, 0, hbs);
|
||||
if (mi_col + 3 * qbs < cm->mi_cols) CSEGS(qbs, bs, 0, 3 * qbs);
|
||||
break;
|
||||
#else
|
||||
case PARTITION_HORZ_A:
|
||||
CSEGS(hbs, hbs, 0, 0);
|
||||
CSEGS(hbs, hbs, 0, hbs);
|
||||
CSEGS(bs, hbs, hbs, 0);
|
||||
break;
|
||||
case PARTITION_HORZ_B:
|
||||
CSEGS(bs, hbs, 0, 0);
|
||||
CSEGS(hbs, hbs, hbs, 0);
|
||||
CSEGS(hbs, hbs, hbs, hbs);
|
||||
break;
|
||||
case PARTITION_VERT_A:
|
||||
CSEGS(hbs, hbs, 0, 0);
|
||||
CSEGS(hbs, hbs, hbs, 0);
|
||||
CSEGS(hbs, bs, 0, hbs);
|
||||
break;
|
||||
case PARTITION_VERT_B:
|
||||
CSEGS(hbs, bs, 0, 0);
|
||||
CSEGS(hbs, hbs, 0, hbs);
|
||||
CSEGS(hbs, hbs, hbs, hbs);
|
||||
break;
|
||||
#endif
|
||||
case PARTITION_SPLIT: {
|
||||
const BLOCK_SIZE subsize = subsize_lookup[PARTITION_SPLIT][bsize];
|
||||
int n;
|
||||
|
|
@ -260,20 +262,13 @@ static void count_segs_sb(const AV1_COMMON *cm, MACROBLOCKD *xd,
|
|||
bh = mi_size_high[mi[0]->mbmi.sb_type];
|
||||
|
||||
if (bw == bs && bh == bs) {
|
||||
count_segs(cm, xd, tile, mi, no_pred_segcounts, temporal_predictor_count,
|
||||
t_unpred_seg_counts, bs, bs, mi_row, mi_col);
|
||||
CSEGS(bs, bs, 0, 0);
|
||||
} else if (bw == bs && bh < bs) {
|
||||
count_segs(cm, xd, tile, mi, no_pred_segcounts, temporal_predictor_count,
|
||||
t_unpred_seg_counts, bs, hbs, mi_row, mi_col);
|
||||
count_segs(cm, xd, tile, mi + hbs * mis, no_pred_segcounts,
|
||||
temporal_predictor_count, t_unpred_seg_counts, bs, hbs,
|
||||
mi_row + hbs, mi_col);
|
||||
CSEGS(bs, hbs, 0, 0);
|
||||
CSEGS(bs, hbs, hbs, 0);
|
||||
} else if (bw < bs && bh == bs) {
|
||||
count_segs(cm, xd, tile, mi, no_pred_segcounts, temporal_predictor_count,
|
||||
t_unpred_seg_counts, hbs, bs, mi_row, mi_col);
|
||||
count_segs(cm, xd, tile, mi + hbs, no_pred_segcounts,
|
||||
temporal_predictor_count, t_unpred_seg_counts, hbs, bs, mi_row,
|
||||
mi_col + hbs);
|
||||
CSEGS(hbs, bs, 0, 0);
|
||||
CSEGS(hbs, bs, 0, hbs);
|
||||
} else {
|
||||
const BLOCK_SIZE subsize = subsize_lookup[PARTITION_SPLIT][bsize];
|
||||
int n;
|
||||
|
|
@ -290,6 +285,8 @@ static void count_segs_sb(const AV1_COMMON *cm, MACROBLOCKD *xd,
|
|||
}
|
||||
}
|
||||
#endif // CONFIG_EXT_PARTITION_TYPES
|
||||
|
||||
#undef CSEGS
|
||||
}
|
||||
|
||||
void av1_choose_segmap_coding_method(AV1_COMMON *cm, MACROBLOCKD *xd) {
|
||||
|
|
|
|||
2
third_party/aom/av1/encoder/segmentation.h
vendored
2
third_party/aom/av1/encoder/segmentation.h
vendored
|
|
@ -37,7 +37,7 @@ void av1_clear_segdata(struct segmentation *seg, int segment_id,
|
|||
//
|
||||
// abs_delta = SEGMENT_DELTADATA (deltas) abs_delta = SEGMENT_ABSDATA (use
|
||||
// the absolute values given).
|
||||
void av1_set_segment_data(struct segmentation *seg, signed char *feature_data,
|
||||
void av1_set_segment_data(struct segmentation *seg, int8_t *feature_data,
|
||||
unsigned char abs_delta);
|
||||
|
||||
void av1_choose_segmap_coding_method(AV1_COMMON *cm, MACROBLOCKD *xd);
|
||||
|
|
|
|||
17
third_party/aom/av1/encoder/speed_features.c
vendored
17
third_party/aom/av1/encoder/speed_features.c
vendored
|
|
@ -172,20 +172,20 @@ static void set_good_speed_features_framesize_independent(AV1_COMP *cpi,
|
|||
#if CONFIG_TX64X64
|
||||
sf->intra_y_mode_mask[TX_64X64] = INTRA_DC_H_V;
|
||||
#if CONFIG_CFL
|
||||
sf->intra_uv_mode_mask[TX_64X64] = UV_INTRA_DC_H_V;
|
||||
sf->intra_uv_mode_mask[TX_64X64] = UV_INTRA_DC_H_V_CFL;
|
||||
#else
|
||||
sf->intra_uv_mode_mask[TX_64X64] = INTRA_DC_H_V;
|
||||
#endif // CONFIG_CFL
|
||||
#endif // CONFIG_TX64X64
|
||||
sf->intra_y_mode_mask[TX_32X32] = INTRA_DC_H_V;
|
||||
#if CONFIG_CFL
|
||||
sf->intra_uv_mode_mask[TX_32X32] = UV_INTRA_DC_H_V;
|
||||
sf->intra_uv_mode_mask[TX_32X32] = UV_INTRA_DC_H_V_CFL;
|
||||
#else
|
||||
sf->intra_uv_mode_mask[TX_32X32] = INTRA_DC_H_V;
|
||||
#endif
|
||||
sf->intra_y_mode_mask[TX_16X16] = INTRA_DC_H_V;
|
||||
#if CONFIG_CFL
|
||||
sf->intra_uv_mode_mask[TX_16X16] = UV_INTRA_DC_H_V;
|
||||
sf->intra_uv_mode_mask[TX_16X16] = UV_INTRA_DC_H_V_CFL;
|
||||
#else
|
||||
sf->intra_uv_mode_mask[TX_16X16] = INTRA_DC_H_V;
|
||||
#endif
|
||||
|
|
@ -196,10 +196,8 @@ static void set_good_speed_features_framesize_independent(AV1_COMP *cpi,
|
|||
// Use transform domain distortion.
|
||||
// Note var-tx expt always uses pixel domain distortion.
|
||||
sf->use_transform_domain_distortion = 1;
|
||||
#if CONFIG_EXT_INTER
|
||||
sf->disable_wedge_search_var_thresh = 100;
|
||||
sf->fast_wedge_sign_estimate = 1;
|
||||
#endif // CONFIG_EXT_INTER
|
||||
}
|
||||
|
||||
if (speed >= 3) {
|
||||
|
|
@ -240,14 +238,14 @@ static void set_good_speed_features_framesize_independent(AV1_COMP *cpi,
|
|||
#if CONFIG_TX64X64
|
||||
sf->intra_y_mode_mask[TX_64X64] = INTRA_DC;
|
||||
#if CONFIG_CFL
|
||||
sf->intra_uv_mode_mask[TX_64X64] = UV_INTRA_DC;
|
||||
sf->intra_uv_mode_mask[TX_64X64] = UV_INTRA_DC_CFL;
|
||||
#else
|
||||
sf->intra_uv_mode_mask[TX_64X64] = INTRA_DC;
|
||||
#endif // CONFIG_CFL
|
||||
#endif // CONFIG_TX64X64
|
||||
sf->intra_y_mode_mask[TX_32X32] = INTRA_DC;
|
||||
#if CONFIG_CFL
|
||||
sf->intra_uv_mode_mask[TX_32X32] = UV_INTRA_DC;
|
||||
sf->intra_uv_mode_mask[TX_32X32] = UV_INTRA_DC_CFL;
|
||||
#else
|
||||
sf->intra_uv_mode_mask[TX_32X32] = INTRA_DC;
|
||||
#endif // CONFIG_CFL
|
||||
|
|
@ -276,7 +274,7 @@ static void set_good_speed_features_framesize_independent(AV1_COMP *cpi,
|
|||
for (i = 0; i < TX_SIZES; ++i) {
|
||||
sf->intra_y_mode_mask[i] = INTRA_DC;
|
||||
#if CONFIG_CFL
|
||||
sf->intra_uv_mode_mask[i] = UV_INTRA_DC;
|
||||
sf->intra_uv_mode_mask[i] = UV_INTRA_DC_CFL;
|
||||
#else
|
||||
sf->intra_uv_mode_mask[i] = INTRA_DC;
|
||||
#endif // CONFIG_CFL
|
||||
|
|
@ -404,6 +402,7 @@ void av1_set_speed_features_framesize_independent(AV1_COMP *cpi) {
|
|||
sf->alt_ref_search_fp = 0;
|
||||
sf->partition_search_type = SEARCH_PARTITION;
|
||||
sf->tx_type_search.prune_mode = NO_PRUNE;
|
||||
sf->tx_type_search.use_skip_flag_prediction = 1;
|
||||
sf->tx_type_search.fast_intra_tx_type_search = 0;
|
||||
sf->tx_type_search.fast_inter_tx_type_search = 0;
|
||||
sf->less_rectangular_check = 0;
|
||||
|
|
@ -422,10 +421,8 @@ void av1_set_speed_features_framesize_independent(AV1_COMP *cpi) {
|
|||
sf->adaptive_interp_filter_search = 0;
|
||||
sf->allow_partition_search_skip = 0;
|
||||
sf->use_upsampled_references = 1;
|
||||
#if CONFIG_EXT_INTER
|
||||
sf->disable_wedge_search_var_thresh = 0;
|
||||
sf->fast_wedge_sign_estimate = 0;
|
||||
#endif // CONFIG_EXT_INTER
|
||||
|
||||
for (i = 0; i < TX_SIZES; i++) {
|
||||
sf->intra_y_mode_mask[i] = INTRA_ALL;
|
||||
|
|
|
|||
35
third_party/aom/av1/encoder/speed_features.h
vendored
35
third_party/aom/av1/encoder/speed_features.h
vendored
|
|
@ -21,31 +21,34 @@ extern "C" {
|
|||
enum {
|
||||
INTRA_ALL = (1 << DC_PRED) | (1 << V_PRED) | (1 << H_PRED) | (1 << D45_PRED) |
|
||||
(1 << D135_PRED) | (1 << D117_PRED) | (1 << D153_PRED) |
|
||||
(1 << D207_PRED) | (1 << D63_PRED) |
|
||||
#if CONFIG_ALT_INTRA
|
||||
(1 << SMOOTH_PRED) |
|
||||
(1 << D207_PRED) | (1 << D63_PRED) | (1 << SMOOTH_PRED) |
|
||||
#if CONFIG_SMOOTH_HV
|
||||
(1 << SMOOTH_V_PRED) | (1 << SMOOTH_H_PRED) |
|
||||
#endif // CONFIG_SMOOTH_HV
|
||||
#endif // CONFIG_ALT_INTRA
|
||||
(1 << TM_PRED),
|
||||
#if CONFIG_CFL
|
||||
UV_INTRA_ALL = (1 << UV_DC_PRED) | (1 << UV_V_PRED) | (1 << UV_H_PRED) |
|
||||
(1 << UV_D45_PRED) | (1 << UV_D135_PRED) |
|
||||
(1 << UV_D117_PRED) | (1 << UV_D153_PRED) |
|
||||
(1 << UV_D207_PRED) | (1 << UV_D63_PRED) |
|
||||
#if CONFIG_ALT_INTRA
|
||||
(1 << UV_SMOOTH_PRED) |
|
||||
#if CONFIG_SMOOTH_HV
|
||||
(1 << UV_SMOOTH_V_PRED) | (1 << UV_SMOOTH_H_PRED) |
|
||||
#endif // CONFIG_SMOOTH_HV
|
||||
#endif // CONFIG_ALT_INTRA
|
||||
(1 << UV_TM_PRED),
|
||||
(1 << UV_TM_PRED) | (1 << UV_CFL_PRED),
|
||||
UV_INTRA_DC = (1 << UV_DC_PRED),
|
||||
UV_INTRA_DC_CFL = (1 << UV_DC_PRED) | (1 << UV_CFL_PRED),
|
||||
UV_INTRA_DC_TM = (1 << UV_DC_PRED) | (1 << UV_TM_PRED),
|
||||
UV_INTRA_DC_TM_CFL =
|
||||
(1 << UV_DC_PRED) | (1 << UV_TM_PRED) | (1 << UV_CFL_PRED),
|
||||
UV_INTRA_DC_H_V = (1 << UV_DC_PRED) | (1 << UV_V_PRED) | (1 << UV_H_PRED),
|
||||
UV_INTRA_DC_H_V_CFL = (1 << UV_DC_PRED) | (1 << UV_V_PRED) |
|
||||
(1 << UV_H_PRED) | (1 << UV_CFL_PRED),
|
||||
UV_INTRA_DC_TM_H_V = (1 << UV_DC_PRED) | (1 << UV_TM_PRED) |
|
||||
(1 << UV_V_PRED) | (1 << UV_H_PRED),
|
||||
UV_INTRA_DC_TM_H_V_CFL = (1 << UV_DC_PRED) | (1 << UV_TM_PRED) |
|
||||
(1 << UV_V_PRED) | (1 << UV_H_PRED) |
|
||||
(1 << UV_CFL_PRED),
|
||||
#endif // CONFIG_CFL
|
||||
INTRA_DC = (1 << DC_PRED),
|
||||
INTRA_DC_TM = (1 << DC_PRED) | (1 << TM_PRED),
|
||||
|
|
@ -54,7 +57,6 @@ enum {
|
|||
(1 << DC_PRED) | (1 << TM_PRED) | (1 << V_PRED) | (1 << H_PRED)
|
||||
};
|
||||
|
||||
#if CONFIG_EXT_INTER
|
||||
enum {
|
||||
#if CONFIG_COMPOUND_SINGLEREF
|
||||
// TODO(zoeliu): To further consider following single ref comp modes:
|
||||
|
|
@ -90,17 +92,6 @@ enum {
|
|||
(1 << NEW_NEARMV) | (1 << NEAR_NEWMV) |
|
||||
(1 << NEAR_NEARMV),
|
||||
};
|
||||
#else // !CONFIG_EXT_INTER
|
||||
enum {
|
||||
INTER_ALL = (1 << NEARESTMV) | (1 << NEARMV) | (1 << ZEROMV) | (1 << NEWMV),
|
||||
INTER_NEAREST = (1 << NEARESTMV),
|
||||
INTER_NEAREST_NEW = (1 << NEARESTMV) | (1 << NEWMV),
|
||||
INTER_NEAREST_ZERO = (1 << NEARESTMV) | (1 << ZEROMV),
|
||||
INTER_NEAREST_NEW_ZERO = (1 << NEARESTMV) | (1 << ZEROMV) | (1 << NEWMV),
|
||||
INTER_NEAREST_NEAR_NEW = (1 << NEARESTMV) | (1 << NEARMV) | (1 << NEWMV),
|
||||
INTER_NEAREST_NEAR_ZERO = (1 << NEARESTMV) | (1 << NEARMV) | (1 << ZEROMV),
|
||||
};
|
||||
#endif // CONFIG_EXT_INTER
|
||||
|
||||
enum {
|
||||
DISABLE_ALL_INTER_SPLIT = (1 << THR_COMP_GA) | (1 << THR_COMP_LA) |
|
||||
|
|
@ -209,6 +200,10 @@ typedef struct {
|
|||
TX_TYPE_PRUNE_MODE prune_mode;
|
||||
int fast_intra_tx_type_search;
|
||||
int fast_inter_tx_type_search;
|
||||
|
||||
// Use a skip flag prediction model to detect blocks with skip = 1 early
|
||||
// and avoid doing full TX type search for such blocks.
|
||||
int use_skip_flag_prediction;
|
||||
} TX_TYPE_SEARCH;
|
||||
|
||||
typedef enum {
|
||||
|
|
@ -409,13 +404,11 @@ typedef struct SPEED_FEATURES {
|
|||
// Choose a very large value (UINT_MAX) to use 8-tap always
|
||||
unsigned int disable_filter_search_var_thresh;
|
||||
|
||||
#if CONFIG_EXT_INTER
|
||||
// A source variance threshold below which wedge search is disabled
|
||||
unsigned int disable_wedge_search_var_thresh;
|
||||
|
||||
// Whether fast wedge sign estimate is used
|
||||
int fast_wedge_sign_estimate;
|
||||
#endif // CONFIG_EXT_INTER
|
||||
|
||||
// These bit masks allow you to enable or disable intra modes for each
|
||||
// transform size separately.
|
||||
|
|
|
|||
41
third_party/aom/av1/encoder/subexp.c
vendored
41
third_party/aom/av1/encoder/subexp.c
vendored
|
|
@ -138,47 +138,6 @@ int av1_prob_diff_update_savings_search(const unsigned int *ct, aom_prob oldp,
|
|||
return bestsavings;
|
||||
}
|
||||
|
||||
int av1_prob_diff_update_savings_search_model(const unsigned int *ct,
|
||||
const aom_prob oldp,
|
||||
aom_prob *bestp, aom_prob upd,
|
||||
int stepsize, int probwt) {
|
||||
int i, old_b, new_b, update_b, savings, bestsavings;
|
||||
int newp;
|
||||
const int step_sign = *bestp > oldp ? -1 : 1;
|
||||
const int step = stepsize * step_sign;
|
||||
const int upd_cost = av1_cost_one(upd) - av1_cost_zero(upd);
|
||||
const aom_prob *newplist, *oldplist;
|
||||
aom_prob bestnewp;
|
||||
oldplist = av1_pareto8_full[oldp - 1];
|
||||
old_b = cost_branch256(ct + 2 * PIVOT_NODE, oldp);
|
||||
for (i = UNCONSTRAINED_NODES; i < ENTROPY_NODES; ++i)
|
||||
old_b += cost_branch256(ct + 2 * i, oldplist[i - UNCONSTRAINED_NODES]);
|
||||
|
||||
bestsavings = 0;
|
||||
bestnewp = oldp;
|
||||
|
||||
assert(stepsize > 0);
|
||||
|
||||
if (old_b > upd_cost + (MIN_DELP_BITS << AV1_PROB_COST_SHIFT)) {
|
||||
for (newp = *bestp; (newp - oldp) * step_sign < 0; newp += step) {
|
||||
if (newp < 1 || newp > 255) continue;
|
||||
newplist = av1_pareto8_full[newp - 1];
|
||||
new_b = cost_branch256(ct + 2 * PIVOT_NODE, newp);
|
||||
for (i = UNCONSTRAINED_NODES; i < ENTROPY_NODES; ++i)
|
||||
new_b += cost_branch256(ct + 2 * i, newplist[i - UNCONSTRAINED_NODES]);
|
||||
update_b = prob_diff_update_cost(newp, oldp) + upd_cost;
|
||||
savings = old_b - new_b - update_b * probwt;
|
||||
if (savings > bestsavings) {
|
||||
bestsavings = savings;
|
||||
bestnewp = newp;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
*bestp = bestnewp;
|
||||
return bestsavings;
|
||||
}
|
||||
|
||||
void av1_cond_prob_diff_update(aom_writer *w, aom_prob *oldp,
|
||||
const unsigned int ct[2], int probwt) {
|
||||
const aom_prob upd = DIFF_UPDATE_PROB;
|
||||
|
|
|
|||
132
third_party/aom/av1/encoder/temporal_filter.c
vendored
132
third_party/aom/av1/encoder/temporal_filter.c
vendored
|
|
@ -44,18 +44,13 @@ static void temporal_filter_predictors_mb_c(
|
|||
ConvolveParams conv_params = get_conv_params(which_mv, which_mv, 0);
|
||||
|
||||
#if USE_TEMPORALFILTER_12TAP
|
||||
#if CONFIG_DUAL_FILTER
|
||||
const InterpFilter interp_filter[4] = { TEMPORALFILTER_12TAP,
|
||||
TEMPORALFILTER_12TAP,
|
||||
TEMPORALFILTER_12TAP,
|
||||
TEMPORALFILTER_12TAP };
|
||||
#else
|
||||
const InterpFilter interp_filter = TEMPORALFILTER_12TAP;
|
||||
#endif
|
||||
const InterpFilters interp_filters =
|
||||
av1_broadcast_interp_filter(TEMPORALFILTER_12TAP);
|
||||
(void)xd;
|
||||
#else
|
||||
const InterpFilter interp_filter = xd->mi[0]->mbmi.interp_filter;
|
||||
const InterpFilters interp_filters = xd->mi[0]->mbmi.interp_filters;
|
||||
#endif // USE_TEMPORALFILTER_12TAP
|
||||
|
||||
#if CONFIG_GLOBAL_MOTION || CONFIG_WARPED_MOTION
|
||||
WarpTypesAllowed warp_types;
|
||||
memset(&warp_types, 0, sizeof(WarpTypesAllowed));
|
||||
|
|
@ -72,7 +67,7 @@ static void temporal_filter_predictors_mb_c(
|
|||
#if CONFIG_HIGHBITDEPTH
|
||||
if (xd->cur_buf->flags & YV12_FLAG_HIGHBITDEPTH) {
|
||||
av1_highbd_build_inter_predictor(y_mb_ptr, stride, &pred[0], 16, &mv, scale,
|
||||
16, 16, which_mv, interp_filter,
|
||||
16, 16, which_mv, interp_filters,
|
||||
#if CONFIG_GLOBAL_MOTION || CONFIG_WARPED_MOTION
|
||||
&warp_types, x, y,
|
||||
#endif // CONFIG_GLOBAL_MOTION || CONFIG_WARPED_MOTION
|
||||
|
|
@ -80,7 +75,7 @@ static void temporal_filter_predictors_mb_c(
|
|||
|
||||
av1_highbd_build_inter_predictor(u_mb_ptr, uv_stride, &pred[256],
|
||||
uv_block_width, &mv, scale, uv_block_width,
|
||||
uv_block_height, which_mv, interp_filter,
|
||||
uv_block_height, which_mv, interp_filters,
|
||||
#if CONFIG_GLOBAL_MOTION || CONFIG_WARPED_MOTION
|
||||
&warp_types, x, y,
|
||||
#endif // CONFIG_GLOBAL_MOTION || CONFIG_WARPED_MOTION
|
||||
|
|
@ -88,7 +83,7 @@ static void temporal_filter_predictors_mb_c(
|
|||
|
||||
av1_highbd_build_inter_predictor(v_mb_ptr, uv_stride, &pred[512],
|
||||
uv_block_width, &mv, scale, uv_block_width,
|
||||
uv_block_height, which_mv, interp_filter,
|
||||
uv_block_height, which_mv, interp_filters,
|
||||
#if CONFIG_GLOBAL_MOTION || CONFIG_WARPED_MOTION
|
||||
&warp_types, x, y,
|
||||
#endif // CONFIG_GLOBAL_MOTION || CONFIG_WARPED_MOTION
|
||||
|
|
@ -97,7 +92,7 @@ static void temporal_filter_predictors_mb_c(
|
|||
}
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
av1_build_inter_predictor(y_mb_ptr, stride, &pred[0], 16, &mv, scale, 16, 16,
|
||||
&conv_params, interp_filter,
|
||||
&conv_params, interp_filters,
|
||||
#if CONFIG_GLOBAL_MOTION || CONFIG_WARPED_MOTION
|
||||
&warp_types, x, y, 0, 0,
|
||||
#endif // CONFIG_GLOBAL_MOTION || CONFIG_WARPED_MOTION
|
||||
|
|
@ -105,7 +100,7 @@ static void temporal_filter_predictors_mb_c(
|
|||
|
||||
av1_build_inter_predictor(u_mb_ptr, uv_stride, &pred[256], uv_block_width,
|
||||
&mv, scale, uv_block_width, uv_block_height,
|
||||
&conv_params, interp_filter,
|
||||
&conv_params, interp_filters,
|
||||
#if CONFIG_GLOBAL_MOTION || CONFIG_WARPED_MOTION
|
||||
&warp_types, x, y, 1, 0,
|
||||
#endif // CONFIG_GLOBAL_MOTION || CONFIG_WARPED_MOTION
|
||||
|
|
@ -113,7 +108,7 @@ static void temporal_filter_predictors_mb_c(
|
|||
|
||||
av1_build_inter_predictor(v_mb_ptr, uv_stride, &pred[512], uv_block_width,
|
||||
&mv, scale, uv_block_width, uv_block_height,
|
||||
&conv_params, interp_filter,
|
||||
&conv_params, interp_filters,
|
||||
#if CONFIG_GLOBAL_MOTION || CONFIG_WARPED_MOTION
|
||||
&warp_types, x, y, 2, 0,
|
||||
#endif // CONFIG_GLOBAL_MOTION || CONFIG_WARPED_MOTION
|
||||
|
|
@ -291,15 +286,30 @@ static int temporal_filter_find_matching_mb_c(AV1_COMP *cpi,
|
|||
|
||||
x->mv_limits = tmp_mv_limits;
|
||||
|
||||
// Ignore mv costing by sending NULL pointer instead of cost array
|
||||
bestsme = cpi->find_fractional_mv_step(
|
||||
x, &best_ref_mv1, cpi->common.allow_high_precision_mv, x->errorperbit,
|
||||
&cpi->fn_ptr[BLOCK_16X16], 0, mv_sf->subpel_iters_per_step,
|
||||
cond_cost_list(cpi, cost_list), NULL, NULL, &distortion, &sse, NULL,
|
||||
#if CONFIG_EXT_INTER
|
||||
NULL, 0, 0,
|
||||
// Ignore mv costing by sending NULL pointer instead of cost array
|
||||
#if CONFIG_AMVR
|
||||
if (cpi->common.cur_frame_mv_precision_level == 1) {
|
||||
const uint8_t *const src_address = x->plane[0].src.buf;
|
||||
const int src_stride = x->plane[0].src.stride;
|
||||
const uint8_t *const y = xd->plane[0].pre[0].buf;
|
||||
const int y_stride = xd->plane[0].pre[0].stride;
|
||||
const int offset = x->best_mv.as_mv.row * y_stride + x->best_mv.as_mv.col;
|
||||
|
||||
x->best_mv.as_mv.row *= 8;
|
||||
x->best_mv.as_mv.col *= 8;
|
||||
|
||||
bestsme = cpi->fn_ptr[BLOCK_16X16].vf(y + offset, y_stride, src_address,
|
||||
src_stride, &sse);
|
||||
} else {
|
||||
#endif
|
||||
bestsme = cpi->find_fractional_mv_step(
|
||||
x, &best_ref_mv1, cpi->common.allow_high_precision_mv, x->errorperbit,
|
||||
&cpi->fn_ptr[BLOCK_16X16], 0, mv_sf->subpel_iters_per_step,
|
||||
cond_cost_list(cpi, cost_list), NULL, NULL, &distortion, &sse, NULL,
|
||||
NULL, 0, 0, 0, 0, 0);
|
||||
#if CONFIG_AMVR
|
||||
}
|
||||
#endif
|
||||
0, 0, 0);
|
||||
|
||||
x->e_mbd.mi[0]->bmi[0].as_mv[0] = x->best_mv;
|
||||
|
||||
|
|
@ -311,6 +321,9 @@ static int temporal_filter_find_matching_mb_c(AV1_COMP *cpi,
|
|||
}
|
||||
|
||||
static void temporal_filter_iterate_c(AV1_COMP *cpi,
|
||||
#if CONFIG_BGSPRITE
|
||||
YV12_BUFFER_CONFIG *target,
|
||||
#endif // CONFIG_BGSPRITE
|
||||
YV12_BUFFER_CONFIG **frames,
|
||||
int frame_count, int alt_ref_index,
|
||||
int strength,
|
||||
|
|
@ -452,9 +465,17 @@ static void temporal_filter_iterate_c(AV1_COMP *cpi,
|
|||
if (mbd->cur_buf->flags & YV12_FLAG_HIGHBITDEPTH) {
|
||||
uint16_t *dst1_16;
|
||||
uint16_t *dst2_16;
|
||||
#if CONFIG_BGSPRITE
|
||||
dst1 = target->y_buffer;
|
||||
#else
|
||||
dst1 = cpi->alt_ref_buffer.y_buffer;
|
||||
#endif // CONFIG_BGSPRITE
|
||||
dst1_16 = CONVERT_TO_SHORTPTR(dst1);
|
||||
#if CONFIG_BGSPRITE
|
||||
stride = target->y_stride;
|
||||
#else
|
||||
stride = cpi->alt_ref_buffer.y_stride;
|
||||
#endif // CONFIG_BGSPRITE
|
||||
byte = mb_y_offset;
|
||||
for (i = 0, k = 0; i < 16; i++) {
|
||||
for (j = 0; j < 16; j++, k++) {
|
||||
|
|
@ -494,8 +515,13 @@ static void temporal_filter_iterate_c(AV1_COMP *cpi,
|
|||
}
|
||||
} else {
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
dst1 = cpi->alt_ref_buffer.y_buffer;
|
||||
stride = cpi->alt_ref_buffer.y_stride;
|
||||
#if CONFIG_BGSPRITE
|
||||
dst1 = target->y_buffer;
|
||||
stride = target->y_stride;
|
||||
#else
|
||||
dst1 = cpi->alt_ref_buffer.y_buffer;
|
||||
stride = cpi->alt_ref_buffer.y_stride;
|
||||
#endif // CONFIG_BGSPRITE
|
||||
byte = mb_y_offset;
|
||||
for (i = 0, k = 0; i < 16; i++) {
|
||||
for (j = 0; j < 16; j++, k++) {
|
||||
|
|
@ -507,10 +533,15 @@ static void temporal_filter_iterate_c(AV1_COMP *cpi,
|
|||
}
|
||||
byte += stride - 16;
|
||||
}
|
||||
|
||||
dst1 = cpi->alt_ref_buffer.u_buffer;
|
||||
dst2 = cpi->alt_ref_buffer.v_buffer;
|
||||
stride = cpi->alt_ref_buffer.uv_stride;
|
||||
#if CONFIG_BGSPRITE
|
||||
dst1 = target->u_buffer;
|
||||
dst2 = target->v_buffer;
|
||||
stride = target->uv_stride;
|
||||
#else
|
||||
dst1 = cpi->alt_ref_buffer.u_buffer;
|
||||
dst2 = cpi->alt_ref_buffer.v_buffer;
|
||||
stride = cpi->alt_ref_buffer.uv_stride;
|
||||
#endif // CONFIG_BGSPRITE
|
||||
byte = mb_uv_offset;
|
||||
for (i = 0, k = 256; i < mb_uv_height; i++) {
|
||||
for (j = 0; j < mb_uv_width; j++, k++) {
|
||||
|
|
@ -604,7 +635,7 @@ static void adjust_arnr_filter(AV1_COMP *cpi, int distance, int group_boost,
|
|||
|
||||
void av1_temporal_filter(AV1_COMP *cpi,
|
||||
#if CONFIG_BGSPRITE
|
||||
YV12_BUFFER_CONFIG *bg,
|
||||
YV12_BUFFER_CONFIG *bg, YV12_BUFFER_CONFIG *target,
|
||||
#endif // CONFIG_BGSPRITE
|
||||
int distance) {
|
||||
RATE_CONTROL *const rc = &cpi->rc;
|
||||
|
|
@ -618,7 +649,7 @@ void av1_temporal_filter(AV1_COMP *cpi,
|
|||
YV12_BUFFER_CONFIG *frames[MAX_LAG_BUFFERS] = { NULL };
|
||||
#if CONFIG_EXT_REFS
|
||||
const GF_GROUP *const gf_group = &cpi->twopass.gf_group;
|
||||
#endif
|
||||
#endif // CONFIG_EXT_REFS
|
||||
|
||||
// Apply context specific adjustments to the arnr filter parameters.
|
||||
adjust_arnr_filter(cpi, distance, rc->gfu_boost, &frames_to_blur, &strength);
|
||||
|
|
@ -627,19 +658,34 @@ void av1_temporal_filter(AV1_COMP *cpi,
|
|||
// case it is more beneficial to use non-zero strength
|
||||
// filtering.
|
||||
#if CONFIG_EXT_REFS
|
||||
if (gf_group->rf_level[gf_group->index] == GF_ARF_LOW) {
|
||||
if (gf_group->update_type[gf_group->index] == INTNL_ARF_UPDATE) {
|
||||
strength = 0;
|
||||
frames_to_blur = 1;
|
||||
}
|
||||
#endif
|
||||
|
||||
#if CONFIG_EXT_REFS
|
||||
if (strength == 0 && frames_to_blur == 1) {
|
||||
cpi->is_arf_filter_off[gf_group->arf_update_idx[gf_group->index]] = 1;
|
||||
} else {
|
||||
cpi->is_arf_filter_off[gf_group->arf_update_idx[gf_group->index]] = 0;
|
||||
int which_arf = gf_group->arf_update_idx[gf_group->index];
|
||||
|
||||
#if USE_GF16_MULTI_LAYER
|
||||
if (cpi->rc.baseline_gf_interval == 16) {
|
||||
// Identify the index to the current ARF.
|
||||
const int num_arfs_in_gf = cpi->num_extra_arfs + 1;
|
||||
int arf_idx;
|
||||
for (arf_idx = 0; arf_idx < num_arfs_in_gf; arf_idx++) {
|
||||
if (gf_group->index == cpi->arf_pos_in_gf[arf_idx]) {
|
||||
which_arf = arf_idx;
|
||||
break;
|
||||
}
|
||||
}
|
||||
assert(arf_idx < num_arfs_in_gf);
|
||||
}
|
||||
#endif
|
||||
#endif // USE_GF16_MULTI_LAYER
|
||||
|
||||
// Set the temporal filtering status for the corresponding OVERLAY frame
|
||||
if (strength == 0 && frames_to_blur == 1)
|
||||
cpi->is_arf_filter_off[which_arf] = 1;
|
||||
else
|
||||
cpi->is_arf_filter_off[which_arf] = 0;
|
||||
#endif // CONFIG_EXT_REFS
|
||||
|
||||
frames_to_blur_backward = (frames_to_blur / 2);
|
||||
frames_to_blur_forward = ((frames_to_blur - 1) / 2);
|
||||
|
|
@ -678,6 +724,10 @@ void av1_temporal_filter(AV1_COMP *cpi,
|
|||
#endif // CONFIG_HIGHBITDEPTH
|
||||
}
|
||||
|
||||
temporal_filter_iterate_c(cpi, frames, frames_to_blur,
|
||||
frames_to_blur_backward, strength, &sf);
|
||||
temporal_filter_iterate_c(cpi,
|
||||
#if CONFIG_BGSPRITE
|
||||
target,
|
||||
#endif // CONFIG_BGSPRITE
|
||||
frames, frames_to_blur, frames_to_blur_backward,
|
||||
strength, &sf);
|
||||
}
|
||||
|
|
|
|||
|
|
@ -18,7 +18,7 @@ extern "C" {
|
|||
|
||||
void av1_temporal_filter(AV1_COMP *cpi,
|
||||
#if CONFIG_BGSPRITE
|
||||
YV12_BUFFER_CONFIG *bg,
|
||||
YV12_BUFFER_CONFIG *bg, YV12_BUFFER_CONFIG *target,
|
||||
#endif // CONFIG_BGSPRITE
|
||||
int distance);
|
||||
|
||||
|
|
|
|||
175
third_party/aom/av1/encoder/tokenize.c
vendored
175
third_party/aom/av1/encoder/tokenize.c
vendored
|
|
@ -315,36 +315,30 @@ static INLINE void add_token(TOKENEXTRA **t,
|
|||
(*t)->eob_val = eob_val;
|
||||
(*t)->first_val = first_val;
|
||||
(*t)++;
|
||||
|
||||
if (token == BLOCK_Z_TOKEN) {
|
||||
update_cdf(*head_cdf, 0, HEAD_TOKENS + 1);
|
||||
} else {
|
||||
if (eob_val != LAST_EOB) {
|
||||
const int symb = 2 * AOMMIN(token, TWO_TOKEN) - eob_val + first_val;
|
||||
update_cdf(*head_cdf, symb, HEAD_TOKENS + first_val);
|
||||
}
|
||||
if (token > ONE_TOKEN)
|
||||
update_cdf(*tail_cdf, token - TWO_TOKEN, TAIL_TOKENS);
|
||||
}
|
||||
}
|
||||
#endif // !CONFIG_PVQ || CONFIG_VAR_TX
|
||||
|
||||
#if CONFIG_PALETTE
|
||||
void av1_tokenize_palette_sb(const AV1_COMP *cpi,
|
||||
const struct ThreadData *const td, int plane,
|
||||
TOKENEXTRA **t, RUN_TYPE dry_run, BLOCK_SIZE bsize,
|
||||
int *rate) {
|
||||
assert(plane == 0 || plane == 1);
|
||||
const MACROBLOCK *const x = &td->mb;
|
||||
const MACROBLOCKD *const xd = &x->e_mbd;
|
||||
const MB_MODE_INFO *const mbmi = &xd->mi[0]->mbmi;
|
||||
const uint8_t *const color_map = xd->plane[plane].color_index_map;
|
||||
const PALETTE_MODE_INFO *const pmi = &mbmi->palette_mode_info;
|
||||
aom_cdf_prob(
|
||||
*palette_cdf)[PALETTE_COLOR_INDEX_CONTEXTS][CDF_SIZE(PALETTE_COLORS)] =
|
||||
plane ? xd->tile_ctx->palette_uv_color_index_cdf
|
||||
: xd->tile_ctx->palette_y_color_index_cdf;
|
||||
int plane_block_width, rows, cols;
|
||||
av1_get_block_dimensions(bsize, plane, xd, &plane_block_width, NULL, &rows,
|
||||
&cols);
|
||||
static int cost_and_tokenize_map(Av1ColorMapParam *param, TOKENEXTRA **t,
|
||||
int calc_rate) {
|
||||
const uint8_t *const color_map = param->color_map;
|
||||
MapCdf map_cdf = param->map_cdf;
|
||||
ColorCost color_cost = param->color_cost;
|
||||
const int plane_block_width = param->plane_width;
|
||||
const int rows = param->rows;
|
||||
const int cols = param->cols;
|
||||
const int n = param->n_colors;
|
||||
|
||||
// The first color index does not use context or entropy.
|
||||
(*t)->token = color_map[0];
|
||||
(*t)->palette_cdf = NULL;
|
||||
(*t)->skip_eob_node = 0;
|
||||
++(*t);
|
||||
|
||||
const int n = pmi->palette_size[plane];
|
||||
const int calc_rate = rate && dry_run == DRY_RUN_COSTCOEFFS;
|
||||
int this_rate = 0;
|
||||
uint8_t color_order[PALETTE_MAX_SIZE];
|
||||
#if CONFIG_PALETTE_THROUGHPUT
|
||||
|
|
@ -360,18 +354,99 @@ void av1_tokenize_palette_sb(const AV1_COMP *cpi,
|
|||
color_map, plane_block_width, i, j, n, color_order, &color_new_idx);
|
||||
assert(color_new_idx >= 0 && color_new_idx < n);
|
||||
if (calc_rate) {
|
||||
this_rate += cpi->palette_y_color_cost[n - PALETTE_MIN_SIZE][color_ctx]
|
||||
[color_new_idx];
|
||||
this_rate +=
|
||||
(*color_cost)[n - PALETTE_MIN_SIZE][color_ctx][color_new_idx];
|
||||
} else {
|
||||
(*t)->token = color_new_idx;
|
||||
(*t)->color_map_cdf = map_cdf[n - PALETTE_MIN_SIZE][color_ctx];
|
||||
++(*t);
|
||||
}
|
||||
(*t)->token = color_new_idx;
|
||||
(*t)->palette_cdf = palette_cdf[n - PALETTE_MIN_SIZE][color_ctx];
|
||||
(*t)->skip_eob_node = 0;
|
||||
++(*t);
|
||||
}
|
||||
}
|
||||
if (rate) *rate += this_rate;
|
||||
if (calc_rate) return this_rate;
|
||||
return 0;
|
||||
}
|
||||
|
||||
static void get_palette_params(const MACROBLOCK *const x, int plane,
|
||||
BLOCK_SIZE bsize, Av1ColorMapParam *params) {
|
||||
const MACROBLOCKD *const xd = &x->e_mbd;
|
||||
const MB_MODE_INFO *const mbmi = &xd->mi[0]->mbmi;
|
||||
const PALETTE_MODE_INFO *const pmi = &mbmi->palette_mode_info;
|
||||
params->color_map = xd->plane[plane].color_index_map;
|
||||
params->map_cdf = plane ? xd->tile_ctx->palette_uv_color_index_cdf
|
||||
: xd->tile_ctx->palette_y_color_index_cdf;
|
||||
params->color_cost =
|
||||
plane ? &x->palette_uv_color_cost : &x->palette_y_color_cost;
|
||||
params->n_colors = pmi->palette_size[plane];
|
||||
av1_get_block_dimensions(bsize, plane, xd, ¶ms->plane_width, NULL,
|
||||
¶ms->rows, ¶ms->cols);
|
||||
}
|
||||
|
||||
#if CONFIG_MRC_TX && SIGNAL_ANY_MRC_MASK
|
||||
static void get_mrc_params(const MACROBLOCK *const x, int block,
|
||||
TX_SIZE tx_size, Av1ColorMapParam *params) {
|
||||
memset(params, 0, sizeof(*params));
|
||||
const MACROBLOCKD *const xd = &x->e_mbd;
|
||||
const MB_MODE_INFO *const mbmi = &xd->mi[0]->mbmi;
|
||||
const int is_inter = is_inter_block(mbmi);
|
||||
params->color_map = BLOCK_OFFSET(xd->mrc_mask, block);
|
||||
params->map_cdf = is_inter ? xd->tile_ctx->mrc_mask_inter_cdf
|
||||
: xd->tile_ctx->mrc_mask_intra_cdf;
|
||||
params->color_cost =
|
||||
is_inter ? &x->mrc_mask_inter_cost : &x->mrc_mask_intra_cost;
|
||||
params->n_colors = 2;
|
||||
params->plane_width = tx_size_wide[tx_size];
|
||||
params->rows = tx_size_high[tx_size];
|
||||
params->cols = tx_size_wide[tx_size];
|
||||
}
|
||||
#endif // CONFIG_MRC_TX && SIGNAL_ANY_MRC_MASK
|
||||
|
||||
static void get_color_map_params(const MACROBLOCK *const x, int plane,
|
||||
int block, BLOCK_SIZE bsize, TX_SIZE tx_size,
|
||||
COLOR_MAP_TYPE type,
|
||||
Av1ColorMapParam *params) {
|
||||
(void)block;
|
||||
(void)tx_size;
|
||||
memset(params, 0, sizeof(*params));
|
||||
switch (type) {
|
||||
case PALETTE_MAP: get_palette_params(x, plane, bsize, params); break;
|
||||
#if CONFIG_MRC_TX && SIGNAL_ANY_MRC_MASK
|
||||
case MRC_MAP: get_mrc_params(x, block, tx_size, params); break;
|
||||
#endif // CONFIG_MRC_TX && SIGNAL_ANY_MRC_MASK
|
||||
default: assert(0 && "Invalid color map type"); return;
|
||||
}
|
||||
}
|
||||
|
||||
int av1_cost_color_map(const MACROBLOCK *const x, int plane, int block,
|
||||
BLOCK_SIZE bsize, TX_SIZE tx_size, COLOR_MAP_TYPE type) {
|
||||
assert(plane == 0 || plane == 1);
|
||||
Av1ColorMapParam color_map_params;
|
||||
get_color_map_params(x, plane, block, bsize, tx_size, type,
|
||||
&color_map_params);
|
||||
return cost_and_tokenize_map(&color_map_params, NULL, 1);
|
||||
}
|
||||
|
||||
void av1_tokenize_color_map(const MACROBLOCK *const x, int plane, int block,
|
||||
TOKENEXTRA **t, BLOCK_SIZE bsize, TX_SIZE tx_size,
|
||||
COLOR_MAP_TYPE type) {
|
||||
assert(plane == 0 || plane == 1);
|
||||
#if CONFIG_MRC_TX
|
||||
if (type == MRC_MAP) {
|
||||
const int is_inter = is_inter_block(&x->e_mbd.mi[0]->mbmi);
|
||||
if ((is_inter && !SIGNAL_MRC_MASK_INTER) ||
|
||||
(!is_inter && !SIGNAL_MRC_MASK_INTRA))
|
||||
return;
|
||||
}
|
||||
#endif // CONFIG_MRC_TX
|
||||
Av1ColorMapParam color_map_params;
|
||||
get_color_map_params(x, plane, block, bsize, tx_size, type,
|
||||
&color_map_params);
|
||||
// The first color index does not use context or entropy.
|
||||
(*t)->token = color_map_params.color_map[0];
|
||||
(*t)->color_map_cdf = NULL;
|
||||
++(*t);
|
||||
cost_and_tokenize_map(&color_map_params, t, 0);
|
||||
}
|
||||
#endif // CONFIG_PALETTE
|
||||
|
||||
#if CONFIG_PVQ
|
||||
static void add_pvq_block(AV1_COMMON *const cm, MACROBLOCK *const x,
|
||||
|
|
@ -410,7 +485,7 @@ static void tokenize_pvq(int plane, int block, int blk_row, int blk_col,
|
|||
|
||||
assert(block < MAX_PVQ_BLOCKS_IN_SB);
|
||||
pvq_info = &x->pvq[block][plane];
|
||||
add_pvq_block((AV1_COMMON * const)cm, x, pvq_info);
|
||||
add_pvq_block((AV1_COMMON * const) cm, x, pvq_info);
|
||||
}
|
||||
#endif // CONFIG_PVQ
|
||||
|
||||
|
|
@ -444,8 +519,6 @@ static void tokenize_b(int plane, int block, int blk_row, int blk_col,
|
|||
av1_get_tx_type(type, xd, blk_row, blk_col, block, tx_size);
|
||||
const SCAN_ORDER *const scan_order = get_scan(cm, tx_size, tx_type, mbmi);
|
||||
const int ref = is_inter_block(mbmi);
|
||||
unsigned int(*const counts)[COEFF_CONTEXTS][ENTROPY_TOKENS] =
|
||||
td->rd_counts.coef_counts[txsize_sqr_map[tx_size]][type][ref];
|
||||
FRAME_CONTEXT *ec_ctx = xd->tile_ctx;
|
||||
aom_cdf_prob(
|
||||
*const coef_head_cdfs)[COEFF_CONTEXTS][CDF_SIZE(ENTROPY_TOKENS)] =
|
||||
|
|
@ -453,13 +526,9 @@ static void tokenize_b(int plane, int block, int blk_row, int blk_col,
|
|||
aom_cdf_prob(
|
||||
*const coef_tail_cdfs)[COEFF_CONTEXTS][CDF_SIZE(ENTROPY_TOKENS)] =
|
||||
ec_ctx->coef_tail_cdfs[txsize_sqr_map[tx_size]][type][ref];
|
||||
unsigned int(*const blockz_count)[2] =
|
||||
td->counts->blockz_count[txsize_sqr_map[tx_size]][type][ref];
|
||||
int eob_val;
|
||||
int first_val = 1;
|
||||
const int seg_eob = get_tx_eob(&cpi->common.seg, segment_id, tx_size);
|
||||
unsigned int(*const eob_branch)[COEFF_CONTEXTS] =
|
||||
td->counts->eob_branch[txsize_sqr_map[tx_size]][type][ref];
|
||||
const int seg_eob = av1_get_tx_eob(&cpi->common.seg, segment_id, tx_size);
|
||||
const uint8_t *const band = get_band_translate(tx_size);
|
||||
int16_t token;
|
||||
EXTRABIT extra;
|
||||
|
|
@ -470,12 +539,15 @@ static void tokenize_b(int plane, int block, int blk_row, int blk_col,
|
|||
nb = scan_order->neighbors;
|
||||
c = 0;
|
||||
|
||||
#if CONFIG_MRC_TX && SIGNAL_ANY_MRC_MASK
|
||||
if (tx_type == MRC_DCT)
|
||||
av1_tokenize_color_map(x, plane, block, &t, plane_bsize, tx_size, MRC_MAP);
|
||||
#endif // CONFIG_MRC_TX && SIGNAL_ANY_MRC_MASK
|
||||
|
||||
if (eob == 0)
|
||||
add_token(&t, &coef_tail_cdfs[band[c]][pt], &coef_head_cdfs[band[c]][pt], 1,
|
||||
1, 0, BLOCK_Z_TOKEN);
|
||||
|
||||
++blockz_count[pt][eob != 0];
|
||||
|
||||
while (c < eob) {
|
||||
int v = qcoeff[scan[c]];
|
||||
first_val = (c == 0);
|
||||
|
|
@ -483,23 +555,13 @@ static void tokenize_b(int plane, int block, int blk_row, int blk_col,
|
|||
if (!v) {
|
||||
add_token(&t, &coef_tail_cdfs[band[c]][pt], &coef_head_cdfs[band[c]][pt],
|
||||
0, first_val, 0, ZERO_TOKEN);
|
||||
++counts[band[c]][pt][ZERO_TOKEN];
|
||||
token_cache[scan[c]] = 0;
|
||||
} else {
|
||||
eob_val =
|
||||
(c + 1 == eob) ? (c + 1 == seg_eob ? LAST_EOB : EARLY_EOB) : NO_EOB;
|
||||
|
||||
av1_get_token_extra(v, &token, &extra);
|
||||
|
||||
add_token(&t, &coef_tail_cdfs[band[c]][pt], &coef_head_cdfs[band[c]][pt],
|
||||
eob_val, first_val, extra, (uint8_t)token);
|
||||
|
||||
if (eob_val != LAST_EOB) {
|
||||
++counts[band[c]][pt][token];
|
||||
++eob_branch[band[c]][pt];
|
||||
counts[band[c]][pt][EOB_TOKEN] += eob_val != NO_EOB;
|
||||
}
|
||||
|
||||
token_cache[scan[c]] = av1_pt_energy_class[token];
|
||||
}
|
||||
++c;
|
||||
|
|
@ -673,7 +735,7 @@ void av1_tokenize_sb_vartx(const AV1_COMP *cpi, ThreadData *td, TOKENEXTRA **t,
|
|||
if (!is_chroma_reference(mi_row, mi_col, bsize,
|
||||
xd->plane[plane].subsampling_x,
|
||||
xd->plane[plane].subsampling_y)) {
|
||||
#if !CONFIG_PVQ || !CONFIG_LV_MAP
|
||||
#if !CONFIG_PVQ && !CONFIG_LV_MAP
|
||||
if (!dry_run) {
|
||||
(*t)->token = EOSB_TOKEN;
|
||||
(*t)++;
|
||||
|
|
@ -691,7 +753,8 @@ void av1_tokenize_sb_vartx(const AV1_COMP *cpi, ThreadData *td, TOKENEXTRA **t,
|
|||
#endif
|
||||
const int mi_width = block_size_wide[plane_bsize] >> tx_size_wide_log2[0];
|
||||
const int mi_height = block_size_high[plane_bsize] >> tx_size_wide_log2[0];
|
||||
const TX_SIZE max_tx_size = get_vartx_max_txsize(mbmi, plane_bsize);
|
||||
const TX_SIZE max_tx_size = get_vartx_max_txsize(
|
||||
mbmi, plane_bsize, pd->subsampling_x || pd->subsampling_y);
|
||||
const BLOCK_SIZE txb_size = txsize_to_bsize[max_tx_size];
|
||||
int bw = block_size_wide[txb_size] >> tx_size_wide_log2[0];
|
||||
int bh = block_size_high[txb_size] >> tx_size_wide_log2[0];
|
||||
|
|
|
|||
25
third_party/aom/av1/encoder/tokenize.h
vendored
25
third_party/aom/av1/encoder/tokenize.h
vendored
|
|
@ -37,15 +37,12 @@ typedef struct {
|
|||
typedef struct {
|
||||
aom_cdf_prob (*tail_cdf)[CDF_SIZE(ENTROPY_TOKENS)];
|
||||
aom_cdf_prob (*head_cdf)[CDF_SIZE(ENTROPY_TOKENS)];
|
||||
#if CONFIG_PALETTE
|
||||
aom_cdf_prob *palette_cdf;
|
||||
#endif // CONFIG_PALETTE
|
||||
aom_cdf_prob *color_map_cdf;
|
||||
int eob_val;
|
||||
int first_val;
|
||||
const aom_prob *context_tree;
|
||||
EXTRABIT extra;
|
||||
uint8_t token;
|
||||
uint8_t skip_eob_node;
|
||||
} TOKENEXTRA;
|
||||
|
||||
extern const aom_tree_index av1_coef_tree[];
|
||||
|
|
@ -77,12 +74,14 @@ void av1_tokenize_sb_vartx(const struct AV1_COMP *cpi, struct ThreadData *td,
|
|||
TOKENEXTRA **t, RUN_TYPE dry_run, int mi_row,
|
||||
int mi_col, BLOCK_SIZE bsize, int *rate);
|
||||
#endif
|
||||
#if CONFIG_PALETTE
|
||||
void av1_tokenize_palette_sb(const struct AV1_COMP *cpi,
|
||||
const struct ThreadData *const td, int plane,
|
||||
TOKENEXTRA **t, RUN_TYPE dry_run, BLOCK_SIZE bsize,
|
||||
int *rate);
|
||||
#endif // CONFIG_PALETTE
|
||||
|
||||
int av1_cost_color_map(const MACROBLOCK *const x, int plane, int block,
|
||||
BLOCK_SIZE bsize, TX_SIZE tx_size, COLOR_MAP_TYPE type);
|
||||
|
||||
void av1_tokenize_color_map(const MACROBLOCK *const x, int plane, int block,
|
||||
TOKENEXTRA **t, BLOCK_SIZE bsize, TX_SIZE tx_size,
|
||||
COLOR_MAP_TYPE type);
|
||||
|
||||
void av1_tokenize_sb(const struct AV1_COMP *cpi, struct ThreadData *td,
|
||||
TOKENEXTRA **t, RUN_TYPE dry_run, BLOCK_SIZE bsize,
|
||||
int *rate, const int mi_row, const int mi_col);
|
||||
|
|
@ -139,13 +138,11 @@ static INLINE int av1_get_token_cost(int v, int16_t *token, int cat6_bits) {
|
|||
return av1_dct_cat_lt_10_value_cost[v];
|
||||
}
|
||||
|
||||
#if !CONFIG_PVQ || CONFIG_VAR_TX
|
||||
static INLINE int get_tx_eob(const struct segmentation *seg, int segment_id,
|
||||
TX_SIZE tx_size) {
|
||||
static INLINE int av1_get_tx_eob(const struct segmentation *seg, int segment_id,
|
||||
TX_SIZE tx_size) {
|
||||
const int eob_max = tx_size_2d[tx_size];
|
||||
return segfeature_active(seg, segment_id, SEG_LVL_SKIP) ? 0 : eob_max;
|
||||
}
|
||||
#endif
|
||||
|
||||
#ifdef __cplusplus
|
||||
} // extern "C"
|
||||
|
|
|
|||
|
|
@ -16,24 +16,24 @@
|
|||
#include "aom_dsp/aom_dsp_common.h"
|
||||
|
||||
static INLINE void read_coeff(const tran_low_t *coeff, __m256i *c) {
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
const __m256i x0 = _mm256_loadu_si256((const __m256i *)coeff);
|
||||
const __m256i x1 = _mm256_loadu_si256((const __m256i *)coeff + 1);
|
||||
*c = _mm256_packs_epi32(x0, x1);
|
||||
*c = _mm256_permute4x64_epi64(*c, 0xD8);
|
||||
#else
|
||||
*c = _mm256_loadu_si256((const __m256i *)coeff);
|
||||
#endif
|
||||
if (sizeof(tran_low_t) == 4) {
|
||||
const __m256i x0 = _mm256_loadu_si256((const __m256i *)coeff);
|
||||
const __m256i x1 = _mm256_loadu_si256((const __m256i *)coeff + 1);
|
||||
*c = _mm256_packs_epi32(x0, x1);
|
||||
*c = _mm256_permute4x64_epi64(*c, 0xD8);
|
||||
} else {
|
||||
*c = _mm256_loadu_si256((const __m256i *)coeff);
|
||||
}
|
||||
}
|
||||
|
||||
static INLINE void write_zero(tran_low_t *qcoeff) {
|
||||
const __m256i zero = _mm256_setzero_si256();
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
_mm256_storeu_si256((__m256i *)qcoeff, zero);
|
||||
_mm256_storeu_si256((__m256i *)qcoeff + 1, zero);
|
||||
#else
|
||||
_mm256_storeu_si256((__m256i *)qcoeff, zero);
|
||||
#endif
|
||||
if (sizeof(tran_low_t) == 4) {
|
||||
_mm256_storeu_si256((__m256i *)qcoeff, zero);
|
||||
_mm256_storeu_si256((__m256i *)qcoeff + 1, zero);
|
||||
} else {
|
||||
_mm256_storeu_si256((__m256i *)qcoeff, zero);
|
||||
}
|
||||
}
|
||||
|
||||
static INLINE void init_one_qp(const __m128i *p, __m256i *qp) {
|
||||
|
|
@ -83,19 +83,16 @@ static INLINE void update_qp(int log_scale, __m256i *thr, __m256i *qp) {
|
|||
_mm256_storeu_si256((__m256i *)addr + 1, x1); \
|
||||
} while (0)
|
||||
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
#define store_two_quan(q, addr1, dq, addr2) \
|
||||
do { \
|
||||
store_quan(q, addr1); \
|
||||
store_quan(dq, addr2); \
|
||||
#define store_two_quan(q, addr1, dq, addr2) \
|
||||
do { \
|
||||
if (sizeof(tran_low_t) == 4) { \
|
||||
store_quan(q, addr1); \
|
||||
store_quan(dq, addr2); \
|
||||
} else { \
|
||||
_mm256_storeu_si256((__m256i *)addr1, q); \
|
||||
_mm256_storeu_si256((__m256i *)addr2, dq); \
|
||||
} \
|
||||
} while (0)
|
||||
#else
|
||||
#define store_two_quan(q, addr1, dq, addr2) \
|
||||
do { \
|
||||
_mm256_storeu_si256((__m256i *)addr1, q); \
|
||||
_mm256_storeu_si256((__m256i *)addr2, dq); \
|
||||
} while (0)
|
||||
#endif
|
||||
|
||||
static INLINE void quantize(const __m256i *thr, const __m256i *qp, __m256i *c,
|
||||
const int16_t *iscan_ptr, tran_low_t *qcoeff,
|
||||
|
|
|
|||
|
|
@ -18,53 +18,53 @@
|
|||
static INLINE void read_coeff(const tran_low_t *coeff, intptr_t offset,
|
||||
__m128i *c0, __m128i *c1) {
|
||||
const tran_low_t *addr = coeff + offset;
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
const __m128i x0 = _mm_load_si128((const __m128i *)addr);
|
||||
const __m128i x1 = _mm_load_si128((const __m128i *)addr + 1);
|
||||
const __m128i x2 = _mm_load_si128((const __m128i *)addr + 2);
|
||||
const __m128i x3 = _mm_load_si128((const __m128i *)addr + 3);
|
||||
*c0 = _mm_packs_epi32(x0, x1);
|
||||
*c1 = _mm_packs_epi32(x2, x3);
|
||||
#else
|
||||
*c0 = _mm_load_si128((const __m128i *)addr);
|
||||
*c1 = _mm_load_si128((const __m128i *)addr + 1);
|
||||
#endif
|
||||
if (sizeof(tran_low_t) == 4) {
|
||||
const __m128i x0 = _mm_load_si128((const __m128i *)addr);
|
||||
const __m128i x1 = _mm_load_si128((const __m128i *)addr + 1);
|
||||
const __m128i x2 = _mm_load_si128((const __m128i *)addr + 2);
|
||||
const __m128i x3 = _mm_load_si128((const __m128i *)addr + 3);
|
||||
*c0 = _mm_packs_epi32(x0, x1);
|
||||
*c1 = _mm_packs_epi32(x2, x3);
|
||||
} else {
|
||||
*c0 = _mm_load_si128((const __m128i *)addr);
|
||||
*c1 = _mm_load_si128((const __m128i *)addr + 1);
|
||||
}
|
||||
}
|
||||
|
||||
static INLINE void write_qcoeff(const __m128i *qc0, const __m128i *qc1,
|
||||
tran_low_t *qcoeff, intptr_t offset) {
|
||||
tran_low_t *addr = qcoeff + offset;
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
const __m128i zero = _mm_setzero_si128();
|
||||
__m128i sign_bits = _mm_cmplt_epi16(*qc0, zero);
|
||||
__m128i y0 = _mm_unpacklo_epi16(*qc0, sign_bits);
|
||||
__m128i y1 = _mm_unpackhi_epi16(*qc0, sign_bits);
|
||||
_mm_store_si128((__m128i *)addr, y0);
|
||||
_mm_store_si128((__m128i *)addr + 1, y1);
|
||||
if (sizeof(tran_low_t) == 4) {
|
||||
const __m128i zero = _mm_setzero_si128();
|
||||
__m128i sign_bits = _mm_cmplt_epi16(*qc0, zero);
|
||||
__m128i y0 = _mm_unpacklo_epi16(*qc0, sign_bits);
|
||||
__m128i y1 = _mm_unpackhi_epi16(*qc0, sign_bits);
|
||||
_mm_store_si128((__m128i *)addr, y0);
|
||||
_mm_store_si128((__m128i *)addr + 1, y1);
|
||||
|
||||
sign_bits = _mm_cmplt_epi16(*qc1, zero);
|
||||
y0 = _mm_unpacklo_epi16(*qc1, sign_bits);
|
||||
y1 = _mm_unpackhi_epi16(*qc1, sign_bits);
|
||||
_mm_store_si128((__m128i *)addr + 2, y0);
|
||||
_mm_store_si128((__m128i *)addr + 3, y1);
|
||||
#else
|
||||
_mm_store_si128((__m128i *)addr, *qc0);
|
||||
_mm_store_si128((__m128i *)addr + 1, *qc1);
|
||||
#endif
|
||||
sign_bits = _mm_cmplt_epi16(*qc1, zero);
|
||||
y0 = _mm_unpacklo_epi16(*qc1, sign_bits);
|
||||
y1 = _mm_unpackhi_epi16(*qc1, sign_bits);
|
||||
_mm_store_si128((__m128i *)addr + 2, y0);
|
||||
_mm_store_si128((__m128i *)addr + 3, y1);
|
||||
} else {
|
||||
_mm_store_si128((__m128i *)addr, *qc0);
|
||||
_mm_store_si128((__m128i *)addr + 1, *qc1);
|
||||
}
|
||||
}
|
||||
|
||||
static INLINE void write_zero(tran_low_t *qcoeff, intptr_t offset) {
|
||||
const __m128i zero = _mm_setzero_si128();
|
||||
tran_low_t *addr = qcoeff + offset;
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
_mm_store_si128((__m128i *)addr, zero);
|
||||
_mm_store_si128((__m128i *)addr + 1, zero);
|
||||
_mm_store_si128((__m128i *)addr + 2, zero);
|
||||
_mm_store_si128((__m128i *)addr + 3, zero);
|
||||
#else
|
||||
_mm_store_si128((__m128i *)addr, zero);
|
||||
_mm_store_si128((__m128i *)addr + 1, zero);
|
||||
#endif
|
||||
if (sizeof(tran_low_t) == 4) {
|
||||
_mm_store_si128((__m128i *)addr, zero);
|
||||
_mm_store_si128((__m128i *)addr + 1, zero);
|
||||
_mm_store_si128((__m128i *)addr + 2, zero);
|
||||
_mm_store_si128((__m128i *)addr + 3, zero);
|
||||
} else {
|
||||
_mm_store_si128((__m128i *)addr, zero);
|
||||
_mm_store_si128((__m128i *)addr + 1, zero);
|
||||
}
|
||||
}
|
||||
|
||||
void av1_quantize_fp_sse2(const tran_low_t *coeff_ptr, intptr_t n_coeffs,
|
||||
|
|
|
|||
461
third_party/aom/av1/encoder/x86/dct_intrin_sse2.c
vendored
461
third_party/aom/av1/encoder/x86/dct_intrin_sse2.c
vendored
|
|
@ -205,7 +205,7 @@ static void fidtx4_sse2(__m128i *in) {
|
|||
void av1_fht4x4_sse2(const int16_t *input, tran_low_t *output, int stride,
|
||||
TxfmParam *txfm_param) {
|
||||
__m128i in[4];
|
||||
int tx_type = txfm_param->tx_type;
|
||||
const TX_TYPE tx_type = txfm_param->tx_type;
|
||||
#if CONFIG_MRC_TX
|
||||
assert(tx_type != MRC_DCT && "Invalid tx type for tx size");
|
||||
#endif
|
||||
|
|
@ -308,447 +308,6 @@ void av1_fht4x4_sse2(const int16_t *input, tran_low_t *output, int stride,
|
|||
}
|
||||
}
|
||||
|
||||
void av1_fdct8x8_quant_sse2(const int16_t *input, int stride,
|
||||
int16_t *coeff_ptr, intptr_t n_coeffs,
|
||||
int skip_block, const int16_t *zbin_ptr,
|
||||
const int16_t *round_ptr, const int16_t *quant_ptr,
|
||||
const int16_t *quant_shift_ptr, int16_t *qcoeff_ptr,
|
||||
int16_t *dqcoeff_ptr, const int16_t *dequant_ptr,
|
||||
uint16_t *eob_ptr, const int16_t *scan_ptr,
|
||||
const int16_t *iscan_ptr) {
|
||||
__m128i zero;
|
||||
int pass;
|
||||
// Constants
|
||||
// When we use them, in one case, they are all the same. In all others
|
||||
// it's a pair of them that we need to repeat four times. This is done
|
||||
// by constructing the 32 bit constant corresponding to that pair.
|
||||
const __m128i k__cospi_p16_p16 = _mm_set1_epi16((int16_t)cospi_16_64);
|
||||
const __m128i k__cospi_p16_m16 = pair_set_epi16(cospi_16_64, -cospi_16_64);
|
||||
const __m128i k__cospi_p24_p08 = pair_set_epi16(cospi_24_64, cospi_8_64);
|
||||
const __m128i k__cospi_m08_p24 = pair_set_epi16(-cospi_8_64, cospi_24_64);
|
||||
const __m128i k__cospi_p28_p04 = pair_set_epi16(cospi_28_64, cospi_4_64);
|
||||
const __m128i k__cospi_m04_p28 = pair_set_epi16(-cospi_4_64, cospi_28_64);
|
||||
const __m128i k__cospi_p12_p20 = pair_set_epi16(cospi_12_64, cospi_20_64);
|
||||
const __m128i k__cospi_m20_p12 = pair_set_epi16(-cospi_20_64, cospi_12_64);
|
||||
const __m128i k__DCT_CONST_ROUNDING = _mm_set1_epi32(DCT_CONST_ROUNDING);
|
||||
// Load input
|
||||
__m128i in0 = _mm_load_si128((const __m128i *)(input + 0 * stride));
|
||||
__m128i in1 = _mm_load_si128((const __m128i *)(input + 1 * stride));
|
||||
__m128i in2 = _mm_load_si128((const __m128i *)(input + 2 * stride));
|
||||
__m128i in3 = _mm_load_si128((const __m128i *)(input + 3 * stride));
|
||||
__m128i in4 = _mm_load_si128((const __m128i *)(input + 4 * stride));
|
||||
__m128i in5 = _mm_load_si128((const __m128i *)(input + 5 * stride));
|
||||
__m128i in6 = _mm_load_si128((const __m128i *)(input + 6 * stride));
|
||||
__m128i in7 = _mm_load_si128((const __m128i *)(input + 7 * stride));
|
||||
__m128i *in[8];
|
||||
int index = 0;
|
||||
|
||||
(void)scan_ptr;
|
||||
(void)zbin_ptr;
|
||||
(void)quant_shift_ptr;
|
||||
(void)coeff_ptr;
|
||||
|
||||
// Pre-condition input (shift by two)
|
||||
in0 = _mm_slli_epi16(in0, 2);
|
||||
in1 = _mm_slli_epi16(in1, 2);
|
||||
in2 = _mm_slli_epi16(in2, 2);
|
||||
in3 = _mm_slli_epi16(in3, 2);
|
||||
in4 = _mm_slli_epi16(in4, 2);
|
||||
in5 = _mm_slli_epi16(in5, 2);
|
||||
in6 = _mm_slli_epi16(in6, 2);
|
||||
in7 = _mm_slli_epi16(in7, 2);
|
||||
|
||||
in[0] = &in0;
|
||||
in[1] = &in1;
|
||||
in[2] = &in2;
|
||||
in[3] = &in3;
|
||||
in[4] = &in4;
|
||||
in[5] = &in5;
|
||||
in[6] = &in6;
|
||||
in[7] = &in7;
|
||||
|
||||
// We do two passes, first the columns, then the rows. The results of the
|
||||
// first pass are transposed so that the same column code can be reused. The
|
||||
// results of the second pass are also transposed so that the rows (processed
|
||||
// as columns) are put back in row positions.
|
||||
for (pass = 0; pass < 2; pass++) {
|
||||
// To store results of each pass before the transpose.
|
||||
__m128i res0, res1, res2, res3, res4, res5, res6, res7;
|
||||
// Add/subtract
|
||||
const __m128i q0 = _mm_add_epi16(in0, in7);
|
||||
const __m128i q1 = _mm_add_epi16(in1, in6);
|
||||
const __m128i q2 = _mm_add_epi16(in2, in5);
|
||||
const __m128i q3 = _mm_add_epi16(in3, in4);
|
||||
const __m128i q4 = _mm_sub_epi16(in3, in4);
|
||||
const __m128i q5 = _mm_sub_epi16(in2, in5);
|
||||
const __m128i q6 = _mm_sub_epi16(in1, in6);
|
||||
const __m128i q7 = _mm_sub_epi16(in0, in7);
|
||||
// Work on first four results
|
||||
{
|
||||
// Add/subtract
|
||||
const __m128i r0 = _mm_add_epi16(q0, q3);
|
||||
const __m128i r1 = _mm_add_epi16(q1, q2);
|
||||
const __m128i r2 = _mm_sub_epi16(q1, q2);
|
||||
const __m128i r3 = _mm_sub_epi16(q0, q3);
|
||||
// Interleave to do the multiply by constants which gets us into 32bits
|
||||
const __m128i t0 = _mm_unpacklo_epi16(r0, r1);
|
||||
const __m128i t1 = _mm_unpackhi_epi16(r0, r1);
|
||||
const __m128i t2 = _mm_unpacklo_epi16(r2, r3);
|
||||
const __m128i t3 = _mm_unpackhi_epi16(r2, r3);
|
||||
const __m128i u0 = _mm_madd_epi16(t0, k__cospi_p16_p16);
|
||||
const __m128i u1 = _mm_madd_epi16(t1, k__cospi_p16_p16);
|
||||
const __m128i u2 = _mm_madd_epi16(t0, k__cospi_p16_m16);
|
||||
const __m128i u3 = _mm_madd_epi16(t1, k__cospi_p16_m16);
|
||||
const __m128i u4 = _mm_madd_epi16(t2, k__cospi_p24_p08);
|
||||
const __m128i u5 = _mm_madd_epi16(t3, k__cospi_p24_p08);
|
||||
const __m128i u6 = _mm_madd_epi16(t2, k__cospi_m08_p24);
|
||||
const __m128i u7 = _mm_madd_epi16(t3, k__cospi_m08_p24);
|
||||
// dct_const_round_shift
|
||||
const __m128i v0 = _mm_add_epi32(u0, k__DCT_CONST_ROUNDING);
|
||||
const __m128i v1 = _mm_add_epi32(u1, k__DCT_CONST_ROUNDING);
|
||||
const __m128i v2 = _mm_add_epi32(u2, k__DCT_CONST_ROUNDING);
|
||||
const __m128i v3 = _mm_add_epi32(u3, k__DCT_CONST_ROUNDING);
|
||||
const __m128i v4 = _mm_add_epi32(u4, k__DCT_CONST_ROUNDING);
|
||||
const __m128i v5 = _mm_add_epi32(u5, k__DCT_CONST_ROUNDING);
|
||||
const __m128i v6 = _mm_add_epi32(u6, k__DCT_CONST_ROUNDING);
|
||||
const __m128i v7 = _mm_add_epi32(u7, k__DCT_CONST_ROUNDING);
|
||||
const __m128i w0 = _mm_srai_epi32(v0, DCT_CONST_BITS);
|
||||
const __m128i w1 = _mm_srai_epi32(v1, DCT_CONST_BITS);
|
||||
const __m128i w2 = _mm_srai_epi32(v2, DCT_CONST_BITS);
|
||||
const __m128i w3 = _mm_srai_epi32(v3, DCT_CONST_BITS);
|
||||
const __m128i w4 = _mm_srai_epi32(v4, DCT_CONST_BITS);
|
||||
const __m128i w5 = _mm_srai_epi32(v5, DCT_CONST_BITS);
|
||||
const __m128i w6 = _mm_srai_epi32(v6, DCT_CONST_BITS);
|
||||
const __m128i w7 = _mm_srai_epi32(v7, DCT_CONST_BITS);
|
||||
// Combine
|
||||
res0 = _mm_packs_epi32(w0, w1);
|
||||
res4 = _mm_packs_epi32(w2, w3);
|
||||
res2 = _mm_packs_epi32(w4, w5);
|
||||
res6 = _mm_packs_epi32(w6, w7);
|
||||
}
|
||||
// Work on next four results
|
||||
{
|
||||
// Interleave to do the multiply by constants which gets us into 32bits
|
||||
const __m128i d0 = _mm_unpacklo_epi16(q6, q5);
|
||||
const __m128i d1 = _mm_unpackhi_epi16(q6, q5);
|
||||
const __m128i e0 = _mm_madd_epi16(d0, k__cospi_p16_m16);
|
||||
const __m128i e1 = _mm_madd_epi16(d1, k__cospi_p16_m16);
|
||||
const __m128i e2 = _mm_madd_epi16(d0, k__cospi_p16_p16);
|
||||
const __m128i e3 = _mm_madd_epi16(d1, k__cospi_p16_p16);
|
||||
// dct_const_round_shift
|
||||
const __m128i f0 = _mm_add_epi32(e0, k__DCT_CONST_ROUNDING);
|
||||
const __m128i f1 = _mm_add_epi32(e1, k__DCT_CONST_ROUNDING);
|
||||
const __m128i f2 = _mm_add_epi32(e2, k__DCT_CONST_ROUNDING);
|
||||
const __m128i f3 = _mm_add_epi32(e3, k__DCT_CONST_ROUNDING);
|
||||
const __m128i s0 = _mm_srai_epi32(f0, DCT_CONST_BITS);
|
||||
const __m128i s1 = _mm_srai_epi32(f1, DCT_CONST_BITS);
|
||||
const __m128i s2 = _mm_srai_epi32(f2, DCT_CONST_BITS);
|
||||
const __m128i s3 = _mm_srai_epi32(f3, DCT_CONST_BITS);
|
||||
// Combine
|
||||
const __m128i r0 = _mm_packs_epi32(s0, s1);
|
||||
const __m128i r1 = _mm_packs_epi32(s2, s3);
|
||||
// Add/subtract
|
||||
const __m128i x0 = _mm_add_epi16(q4, r0);
|
||||
const __m128i x1 = _mm_sub_epi16(q4, r0);
|
||||
const __m128i x2 = _mm_sub_epi16(q7, r1);
|
||||
const __m128i x3 = _mm_add_epi16(q7, r1);
|
||||
// Interleave to do the multiply by constants which gets us into 32bits
|
||||
const __m128i t0 = _mm_unpacklo_epi16(x0, x3);
|
||||
const __m128i t1 = _mm_unpackhi_epi16(x0, x3);
|
||||
const __m128i t2 = _mm_unpacklo_epi16(x1, x2);
|
||||
const __m128i t3 = _mm_unpackhi_epi16(x1, x2);
|
||||
const __m128i u0 = _mm_madd_epi16(t0, k__cospi_p28_p04);
|
||||
const __m128i u1 = _mm_madd_epi16(t1, k__cospi_p28_p04);
|
||||
const __m128i u2 = _mm_madd_epi16(t0, k__cospi_m04_p28);
|
||||
const __m128i u3 = _mm_madd_epi16(t1, k__cospi_m04_p28);
|
||||
const __m128i u4 = _mm_madd_epi16(t2, k__cospi_p12_p20);
|
||||
const __m128i u5 = _mm_madd_epi16(t3, k__cospi_p12_p20);
|
||||
const __m128i u6 = _mm_madd_epi16(t2, k__cospi_m20_p12);
|
||||
const __m128i u7 = _mm_madd_epi16(t3, k__cospi_m20_p12);
|
||||
// dct_const_round_shift
|
||||
const __m128i v0 = _mm_add_epi32(u0, k__DCT_CONST_ROUNDING);
|
||||
const __m128i v1 = _mm_add_epi32(u1, k__DCT_CONST_ROUNDING);
|
||||
const __m128i v2 = _mm_add_epi32(u2, k__DCT_CONST_ROUNDING);
|
||||
const __m128i v3 = _mm_add_epi32(u3, k__DCT_CONST_ROUNDING);
|
||||
const __m128i v4 = _mm_add_epi32(u4, k__DCT_CONST_ROUNDING);
|
||||
const __m128i v5 = _mm_add_epi32(u5, k__DCT_CONST_ROUNDING);
|
||||
const __m128i v6 = _mm_add_epi32(u6, k__DCT_CONST_ROUNDING);
|
||||
const __m128i v7 = _mm_add_epi32(u7, k__DCT_CONST_ROUNDING);
|
||||
const __m128i w0 = _mm_srai_epi32(v0, DCT_CONST_BITS);
|
||||
const __m128i w1 = _mm_srai_epi32(v1, DCT_CONST_BITS);
|
||||
const __m128i w2 = _mm_srai_epi32(v2, DCT_CONST_BITS);
|
||||
const __m128i w3 = _mm_srai_epi32(v3, DCT_CONST_BITS);
|
||||
const __m128i w4 = _mm_srai_epi32(v4, DCT_CONST_BITS);
|
||||
const __m128i w5 = _mm_srai_epi32(v5, DCT_CONST_BITS);
|
||||
const __m128i w6 = _mm_srai_epi32(v6, DCT_CONST_BITS);
|
||||
const __m128i w7 = _mm_srai_epi32(v7, DCT_CONST_BITS);
|
||||
// Combine
|
||||
res1 = _mm_packs_epi32(w0, w1);
|
||||
res7 = _mm_packs_epi32(w2, w3);
|
||||
res5 = _mm_packs_epi32(w4, w5);
|
||||
res3 = _mm_packs_epi32(w6, w7);
|
||||
}
|
||||
// Transpose the 8x8.
|
||||
{
|
||||
// 00 01 02 03 04 05 06 07
|
||||
// 10 11 12 13 14 15 16 17
|
||||
// 20 21 22 23 24 25 26 27
|
||||
// 30 31 32 33 34 35 36 37
|
||||
// 40 41 42 43 44 45 46 47
|
||||
// 50 51 52 53 54 55 56 57
|
||||
// 60 61 62 63 64 65 66 67
|
||||
// 70 71 72 73 74 75 76 77
|
||||
const __m128i tr0_0 = _mm_unpacklo_epi16(res0, res1);
|
||||
const __m128i tr0_1 = _mm_unpacklo_epi16(res2, res3);
|
||||
const __m128i tr0_2 = _mm_unpackhi_epi16(res0, res1);
|
||||
const __m128i tr0_3 = _mm_unpackhi_epi16(res2, res3);
|
||||
const __m128i tr0_4 = _mm_unpacklo_epi16(res4, res5);
|
||||
const __m128i tr0_5 = _mm_unpacklo_epi16(res6, res7);
|
||||
const __m128i tr0_6 = _mm_unpackhi_epi16(res4, res5);
|
||||
const __m128i tr0_7 = _mm_unpackhi_epi16(res6, res7);
|
||||
// 00 10 01 11 02 12 03 13
|
||||
// 20 30 21 31 22 32 23 33
|
||||
// 04 14 05 15 06 16 07 17
|
||||
// 24 34 25 35 26 36 27 37
|
||||
// 40 50 41 51 42 52 43 53
|
||||
// 60 70 61 71 62 72 63 73
|
||||
// 54 54 55 55 56 56 57 57
|
||||
// 64 74 65 75 66 76 67 77
|
||||
const __m128i tr1_0 = _mm_unpacklo_epi32(tr0_0, tr0_1);
|
||||
const __m128i tr1_1 = _mm_unpacklo_epi32(tr0_2, tr0_3);
|
||||
const __m128i tr1_2 = _mm_unpackhi_epi32(tr0_0, tr0_1);
|
||||
const __m128i tr1_3 = _mm_unpackhi_epi32(tr0_2, tr0_3);
|
||||
const __m128i tr1_4 = _mm_unpacklo_epi32(tr0_4, tr0_5);
|
||||
const __m128i tr1_5 = _mm_unpacklo_epi32(tr0_6, tr0_7);
|
||||
const __m128i tr1_6 = _mm_unpackhi_epi32(tr0_4, tr0_5);
|
||||
const __m128i tr1_7 = _mm_unpackhi_epi32(tr0_6, tr0_7);
|
||||
// 00 10 20 30 01 11 21 31
|
||||
// 40 50 60 70 41 51 61 71
|
||||
// 02 12 22 32 03 13 23 33
|
||||
// 42 52 62 72 43 53 63 73
|
||||
// 04 14 24 34 05 15 21 36
|
||||
// 44 54 64 74 45 55 61 76
|
||||
// 06 16 26 36 07 17 27 37
|
||||
// 46 56 66 76 47 57 67 77
|
||||
in0 = _mm_unpacklo_epi64(tr1_0, tr1_4);
|
||||
in1 = _mm_unpackhi_epi64(tr1_0, tr1_4);
|
||||
in2 = _mm_unpacklo_epi64(tr1_2, tr1_6);
|
||||
in3 = _mm_unpackhi_epi64(tr1_2, tr1_6);
|
||||
in4 = _mm_unpacklo_epi64(tr1_1, tr1_5);
|
||||
in5 = _mm_unpackhi_epi64(tr1_1, tr1_5);
|
||||
in6 = _mm_unpacklo_epi64(tr1_3, tr1_7);
|
||||
in7 = _mm_unpackhi_epi64(tr1_3, tr1_7);
|
||||
// 00 10 20 30 40 50 60 70
|
||||
// 01 11 21 31 41 51 61 71
|
||||
// 02 12 22 32 42 52 62 72
|
||||
// 03 13 23 33 43 53 63 73
|
||||
// 04 14 24 34 44 54 64 74
|
||||
// 05 15 25 35 45 55 65 75
|
||||
// 06 16 26 36 46 56 66 76
|
||||
// 07 17 27 37 47 57 67 77
|
||||
}
|
||||
}
|
||||
// Post-condition output and store it
|
||||
{
|
||||
// Post-condition (division by two)
|
||||
// division of two 16 bits signed numbers using shifts
|
||||
// n / 2 = (n - (n >> 15)) >> 1
|
||||
const __m128i sign_in0 = _mm_srai_epi16(in0, 15);
|
||||
const __m128i sign_in1 = _mm_srai_epi16(in1, 15);
|
||||
const __m128i sign_in2 = _mm_srai_epi16(in2, 15);
|
||||
const __m128i sign_in3 = _mm_srai_epi16(in3, 15);
|
||||
const __m128i sign_in4 = _mm_srai_epi16(in4, 15);
|
||||
const __m128i sign_in5 = _mm_srai_epi16(in5, 15);
|
||||
const __m128i sign_in6 = _mm_srai_epi16(in6, 15);
|
||||
const __m128i sign_in7 = _mm_srai_epi16(in7, 15);
|
||||
in0 = _mm_sub_epi16(in0, sign_in0);
|
||||
in1 = _mm_sub_epi16(in1, sign_in1);
|
||||
in2 = _mm_sub_epi16(in2, sign_in2);
|
||||
in3 = _mm_sub_epi16(in3, sign_in3);
|
||||
in4 = _mm_sub_epi16(in4, sign_in4);
|
||||
in5 = _mm_sub_epi16(in5, sign_in5);
|
||||
in6 = _mm_sub_epi16(in6, sign_in6);
|
||||
in7 = _mm_sub_epi16(in7, sign_in7);
|
||||
in0 = _mm_srai_epi16(in0, 1);
|
||||
in1 = _mm_srai_epi16(in1, 1);
|
||||
in2 = _mm_srai_epi16(in2, 1);
|
||||
in3 = _mm_srai_epi16(in3, 1);
|
||||
in4 = _mm_srai_epi16(in4, 1);
|
||||
in5 = _mm_srai_epi16(in5, 1);
|
||||
in6 = _mm_srai_epi16(in6, 1);
|
||||
in7 = _mm_srai_epi16(in7, 1);
|
||||
}
|
||||
|
||||
iscan_ptr += n_coeffs;
|
||||
qcoeff_ptr += n_coeffs;
|
||||
dqcoeff_ptr += n_coeffs;
|
||||
n_coeffs = -n_coeffs;
|
||||
zero = _mm_setzero_si128();
|
||||
|
||||
if (!skip_block) {
|
||||
__m128i eob;
|
||||
__m128i round, quant, dequant;
|
||||
{
|
||||
__m128i coeff0, coeff1;
|
||||
|
||||
// Setup global values
|
||||
{
|
||||
round = _mm_load_si128((const __m128i *)round_ptr);
|
||||
quant = _mm_load_si128((const __m128i *)quant_ptr);
|
||||
dequant = _mm_load_si128((const __m128i *)dequant_ptr);
|
||||
}
|
||||
|
||||
{
|
||||
__m128i coeff0_sign, coeff1_sign;
|
||||
__m128i qcoeff0, qcoeff1;
|
||||
__m128i qtmp0, qtmp1;
|
||||
// Do DC and first 15 AC
|
||||
coeff0 = *in[0];
|
||||
coeff1 = *in[1];
|
||||
|
||||
// Poor man's sign extract
|
||||
coeff0_sign = _mm_srai_epi16(coeff0, 15);
|
||||
coeff1_sign = _mm_srai_epi16(coeff1, 15);
|
||||
qcoeff0 = _mm_xor_si128(coeff0, coeff0_sign);
|
||||
qcoeff1 = _mm_xor_si128(coeff1, coeff1_sign);
|
||||
qcoeff0 = _mm_sub_epi16(qcoeff0, coeff0_sign);
|
||||
qcoeff1 = _mm_sub_epi16(qcoeff1, coeff1_sign);
|
||||
|
||||
qcoeff0 = _mm_adds_epi16(qcoeff0, round);
|
||||
round = _mm_unpackhi_epi64(round, round);
|
||||
qcoeff1 = _mm_adds_epi16(qcoeff1, round);
|
||||
qtmp0 = _mm_mulhi_epi16(qcoeff0, quant);
|
||||
quant = _mm_unpackhi_epi64(quant, quant);
|
||||
qtmp1 = _mm_mulhi_epi16(qcoeff1, quant);
|
||||
|
||||
// Reinsert signs
|
||||
qcoeff0 = _mm_xor_si128(qtmp0, coeff0_sign);
|
||||
qcoeff1 = _mm_xor_si128(qtmp1, coeff1_sign);
|
||||
qcoeff0 = _mm_sub_epi16(qcoeff0, coeff0_sign);
|
||||
qcoeff1 = _mm_sub_epi16(qcoeff1, coeff1_sign);
|
||||
|
||||
_mm_store_si128((__m128i *)(qcoeff_ptr + n_coeffs), qcoeff0);
|
||||
_mm_store_si128((__m128i *)(qcoeff_ptr + n_coeffs) + 1, qcoeff1);
|
||||
|
||||
coeff0 = _mm_mullo_epi16(qcoeff0, dequant);
|
||||
dequant = _mm_unpackhi_epi64(dequant, dequant);
|
||||
coeff1 = _mm_mullo_epi16(qcoeff1, dequant);
|
||||
|
||||
_mm_store_si128((__m128i *)(dqcoeff_ptr + n_coeffs), coeff0);
|
||||
_mm_store_si128((__m128i *)(dqcoeff_ptr + n_coeffs) + 1, coeff1);
|
||||
}
|
||||
|
||||
{
|
||||
// Scan for eob
|
||||
__m128i zero_coeff0, zero_coeff1;
|
||||
__m128i nzero_coeff0, nzero_coeff1;
|
||||
__m128i iscan0, iscan1;
|
||||
__m128i eob1;
|
||||
zero_coeff0 = _mm_cmpeq_epi16(coeff0, zero);
|
||||
zero_coeff1 = _mm_cmpeq_epi16(coeff1, zero);
|
||||
nzero_coeff0 = _mm_cmpeq_epi16(zero_coeff0, zero);
|
||||
nzero_coeff1 = _mm_cmpeq_epi16(zero_coeff1, zero);
|
||||
iscan0 = _mm_load_si128((const __m128i *)(iscan_ptr + n_coeffs));
|
||||
iscan1 = _mm_load_si128((const __m128i *)(iscan_ptr + n_coeffs) + 1);
|
||||
// Add one to convert from indices to counts
|
||||
iscan0 = _mm_sub_epi16(iscan0, nzero_coeff0);
|
||||
iscan1 = _mm_sub_epi16(iscan1, nzero_coeff1);
|
||||
eob = _mm_and_si128(iscan0, nzero_coeff0);
|
||||
eob1 = _mm_and_si128(iscan1, nzero_coeff1);
|
||||
eob = _mm_max_epi16(eob, eob1);
|
||||
}
|
||||
n_coeffs += 8 * 2;
|
||||
}
|
||||
|
||||
// AC only loop
|
||||
index = 2;
|
||||
while (n_coeffs < 0) {
|
||||
__m128i coeff0, coeff1;
|
||||
{
|
||||
__m128i coeff0_sign, coeff1_sign;
|
||||
__m128i qcoeff0, qcoeff1;
|
||||
__m128i qtmp0, qtmp1;
|
||||
|
||||
assert(index < (int)(sizeof(in) / sizeof(in[0])) - 1);
|
||||
coeff0 = *in[index];
|
||||
coeff1 = *in[index + 1];
|
||||
|
||||
// Poor man's sign extract
|
||||
coeff0_sign = _mm_srai_epi16(coeff0, 15);
|
||||
coeff1_sign = _mm_srai_epi16(coeff1, 15);
|
||||
qcoeff0 = _mm_xor_si128(coeff0, coeff0_sign);
|
||||
qcoeff1 = _mm_xor_si128(coeff1, coeff1_sign);
|
||||
qcoeff0 = _mm_sub_epi16(qcoeff0, coeff0_sign);
|
||||
qcoeff1 = _mm_sub_epi16(qcoeff1, coeff1_sign);
|
||||
|
||||
qcoeff0 = _mm_adds_epi16(qcoeff0, round);
|
||||
qcoeff1 = _mm_adds_epi16(qcoeff1, round);
|
||||
qtmp0 = _mm_mulhi_epi16(qcoeff0, quant);
|
||||
qtmp1 = _mm_mulhi_epi16(qcoeff1, quant);
|
||||
|
||||
// Reinsert signs
|
||||
qcoeff0 = _mm_xor_si128(qtmp0, coeff0_sign);
|
||||
qcoeff1 = _mm_xor_si128(qtmp1, coeff1_sign);
|
||||
qcoeff0 = _mm_sub_epi16(qcoeff0, coeff0_sign);
|
||||
qcoeff1 = _mm_sub_epi16(qcoeff1, coeff1_sign);
|
||||
|
||||
_mm_store_si128((__m128i *)(qcoeff_ptr + n_coeffs), qcoeff0);
|
||||
_mm_store_si128((__m128i *)(qcoeff_ptr + n_coeffs) + 1, qcoeff1);
|
||||
|
||||
coeff0 = _mm_mullo_epi16(qcoeff0, dequant);
|
||||
coeff1 = _mm_mullo_epi16(qcoeff1, dequant);
|
||||
|
||||
_mm_store_si128((__m128i *)(dqcoeff_ptr + n_coeffs), coeff0);
|
||||
_mm_store_si128((__m128i *)(dqcoeff_ptr + n_coeffs) + 1, coeff1);
|
||||
}
|
||||
|
||||
{
|
||||
// Scan for eob
|
||||
__m128i zero_coeff0, zero_coeff1;
|
||||
__m128i nzero_coeff0, nzero_coeff1;
|
||||
__m128i iscan0, iscan1;
|
||||
__m128i eob0, eob1;
|
||||
zero_coeff0 = _mm_cmpeq_epi16(coeff0, zero);
|
||||
zero_coeff1 = _mm_cmpeq_epi16(coeff1, zero);
|
||||
nzero_coeff0 = _mm_cmpeq_epi16(zero_coeff0, zero);
|
||||
nzero_coeff1 = _mm_cmpeq_epi16(zero_coeff1, zero);
|
||||
iscan0 = _mm_load_si128((const __m128i *)(iscan_ptr + n_coeffs));
|
||||
iscan1 = _mm_load_si128((const __m128i *)(iscan_ptr + n_coeffs) + 1);
|
||||
// Add one to convert from indices to counts
|
||||
iscan0 = _mm_sub_epi16(iscan0, nzero_coeff0);
|
||||
iscan1 = _mm_sub_epi16(iscan1, nzero_coeff1);
|
||||
eob0 = _mm_and_si128(iscan0, nzero_coeff0);
|
||||
eob1 = _mm_and_si128(iscan1, nzero_coeff1);
|
||||
eob0 = _mm_max_epi16(eob0, eob1);
|
||||
eob = _mm_max_epi16(eob, eob0);
|
||||
}
|
||||
n_coeffs += 8 * 2;
|
||||
index += 2;
|
||||
}
|
||||
|
||||
// Accumulate EOB
|
||||
{
|
||||
__m128i eob_shuffled;
|
||||
eob_shuffled = _mm_shuffle_epi32(eob, 0xe);
|
||||
eob = _mm_max_epi16(eob, eob_shuffled);
|
||||
eob_shuffled = _mm_shufflelo_epi16(eob, 0xe);
|
||||
eob = _mm_max_epi16(eob, eob_shuffled);
|
||||
eob_shuffled = _mm_shufflelo_epi16(eob, 0x1);
|
||||
eob = _mm_max_epi16(eob, eob_shuffled);
|
||||
*eob_ptr = _mm_extract_epi16(eob, 1);
|
||||
}
|
||||
} else {
|
||||
do {
|
||||
_mm_store_si128((__m128i *)(dqcoeff_ptr + n_coeffs), zero);
|
||||
_mm_store_si128((__m128i *)(dqcoeff_ptr + n_coeffs) + 1, zero);
|
||||
_mm_store_si128((__m128i *)(qcoeff_ptr + n_coeffs), zero);
|
||||
_mm_store_si128((__m128i *)(qcoeff_ptr + n_coeffs) + 1, zero);
|
||||
n_coeffs += 8 * 2;
|
||||
} while (n_coeffs < 0);
|
||||
*eob_ptr = 0;
|
||||
}
|
||||
}
|
||||
|
||||
// load 8x8 array
|
||||
static INLINE void load_buffer_8x8(const int16_t *input, __m128i *in,
|
||||
int stride, int flipud, int fliplr) {
|
||||
|
|
@ -1307,7 +866,7 @@ static void fidtx8_sse2(__m128i *in) {
|
|||
void av1_fht8x8_sse2(const int16_t *input, tran_low_t *output, int stride,
|
||||
TxfmParam *txfm_param) {
|
||||
__m128i in[8];
|
||||
int tx_type = txfm_param->tx_type;
|
||||
const TX_TYPE tx_type = txfm_param->tx_type;
|
||||
#if CONFIG_MRC_TX
|
||||
assert(tx_type != MRC_DCT && "Invalid tx type for tx size");
|
||||
#endif
|
||||
|
|
@ -2344,7 +1903,7 @@ static void fidtx16_sse2(__m128i *in0, __m128i *in1) {
|
|||
void av1_fht16x16_sse2(const int16_t *input, tran_low_t *output, int stride,
|
||||
TxfmParam *txfm_param) {
|
||||
__m128i in0[16], in1[16];
|
||||
int tx_type = txfm_param->tx_type;
|
||||
const TX_TYPE tx_type = txfm_param->tx_type;
|
||||
#if CONFIG_MRC_TX
|
||||
assert(tx_type != MRC_DCT && "Invalid tx type for tx size");
|
||||
#endif
|
||||
|
|
@ -2564,7 +2123,7 @@ static INLINE void write_buffer_4x8(tran_low_t *output, __m128i *res) {
|
|||
void av1_fht4x8_sse2(const int16_t *input, tran_low_t *output, int stride,
|
||||
TxfmParam *txfm_param) {
|
||||
__m128i in[8];
|
||||
int tx_type = txfm_param->tx_type;
|
||||
const TX_TYPE tx_type = txfm_param->tx_type;
|
||||
#if CONFIG_MRC_TX
|
||||
assert(tx_type != MRC_DCT && "Invalid tx type for tx size");
|
||||
#endif
|
||||
|
|
@ -2742,7 +2301,7 @@ static INLINE void write_buffer_8x4(tran_low_t *output, __m128i *res) {
|
|||
void av1_fht8x4_sse2(const int16_t *input, tran_low_t *output, int stride,
|
||||
TxfmParam *txfm_param) {
|
||||
__m128i in[8];
|
||||
int tx_type = txfm_param->tx_type;
|
||||
const TX_TYPE tx_type = txfm_param->tx_type;
|
||||
#if CONFIG_MRC_TX
|
||||
assert(tx_type != MRC_DCT && "Invalid tx type for tx size");
|
||||
#endif
|
||||
|
|
@ -2886,7 +2445,7 @@ static void row_8x16_rounding(__m128i *in, int bits) {
|
|||
void av1_fht8x16_sse2(const int16_t *input, tran_low_t *output, int stride,
|
||||
TxfmParam *txfm_param) {
|
||||
__m128i in[16];
|
||||
int tx_type = txfm_param->tx_type;
|
||||
const TX_TYPE tx_type = txfm_param->tx_type;
|
||||
#if CONFIG_MRC_TX
|
||||
assert(tx_type != MRC_DCT && "Invalid tx type for tx size");
|
||||
#endif
|
||||
|
|
@ -3071,7 +2630,7 @@ static INLINE void load_buffer_16x8(const int16_t *input, __m128i *in,
|
|||
void av1_fht16x8_sse2(const int16_t *input, tran_low_t *output, int stride,
|
||||
TxfmParam *txfm_param) {
|
||||
__m128i in[16];
|
||||
int tx_type = txfm_param->tx_type;
|
||||
const TX_TYPE tx_type = txfm_param->tx_type;
|
||||
#if CONFIG_MRC_TX
|
||||
assert(tx_type != MRC_DCT && "Invalid tx type for tx size");
|
||||
#endif
|
||||
|
|
@ -3385,7 +2944,7 @@ static INLINE void fhalfright32_16col(__m128i *tl, __m128i *tr, __m128i *bl,
|
|||
void av1_fht16x32_sse2(const int16_t *input, tran_low_t *output, int stride,
|
||||
TxfmParam *txfm_param) {
|
||||
__m128i intl[16], intr[16], inbl[16], inbr[16];
|
||||
int tx_type = txfm_param->tx_type;
|
||||
const TX_TYPE tx_type = txfm_param->tx_type;
|
||||
#if CONFIG_MRC_TX
|
||||
assert(tx_type != MRC_DCT && "Invalid tx type for tx size");
|
||||
#endif
|
||||
|
|
@ -3578,7 +3137,7 @@ static INLINE void write_buffer_32x16(tran_low_t *output, __m128i *res0,
|
|||
void av1_fht32x16_sse2(const int16_t *input, tran_low_t *output, int stride,
|
||||
TxfmParam *txfm_param) {
|
||||
__m128i in0[16], in1[16], in2[16], in3[16];
|
||||
int tx_type = txfm_param->tx_type;
|
||||
const TX_TYPE tx_type = txfm_param->tx_type;
|
||||
#if CONFIG_MRC_TX
|
||||
assert(tx_type != MRC_DCT && "Invalid tx type for tx size");
|
||||
#endif
|
||||
|
|
@ -3822,7 +3381,7 @@ static INLINE void write_buffer_32x32(__m128i *in0, __m128i *in1, __m128i *in2,
|
|||
void av1_fht32x32_sse2(const int16_t *input, tran_low_t *output, int stride,
|
||||
TxfmParam *txfm_param) {
|
||||
__m128i in0[32], in1[32], in2[32], in3[32];
|
||||
int tx_type = txfm_param->tx_type;
|
||||
const TX_TYPE tx_type = txfm_param->tx_type;
|
||||
#if CONFIG_MRC_TX
|
||||
assert(tx_type != MRC_DCT && "No 32x32 sse2 MRC_DCT implementation");
|
||||
#endif
|
||||
|
|
|
|||
469
third_party/aom/av1/encoder/x86/dct_ssse3.c
vendored
469
third_party/aom/av1/encoder/x86/dct_ssse3.c
vendored
|
|
@ -1,469 +0,0 @@
|
|||
/*
|
||||
* Copyright (c) 2016, Alliance for Open Media. All rights reserved
|
||||
*
|
||||
* This source code is subject to the terms of the BSD 2 Clause License and
|
||||
* the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
|
||||
* was not distributed with this source code in the LICENSE file, you can
|
||||
* obtain it at www.aomedia.org/license/software. If the Alliance for Open
|
||||
* Media Patent License 1.0 was not distributed with this source code in the
|
||||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#include <assert.h>
|
||||
#if defined(_MSC_VER) && _MSC_VER <= 1500
|
||||
// Need to include math.h before calling tmmintrin.h/intrin.h
|
||||
// in certain versions of MSVS.
|
||||
#include <math.h>
|
||||
#endif
|
||||
#include <tmmintrin.h> // SSSE3
|
||||
|
||||
#include "./av1_rtcd.h"
|
||||
#include "aom_dsp/x86/inv_txfm_sse2.h"
|
||||
#include "aom_dsp/x86/txfm_common_sse2.h"
|
||||
|
||||
void av1_fdct8x8_quant_ssse3(
|
||||
const int16_t *input, int stride, int16_t *coeff_ptr, intptr_t n_coeffs,
|
||||
int skip_block, const int16_t *zbin_ptr, const int16_t *round_ptr,
|
||||
const int16_t *quant_ptr, const int16_t *quant_shift_ptr,
|
||||
int16_t *qcoeff_ptr, int16_t *dqcoeff_ptr, const int16_t *dequant_ptr,
|
||||
uint16_t *eob_ptr, const int16_t *scan_ptr, const int16_t *iscan_ptr) {
|
||||
__m128i zero;
|
||||
int pass;
|
||||
// Constants
|
||||
// When we use them, in one case, they are all the same. In all others
|
||||
// it's a pair of them that we need to repeat four times. This is done
|
||||
// by constructing the 32 bit constant corresponding to that pair.
|
||||
const __m128i k__dual_p16_p16 = dual_set_epi16(23170, 23170);
|
||||
const __m128i k__cospi_p16_p16 = _mm_set1_epi16((int16_t)cospi_16_64);
|
||||
const __m128i k__cospi_p16_m16 = pair_set_epi16(cospi_16_64, -cospi_16_64);
|
||||
const __m128i k__cospi_p24_p08 = pair_set_epi16(cospi_24_64, cospi_8_64);
|
||||
const __m128i k__cospi_m08_p24 = pair_set_epi16(-cospi_8_64, cospi_24_64);
|
||||
const __m128i k__cospi_p28_p04 = pair_set_epi16(cospi_28_64, cospi_4_64);
|
||||
const __m128i k__cospi_m04_p28 = pair_set_epi16(-cospi_4_64, cospi_28_64);
|
||||
const __m128i k__cospi_p12_p20 = pair_set_epi16(cospi_12_64, cospi_20_64);
|
||||
const __m128i k__cospi_m20_p12 = pair_set_epi16(-cospi_20_64, cospi_12_64);
|
||||
const __m128i k__DCT_CONST_ROUNDING = _mm_set1_epi32(DCT_CONST_ROUNDING);
|
||||
// Load input
|
||||
__m128i in0 = _mm_load_si128((const __m128i *)(input + 0 * stride));
|
||||
__m128i in1 = _mm_load_si128((const __m128i *)(input + 1 * stride));
|
||||
__m128i in2 = _mm_load_si128((const __m128i *)(input + 2 * stride));
|
||||
__m128i in3 = _mm_load_si128((const __m128i *)(input + 3 * stride));
|
||||
__m128i in4 = _mm_load_si128((const __m128i *)(input + 4 * stride));
|
||||
__m128i in5 = _mm_load_si128((const __m128i *)(input + 5 * stride));
|
||||
__m128i in6 = _mm_load_si128((const __m128i *)(input + 6 * stride));
|
||||
__m128i in7 = _mm_load_si128((const __m128i *)(input + 7 * stride));
|
||||
__m128i *in[8];
|
||||
int index = 0;
|
||||
|
||||
(void)scan_ptr;
|
||||
(void)zbin_ptr;
|
||||
(void)quant_shift_ptr;
|
||||
(void)coeff_ptr;
|
||||
|
||||
// Pre-condition input (shift by two)
|
||||
in0 = _mm_slli_epi16(in0, 2);
|
||||
in1 = _mm_slli_epi16(in1, 2);
|
||||
in2 = _mm_slli_epi16(in2, 2);
|
||||
in3 = _mm_slli_epi16(in3, 2);
|
||||
in4 = _mm_slli_epi16(in4, 2);
|
||||
in5 = _mm_slli_epi16(in5, 2);
|
||||
in6 = _mm_slli_epi16(in6, 2);
|
||||
in7 = _mm_slli_epi16(in7, 2);
|
||||
|
||||
in[0] = &in0;
|
||||
in[1] = &in1;
|
||||
in[2] = &in2;
|
||||
in[3] = &in3;
|
||||
in[4] = &in4;
|
||||
in[5] = &in5;
|
||||
in[6] = &in6;
|
||||
in[7] = &in7;
|
||||
|
||||
// We do two passes, first the columns, then the rows. The results of the
|
||||
// first pass are transposed so that the same column code can be reused. The
|
||||
// results of the second pass are also transposed so that the rows (processed
|
||||
// as columns) are put back in row positions.
|
||||
for (pass = 0; pass < 2; pass++) {
|
||||
// To store results of each pass before the transpose.
|
||||
__m128i res0, res1, res2, res3, res4, res5, res6, res7;
|
||||
// Add/subtract
|
||||
const __m128i q0 = _mm_add_epi16(in0, in7);
|
||||
const __m128i q1 = _mm_add_epi16(in1, in6);
|
||||
const __m128i q2 = _mm_add_epi16(in2, in5);
|
||||
const __m128i q3 = _mm_add_epi16(in3, in4);
|
||||
const __m128i q4 = _mm_sub_epi16(in3, in4);
|
||||
const __m128i q5 = _mm_sub_epi16(in2, in5);
|
||||
const __m128i q6 = _mm_sub_epi16(in1, in6);
|
||||
const __m128i q7 = _mm_sub_epi16(in0, in7);
|
||||
// Work on first four results
|
||||
{
|
||||
// Add/subtract
|
||||
const __m128i r0 = _mm_add_epi16(q0, q3);
|
||||
const __m128i r1 = _mm_add_epi16(q1, q2);
|
||||
const __m128i r2 = _mm_sub_epi16(q1, q2);
|
||||
const __m128i r3 = _mm_sub_epi16(q0, q3);
|
||||
// Interleave to do the multiply by constants which gets us into 32bits
|
||||
const __m128i t0 = _mm_unpacklo_epi16(r0, r1);
|
||||
const __m128i t1 = _mm_unpackhi_epi16(r0, r1);
|
||||
const __m128i t2 = _mm_unpacklo_epi16(r2, r3);
|
||||
const __m128i t3 = _mm_unpackhi_epi16(r2, r3);
|
||||
|
||||
const __m128i u0 = _mm_madd_epi16(t0, k__cospi_p16_p16);
|
||||
const __m128i u1 = _mm_madd_epi16(t1, k__cospi_p16_p16);
|
||||
const __m128i u2 = _mm_madd_epi16(t0, k__cospi_p16_m16);
|
||||
const __m128i u3 = _mm_madd_epi16(t1, k__cospi_p16_m16);
|
||||
|
||||
const __m128i u4 = _mm_madd_epi16(t2, k__cospi_p24_p08);
|
||||
const __m128i u5 = _mm_madd_epi16(t3, k__cospi_p24_p08);
|
||||
const __m128i u6 = _mm_madd_epi16(t2, k__cospi_m08_p24);
|
||||
const __m128i u7 = _mm_madd_epi16(t3, k__cospi_m08_p24);
|
||||
// dct_const_round_shift
|
||||
|
||||
const __m128i v0 = _mm_add_epi32(u0, k__DCT_CONST_ROUNDING);
|
||||
const __m128i v1 = _mm_add_epi32(u1, k__DCT_CONST_ROUNDING);
|
||||
const __m128i v2 = _mm_add_epi32(u2, k__DCT_CONST_ROUNDING);
|
||||
const __m128i v3 = _mm_add_epi32(u3, k__DCT_CONST_ROUNDING);
|
||||
|
||||
const __m128i v4 = _mm_add_epi32(u4, k__DCT_CONST_ROUNDING);
|
||||
const __m128i v5 = _mm_add_epi32(u5, k__DCT_CONST_ROUNDING);
|
||||
const __m128i v6 = _mm_add_epi32(u6, k__DCT_CONST_ROUNDING);
|
||||
const __m128i v7 = _mm_add_epi32(u7, k__DCT_CONST_ROUNDING);
|
||||
|
||||
const __m128i w0 = _mm_srai_epi32(v0, DCT_CONST_BITS);
|
||||
const __m128i w1 = _mm_srai_epi32(v1, DCT_CONST_BITS);
|
||||
const __m128i w2 = _mm_srai_epi32(v2, DCT_CONST_BITS);
|
||||
const __m128i w3 = _mm_srai_epi32(v3, DCT_CONST_BITS);
|
||||
|
||||
const __m128i w4 = _mm_srai_epi32(v4, DCT_CONST_BITS);
|
||||
const __m128i w5 = _mm_srai_epi32(v5, DCT_CONST_BITS);
|
||||
const __m128i w6 = _mm_srai_epi32(v6, DCT_CONST_BITS);
|
||||
const __m128i w7 = _mm_srai_epi32(v7, DCT_CONST_BITS);
|
||||
// Combine
|
||||
|
||||
res0 = _mm_packs_epi32(w0, w1);
|
||||
res4 = _mm_packs_epi32(w2, w3);
|
||||
res2 = _mm_packs_epi32(w4, w5);
|
||||
res6 = _mm_packs_epi32(w6, w7);
|
||||
}
|
||||
// Work on next four results
|
||||
{
|
||||
// Interleave to do the multiply by constants which gets us into 32bits
|
||||
const __m128i d0 = _mm_sub_epi16(q6, q5);
|
||||
const __m128i d1 = _mm_add_epi16(q6, q5);
|
||||
const __m128i r0 = _mm_mulhrs_epi16(d0, k__dual_p16_p16);
|
||||
const __m128i r1 = _mm_mulhrs_epi16(d1, k__dual_p16_p16);
|
||||
|
||||
// Add/subtract
|
||||
const __m128i x0 = _mm_add_epi16(q4, r0);
|
||||
const __m128i x1 = _mm_sub_epi16(q4, r0);
|
||||
const __m128i x2 = _mm_sub_epi16(q7, r1);
|
||||
const __m128i x3 = _mm_add_epi16(q7, r1);
|
||||
// Interleave to do the multiply by constants which gets us into 32bits
|
||||
const __m128i t0 = _mm_unpacklo_epi16(x0, x3);
|
||||
const __m128i t1 = _mm_unpackhi_epi16(x0, x3);
|
||||
const __m128i t2 = _mm_unpacklo_epi16(x1, x2);
|
||||
const __m128i t3 = _mm_unpackhi_epi16(x1, x2);
|
||||
const __m128i u0 = _mm_madd_epi16(t0, k__cospi_p28_p04);
|
||||
const __m128i u1 = _mm_madd_epi16(t1, k__cospi_p28_p04);
|
||||
const __m128i u2 = _mm_madd_epi16(t0, k__cospi_m04_p28);
|
||||
const __m128i u3 = _mm_madd_epi16(t1, k__cospi_m04_p28);
|
||||
const __m128i u4 = _mm_madd_epi16(t2, k__cospi_p12_p20);
|
||||
const __m128i u5 = _mm_madd_epi16(t3, k__cospi_p12_p20);
|
||||
const __m128i u6 = _mm_madd_epi16(t2, k__cospi_m20_p12);
|
||||
const __m128i u7 = _mm_madd_epi16(t3, k__cospi_m20_p12);
|
||||
// dct_const_round_shift
|
||||
const __m128i v0 = _mm_add_epi32(u0, k__DCT_CONST_ROUNDING);
|
||||
const __m128i v1 = _mm_add_epi32(u1, k__DCT_CONST_ROUNDING);
|
||||
const __m128i v2 = _mm_add_epi32(u2, k__DCT_CONST_ROUNDING);
|
||||
const __m128i v3 = _mm_add_epi32(u3, k__DCT_CONST_ROUNDING);
|
||||
const __m128i v4 = _mm_add_epi32(u4, k__DCT_CONST_ROUNDING);
|
||||
const __m128i v5 = _mm_add_epi32(u5, k__DCT_CONST_ROUNDING);
|
||||
const __m128i v6 = _mm_add_epi32(u6, k__DCT_CONST_ROUNDING);
|
||||
const __m128i v7 = _mm_add_epi32(u7, k__DCT_CONST_ROUNDING);
|
||||
const __m128i w0 = _mm_srai_epi32(v0, DCT_CONST_BITS);
|
||||
const __m128i w1 = _mm_srai_epi32(v1, DCT_CONST_BITS);
|
||||
const __m128i w2 = _mm_srai_epi32(v2, DCT_CONST_BITS);
|
||||
const __m128i w3 = _mm_srai_epi32(v3, DCT_CONST_BITS);
|
||||
const __m128i w4 = _mm_srai_epi32(v4, DCT_CONST_BITS);
|
||||
const __m128i w5 = _mm_srai_epi32(v5, DCT_CONST_BITS);
|
||||
const __m128i w6 = _mm_srai_epi32(v6, DCT_CONST_BITS);
|
||||
const __m128i w7 = _mm_srai_epi32(v7, DCT_CONST_BITS);
|
||||
// Combine
|
||||
res1 = _mm_packs_epi32(w0, w1);
|
||||
res7 = _mm_packs_epi32(w2, w3);
|
||||
res5 = _mm_packs_epi32(w4, w5);
|
||||
res3 = _mm_packs_epi32(w6, w7);
|
||||
}
|
||||
// Transpose the 8x8.
|
||||
{
|
||||
// 00 01 02 03 04 05 06 07
|
||||
// 10 11 12 13 14 15 16 17
|
||||
// 20 21 22 23 24 25 26 27
|
||||
// 30 31 32 33 34 35 36 37
|
||||
// 40 41 42 43 44 45 46 47
|
||||
// 50 51 52 53 54 55 56 57
|
||||
// 60 61 62 63 64 65 66 67
|
||||
// 70 71 72 73 74 75 76 77
|
||||
const __m128i tr0_0 = _mm_unpacklo_epi16(res0, res1);
|
||||
const __m128i tr0_1 = _mm_unpacklo_epi16(res2, res3);
|
||||
const __m128i tr0_2 = _mm_unpackhi_epi16(res0, res1);
|
||||
const __m128i tr0_3 = _mm_unpackhi_epi16(res2, res3);
|
||||
const __m128i tr0_4 = _mm_unpacklo_epi16(res4, res5);
|
||||
const __m128i tr0_5 = _mm_unpacklo_epi16(res6, res7);
|
||||
const __m128i tr0_6 = _mm_unpackhi_epi16(res4, res5);
|
||||
const __m128i tr0_7 = _mm_unpackhi_epi16(res6, res7);
|
||||
// 00 10 01 11 02 12 03 13
|
||||
// 20 30 21 31 22 32 23 33
|
||||
// 04 14 05 15 06 16 07 17
|
||||
// 24 34 25 35 26 36 27 37
|
||||
// 40 50 41 51 42 52 43 53
|
||||
// 60 70 61 71 62 72 63 73
|
||||
// 54 54 55 55 56 56 57 57
|
||||
// 64 74 65 75 66 76 67 77
|
||||
const __m128i tr1_0 = _mm_unpacklo_epi32(tr0_0, tr0_1);
|
||||
const __m128i tr1_1 = _mm_unpacklo_epi32(tr0_2, tr0_3);
|
||||
const __m128i tr1_2 = _mm_unpackhi_epi32(tr0_0, tr0_1);
|
||||
const __m128i tr1_3 = _mm_unpackhi_epi32(tr0_2, tr0_3);
|
||||
const __m128i tr1_4 = _mm_unpacklo_epi32(tr0_4, tr0_5);
|
||||
const __m128i tr1_5 = _mm_unpacklo_epi32(tr0_6, tr0_7);
|
||||
const __m128i tr1_6 = _mm_unpackhi_epi32(tr0_4, tr0_5);
|
||||
const __m128i tr1_7 = _mm_unpackhi_epi32(tr0_6, tr0_7);
|
||||
// 00 10 20 30 01 11 21 31
|
||||
// 40 50 60 70 41 51 61 71
|
||||
// 02 12 22 32 03 13 23 33
|
||||
// 42 52 62 72 43 53 63 73
|
||||
// 04 14 24 34 05 15 21 36
|
||||
// 44 54 64 74 45 55 61 76
|
||||
// 06 16 26 36 07 17 27 37
|
||||
// 46 56 66 76 47 57 67 77
|
||||
in0 = _mm_unpacklo_epi64(tr1_0, tr1_4);
|
||||
in1 = _mm_unpackhi_epi64(tr1_0, tr1_4);
|
||||
in2 = _mm_unpacklo_epi64(tr1_2, tr1_6);
|
||||
in3 = _mm_unpackhi_epi64(tr1_2, tr1_6);
|
||||
in4 = _mm_unpacklo_epi64(tr1_1, tr1_5);
|
||||
in5 = _mm_unpackhi_epi64(tr1_1, tr1_5);
|
||||
in6 = _mm_unpacklo_epi64(tr1_3, tr1_7);
|
||||
in7 = _mm_unpackhi_epi64(tr1_3, tr1_7);
|
||||
// 00 10 20 30 40 50 60 70
|
||||
// 01 11 21 31 41 51 61 71
|
||||
// 02 12 22 32 42 52 62 72
|
||||
// 03 13 23 33 43 53 63 73
|
||||
// 04 14 24 34 44 54 64 74
|
||||
// 05 15 25 35 45 55 65 75
|
||||
// 06 16 26 36 46 56 66 76
|
||||
// 07 17 27 37 47 57 67 77
|
||||
}
|
||||
}
|
||||
// Post-condition output and store it
|
||||
{
|
||||
// Post-condition (division by two)
|
||||
// division of two 16 bits signed numbers using shifts
|
||||
// n / 2 = (n - (n >> 15)) >> 1
|
||||
const __m128i sign_in0 = _mm_srai_epi16(in0, 15);
|
||||
const __m128i sign_in1 = _mm_srai_epi16(in1, 15);
|
||||
const __m128i sign_in2 = _mm_srai_epi16(in2, 15);
|
||||
const __m128i sign_in3 = _mm_srai_epi16(in3, 15);
|
||||
const __m128i sign_in4 = _mm_srai_epi16(in4, 15);
|
||||
const __m128i sign_in5 = _mm_srai_epi16(in5, 15);
|
||||
const __m128i sign_in6 = _mm_srai_epi16(in6, 15);
|
||||
const __m128i sign_in7 = _mm_srai_epi16(in7, 15);
|
||||
in0 = _mm_sub_epi16(in0, sign_in0);
|
||||
in1 = _mm_sub_epi16(in1, sign_in1);
|
||||
in2 = _mm_sub_epi16(in2, sign_in2);
|
||||
in3 = _mm_sub_epi16(in3, sign_in3);
|
||||
in4 = _mm_sub_epi16(in4, sign_in4);
|
||||
in5 = _mm_sub_epi16(in5, sign_in5);
|
||||
in6 = _mm_sub_epi16(in6, sign_in6);
|
||||
in7 = _mm_sub_epi16(in7, sign_in7);
|
||||
in0 = _mm_srai_epi16(in0, 1);
|
||||
in1 = _mm_srai_epi16(in1, 1);
|
||||
in2 = _mm_srai_epi16(in2, 1);
|
||||
in3 = _mm_srai_epi16(in3, 1);
|
||||
in4 = _mm_srai_epi16(in4, 1);
|
||||
in5 = _mm_srai_epi16(in5, 1);
|
||||
in6 = _mm_srai_epi16(in6, 1);
|
||||
in7 = _mm_srai_epi16(in7, 1);
|
||||
}
|
||||
|
||||
iscan_ptr += n_coeffs;
|
||||
qcoeff_ptr += n_coeffs;
|
||||
dqcoeff_ptr += n_coeffs;
|
||||
n_coeffs = -n_coeffs;
|
||||
zero = _mm_setzero_si128();
|
||||
|
||||
if (!skip_block) {
|
||||
__m128i eob;
|
||||
__m128i round, quant, dequant, thr;
|
||||
int16_t nzflag;
|
||||
{
|
||||
__m128i coeff0, coeff1;
|
||||
|
||||
// Setup global values
|
||||
{
|
||||
round = _mm_load_si128((const __m128i *)round_ptr);
|
||||
quant = _mm_load_si128((const __m128i *)quant_ptr);
|
||||
dequant = _mm_load_si128((const __m128i *)dequant_ptr);
|
||||
}
|
||||
|
||||
{
|
||||
__m128i coeff0_sign, coeff1_sign;
|
||||
__m128i qcoeff0, qcoeff1;
|
||||
__m128i qtmp0, qtmp1;
|
||||
// Do DC and first 15 AC
|
||||
coeff0 = *in[0];
|
||||
coeff1 = *in[1];
|
||||
|
||||
// Poor man's sign extract
|
||||
coeff0_sign = _mm_srai_epi16(coeff0, 15);
|
||||
coeff1_sign = _mm_srai_epi16(coeff1, 15);
|
||||
qcoeff0 = _mm_xor_si128(coeff0, coeff0_sign);
|
||||
qcoeff1 = _mm_xor_si128(coeff1, coeff1_sign);
|
||||
qcoeff0 = _mm_sub_epi16(qcoeff0, coeff0_sign);
|
||||
qcoeff1 = _mm_sub_epi16(qcoeff1, coeff1_sign);
|
||||
|
||||
qcoeff0 = _mm_adds_epi16(qcoeff0, round);
|
||||
round = _mm_unpackhi_epi64(round, round);
|
||||
qcoeff1 = _mm_adds_epi16(qcoeff1, round);
|
||||
qtmp0 = _mm_mulhi_epi16(qcoeff0, quant);
|
||||
quant = _mm_unpackhi_epi64(quant, quant);
|
||||
qtmp1 = _mm_mulhi_epi16(qcoeff1, quant);
|
||||
|
||||
// Reinsert signs
|
||||
qcoeff0 = _mm_xor_si128(qtmp0, coeff0_sign);
|
||||
qcoeff1 = _mm_xor_si128(qtmp1, coeff1_sign);
|
||||
qcoeff0 = _mm_sub_epi16(qcoeff0, coeff0_sign);
|
||||
qcoeff1 = _mm_sub_epi16(qcoeff1, coeff1_sign);
|
||||
|
||||
_mm_store_si128((__m128i *)(qcoeff_ptr + n_coeffs), qcoeff0);
|
||||
_mm_store_si128((__m128i *)(qcoeff_ptr + n_coeffs) + 1, qcoeff1);
|
||||
|
||||
coeff0 = _mm_mullo_epi16(qcoeff0, dequant);
|
||||
dequant = _mm_unpackhi_epi64(dequant, dequant);
|
||||
coeff1 = _mm_mullo_epi16(qcoeff1, dequant);
|
||||
|
||||
_mm_store_si128((__m128i *)(dqcoeff_ptr + n_coeffs), coeff0);
|
||||
_mm_store_si128((__m128i *)(dqcoeff_ptr + n_coeffs) + 1, coeff1);
|
||||
}
|
||||
|
||||
{
|
||||
// Scan for eob
|
||||
__m128i zero_coeff0, zero_coeff1;
|
||||
__m128i nzero_coeff0, nzero_coeff1;
|
||||
__m128i iscan0, iscan1;
|
||||
__m128i eob1;
|
||||
zero_coeff0 = _mm_cmpeq_epi16(coeff0, zero);
|
||||
zero_coeff1 = _mm_cmpeq_epi16(coeff1, zero);
|
||||
nzero_coeff0 = _mm_cmpeq_epi16(zero_coeff0, zero);
|
||||
nzero_coeff1 = _mm_cmpeq_epi16(zero_coeff1, zero);
|
||||
iscan0 = _mm_load_si128((const __m128i *)(iscan_ptr + n_coeffs));
|
||||
iscan1 = _mm_load_si128((const __m128i *)(iscan_ptr + n_coeffs) + 1);
|
||||
// Add one to convert from indices to counts
|
||||
iscan0 = _mm_sub_epi16(iscan0, nzero_coeff0);
|
||||
iscan1 = _mm_sub_epi16(iscan1, nzero_coeff1);
|
||||
eob = _mm_and_si128(iscan0, nzero_coeff0);
|
||||
eob1 = _mm_and_si128(iscan1, nzero_coeff1);
|
||||
eob = _mm_max_epi16(eob, eob1);
|
||||
}
|
||||
n_coeffs += 8 * 2;
|
||||
}
|
||||
|
||||
// AC only loop
|
||||
index = 2;
|
||||
thr = _mm_srai_epi16(dequant, 1);
|
||||
while (n_coeffs < 0) {
|
||||
__m128i coeff0, coeff1;
|
||||
{
|
||||
__m128i coeff0_sign, coeff1_sign;
|
||||
__m128i qcoeff0, qcoeff1;
|
||||
__m128i qtmp0, qtmp1;
|
||||
|
||||
assert(index < (int)(sizeof(in) / sizeof(in[0])) - 1);
|
||||
coeff0 = *in[index];
|
||||
coeff1 = *in[index + 1];
|
||||
|
||||
// Poor man's sign extract
|
||||
coeff0_sign = _mm_srai_epi16(coeff0, 15);
|
||||
coeff1_sign = _mm_srai_epi16(coeff1, 15);
|
||||
qcoeff0 = _mm_xor_si128(coeff0, coeff0_sign);
|
||||
qcoeff1 = _mm_xor_si128(coeff1, coeff1_sign);
|
||||
qcoeff0 = _mm_sub_epi16(qcoeff0, coeff0_sign);
|
||||
qcoeff1 = _mm_sub_epi16(qcoeff1, coeff1_sign);
|
||||
|
||||
nzflag = _mm_movemask_epi8(_mm_cmpgt_epi16(qcoeff0, thr)) |
|
||||
_mm_movemask_epi8(_mm_cmpgt_epi16(qcoeff1, thr));
|
||||
|
||||
if (nzflag) {
|
||||
qcoeff0 = _mm_adds_epi16(qcoeff0, round);
|
||||
qcoeff1 = _mm_adds_epi16(qcoeff1, round);
|
||||
qtmp0 = _mm_mulhi_epi16(qcoeff0, quant);
|
||||
qtmp1 = _mm_mulhi_epi16(qcoeff1, quant);
|
||||
|
||||
// Reinsert signs
|
||||
qcoeff0 = _mm_xor_si128(qtmp0, coeff0_sign);
|
||||
qcoeff1 = _mm_xor_si128(qtmp1, coeff1_sign);
|
||||
qcoeff0 = _mm_sub_epi16(qcoeff0, coeff0_sign);
|
||||
qcoeff1 = _mm_sub_epi16(qcoeff1, coeff1_sign);
|
||||
|
||||
_mm_store_si128((__m128i *)(qcoeff_ptr + n_coeffs), qcoeff0);
|
||||
_mm_store_si128((__m128i *)(qcoeff_ptr + n_coeffs) + 1, qcoeff1);
|
||||
|
||||
coeff0 = _mm_mullo_epi16(qcoeff0, dequant);
|
||||
coeff1 = _mm_mullo_epi16(qcoeff1, dequant);
|
||||
|
||||
_mm_store_si128((__m128i *)(dqcoeff_ptr + n_coeffs), coeff0);
|
||||
_mm_store_si128((__m128i *)(dqcoeff_ptr + n_coeffs) + 1, coeff1);
|
||||
} else {
|
||||
_mm_store_si128((__m128i *)(qcoeff_ptr + n_coeffs), zero);
|
||||
_mm_store_si128((__m128i *)(qcoeff_ptr + n_coeffs) + 1, zero);
|
||||
|
||||
_mm_store_si128((__m128i *)(dqcoeff_ptr + n_coeffs), zero);
|
||||
_mm_store_si128((__m128i *)(dqcoeff_ptr + n_coeffs) + 1, zero);
|
||||
}
|
||||
}
|
||||
|
||||
if (nzflag) {
|
||||
// Scan for eob
|
||||
__m128i zero_coeff0, zero_coeff1;
|
||||
__m128i nzero_coeff0, nzero_coeff1;
|
||||
__m128i iscan0, iscan1;
|
||||
__m128i eob0, eob1;
|
||||
zero_coeff0 = _mm_cmpeq_epi16(coeff0, zero);
|
||||
zero_coeff1 = _mm_cmpeq_epi16(coeff1, zero);
|
||||
nzero_coeff0 = _mm_cmpeq_epi16(zero_coeff0, zero);
|
||||
nzero_coeff1 = _mm_cmpeq_epi16(zero_coeff1, zero);
|
||||
iscan0 = _mm_load_si128((const __m128i *)(iscan_ptr + n_coeffs));
|
||||
iscan1 = _mm_load_si128((const __m128i *)(iscan_ptr + n_coeffs) + 1);
|
||||
// Add one to convert from indices to counts
|
||||
iscan0 = _mm_sub_epi16(iscan0, nzero_coeff0);
|
||||
iscan1 = _mm_sub_epi16(iscan1, nzero_coeff1);
|
||||
eob0 = _mm_and_si128(iscan0, nzero_coeff0);
|
||||
eob1 = _mm_and_si128(iscan1, nzero_coeff1);
|
||||
eob0 = _mm_max_epi16(eob0, eob1);
|
||||
eob = _mm_max_epi16(eob, eob0);
|
||||
}
|
||||
n_coeffs += 8 * 2;
|
||||
index += 2;
|
||||
}
|
||||
|
||||
// Accumulate EOB
|
||||
{
|
||||
__m128i eob_shuffled;
|
||||
eob_shuffled = _mm_shuffle_epi32(eob, 0xe);
|
||||
eob = _mm_max_epi16(eob, eob_shuffled);
|
||||
eob_shuffled = _mm_shufflelo_epi16(eob, 0xe);
|
||||
eob = _mm_max_epi16(eob, eob_shuffled);
|
||||
eob_shuffled = _mm_shufflelo_epi16(eob, 0x1);
|
||||
eob = _mm_max_epi16(eob, eob_shuffled);
|
||||
*eob_ptr = _mm_extract_epi16(eob, 1);
|
||||
}
|
||||
} else {
|
||||
do {
|
||||
_mm_store_si128((__m128i *)(dqcoeff_ptr + n_coeffs), zero);
|
||||
_mm_store_si128((__m128i *)(dqcoeff_ptr + n_coeffs) + 1, zero);
|
||||
_mm_store_si128((__m128i *)(qcoeff_ptr + n_coeffs), zero);
|
||||
_mm_store_si128((__m128i *)(qcoeff_ptr + n_coeffs) + 1, zero);
|
||||
n_coeffs += 8 * 2;
|
||||
} while (n_coeffs < 0);
|
||||
*eob_ptr = 0;
|
||||
}
|
||||
}
|
||||
|
|
@ -17,14 +17,15 @@
|
|||
static INLINE void read_coeff(const tran_low_t *coeff, intptr_t offset,
|
||||
__m256i *c) {
|
||||
const tran_low_t *addr = coeff + offset;
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
const __m256i x0 = _mm256_loadu_si256((const __m256i *)addr);
|
||||
const __m256i x1 = _mm256_loadu_si256((const __m256i *)addr + 1);
|
||||
const __m256i y = _mm256_packs_epi32(x0, x1);
|
||||
*c = _mm256_permute4x64_epi64(y, 0xD8);
|
||||
#else
|
||||
*c = _mm256_loadu_si256((const __m256i *)addr);
|
||||
#endif
|
||||
|
||||
if (sizeof(tran_low_t) == 4) {
|
||||
const __m256i x0 = _mm256_loadu_si256((const __m256i *)addr);
|
||||
const __m256i x1 = _mm256_loadu_si256((const __m256i *)addr + 1);
|
||||
const __m256i y = _mm256_packs_epi32(x0, x1);
|
||||
*c = _mm256_permute4x64_epi64(y, 0xD8);
|
||||
} else {
|
||||
*c = _mm256_loadu_si256((const __m256i *)addr);
|
||||
}
|
||||
}
|
||||
|
||||
int64_t av1_block_error_avx2(const tran_low_t *coeff, const tran_low_t *dqcoeff,
|
||||
|
|
|
|||
|
|
@ -195,7 +195,7 @@ static void fadst4x4_sse4_1(__m128i *in, int bit) {
|
|||
}
|
||||
|
||||
void av1_fwd_txfm2d_4x4_sse4_1(const int16_t *input, int32_t *coeff,
|
||||
int input_stride, int tx_type, int bd) {
|
||||
int input_stride, TX_TYPE tx_type, int bd) {
|
||||
__m128i in[4];
|
||||
const TXFM_1D_CFG *row_cfg = NULL;
|
||||
const TXFM_1D_CFG *col_cfg = NULL;
|
||||
|
|
@ -926,7 +926,7 @@ static void fadst8x8_sse4_1(__m128i *in, __m128i *out, int bit) {
|
|||
}
|
||||
|
||||
void av1_fwd_txfm2d_8x8_sse4_1(const int16_t *input, int32_t *coeff, int stride,
|
||||
int tx_type, int bd) {
|
||||
TX_TYPE tx_type, int bd) {
|
||||
__m128i in[16], out[16];
|
||||
const TXFM_1D_CFG *row_cfg = NULL;
|
||||
const TXFM_1D_CFG *col_cfg = NULL;
|
||||
|
|
@ -1800,7 +1800,7 @@ static void write_buffer_16x16(const __m128i *in, int32_t *output) {
|
|||
}
|
||||
|
||||
void av1_fwd_txfm2d_16x16_sse4_1(const int16_t *input, int32_t *coeff,
|
||||
int stride, int tx_type, int bd) {
|
||||
int stride, TX_TYPE tx_type, int bd) {
|
||||
__m128i in[64], out[64];
|
||||
const TXFM_1D_CFG *row_cfg = NULL;
|
||||
const TXFM_1D_CFG *col_cfg = NULL;
|
||||
|
|
|
|||
|
|
@ -916,7 +916,7 @@ static void fidtx16_avx2(__m256i *in) {
|
|||
void av1_fht16x16_avx2(const int16_t *input, tran_low_t *output, int stride,
|
||||
TxfmParam *txfm_param) {
|
||||
__m256i in[16];
|
||||
int tx_type = txfm_param->tx_type;
|
||||
const TX_TYPE tx_type = txfm_param->tx_type;
|
||||
#if CONFIG_MRC_TX
|
||||
assert(tx_type != MRC_DCT && "Invalid tx type for tx size");
|
||||
#endif
|
||||
|
|
@ -1516,7 +1516,7 @@ void av1_fht32x32_avx2(const int16_t *input, tran_low_t *output, int stride,
|
|||
TxfmParam *txfm_param) {
|
||||
__m256i in0[32]; // left 32 columns
|
||||
__m256i in1[32]; // right 32 columns
|
||||
int tx_type = txfm_param->tx_type;
|
||||
const TX_TYPE tx_type = txfm_param->tx_type;
|
||||
#if CONFIG_MRC_TX
|
||||
assert(tx_type != MRC_DCT && "No avx2 32x32 implementation of MRC_DCT");
|
||||
#endif
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue