mirror of
https://repo.dactyloidae.xyz/Dactyloidae/UXP.git
synced 2026-10-09 08:47:31 +09:00
Update aom to slightly newer commit ID
This commit is contained in:
parent
1c8af26369
commit
76f1e6edca
311 changed files with 55292 additions and 33790 deletions
|
|
@ -353,8 +353,8 @@ void av1_cyclic_refresh_check_golden_update(AV1_COMP *const cpi) {
|
|||
// frame because of the camera movement, set this frame as the golden frame.
|
||||
// Use 70% and 5% as the thresholds for golden frame refreshing.
|
||||
// Also, force this frame as a golden update frame if this frame will change
|
||||
// the resolution (resize_pending != 0).
|
||||
if (cpi->resize_pending != 0 ||
|
||||
// the resolution (av1_resize_pending != 0).
|
||||
if (av1_resize_pending(cpi) ||
|
||||
(cnt1 * 10 > (70 * rows * cols) && cnt2 * 20 < cnt1)) {
|
||||
av1_cyclic_refresh_set_golden_update(cpi);
|
||||
rc->frames_till_gf_update_due = rc->baseline_gf_interval;
|
||||
|
|
|
|||
54
third_party/aom/av1/encoder/av1_quantize.c
vendored
54
third_party/aom/av1/encoder/av1_quantize.c
vendored
|
|
@ -1594,50 +1594,48 @@ static int get_qzbin_factor(int q, aom_bit_depth_t bit_depth) {
|
|||
#endif
|
||||
}
|
||||
|
||||
void av1_init_quantizer(AV1_COMP *cpi) {
|
||||
AV1_COMMON *const cm = &cpi->common;
|
||||
QUANTS *const quants = &cpi->quants;
|
||||
void av1_build_quantizer(aom_bit_depth_t bit_depth, int y_dc_delta_q,
|
||||
int uv_dc_delta_q, int uv_ac_delta_q,
|
||||
QUANTS *const quants, Dequants *const deq) {
|
||||
int i, q, quant;
|
||||
#if CONFIG_NEW_QUANT
|
||||
int dq;
|
||||
#endif
|
||||
|
||||
for (q = 0; q < QINDEX_RANGE; q++) {
|
||||
const int qzbin_factor = get_qzbin_factor(q, cm->bit_depth);
|
||||
const int qzbin_factor = get_qzbin_factor(q, bit_depth);
|
||||
const int qrounding_factor = q == 0 ? 64 : 48;
|
||||
|
||||
for (i = 0; i < 2; ++i) {
|
||||
int qrounding_factor_fp = 64;
|
||||
// y
|
||||
quant = i == 0 ? av1_dc_quant(q, cm->y_dc_delta_q, cm->bit_depth)
|
||||
: av1_ac_quant(q, 0, cm->bit_depth);
|
||||
quant = i == 0 ? av1_dc_quant(q, y_dc_delta_q, bit_depth)
|
||||
: av1_ac_quant(q, 0, bit_depth);
|
||||
invert_quant(&quants->y_quant[q][i], &quants->y_quant_shift[q][i], quant);
|
||||
quants->y_quant_fp[q][i] = (1 << 16) / quant;
|
||||
quants->y_round_fp[q][i] = (qrounding_factor_fp * quant) >> 7;
|
||||
quants->y_zbin[q][i] = ROUND_POWER_OF_TWO(qzbin_factor * quant, 7);
|
||||
quants->y_round[q][i] = (qrounding_factor * quant) >> 7;
|
||||
cpi->y_dequant[q][i] = quant;
|
||||
deq->y_dequant[q][i] = quant;
|
||||
|
||||
// uv
|
||||
quant = i == 0 ? av1_dc_quant(q, cm->uv_dc_delta_q, cm->bit_depth)
|
||||
: av1_ac_quant(q, cm->uv_ac_delta_q, cm->bit_depth);
|
||||
quant = i == 0 ? av1_dc_quant(q, uv_dc_delta_q, bit_depth)
|
||||
: av1_ac_quant(q, uv_ac_delta_q, bit_depth);
|
||||
invert_quant(&quants->uv_quant[q][i], &quants->uv_quant_shift[q][i],
|
||||
quant);
|
||||
quants->uv_quant_fp[q][i] = (1 << 16) / quant;
|
||||
quants->uv_round_fp[q][i] = (qrounding_factor_fp * quant) >> 7;
|
||||
quants->uv_zbin[q][i] = ROUND_POWER_OF_TWO(qzbin_factor * quant, 7);
|
||||
quants->uv_round[q][i] = (qrounding_factor * quant) >> 7;
|
||||
cpi->uv_dequant[q][i] = quant;
|
||||
deq->uv_dequant[q][i] = quant;
|
||||
}
|
||||
|
||||
#if CONFIG_NEW_QUANT
|
||||
int dq;
|
||||
for (dq = 0; dq < QUANT_PROFILES; dq++) {
|
||||
for (i = 0; i < COEF_BANDS; i++) {
|
||||
const int y_quant = cpi->y_dequant[q][i != 0];
|
||||
const int uvquant = cpi->uv_dequant[q][i != 0];
|
||||
av1_get_dequant_val_nuq(y_quant, i, cpi->y_dequant_val_nuq[dq][q][i],
|
||||
const int y_quant = deq->y_dequant[q][i != 0];
|
||||
const int uvquant = deq->uv_dequant[q][i != 0];
|
||||
av1_get_dequant_val_nuq(y_quant, i, deq->y_dequant_val_nuq[dq][q][i],
|
||||
quants->y_cuml_bins_nuq[dq][q][i], dq);
|
||||
av1_get_dequant_val_nuq(uvquant, i, cpi->uv_dequant_val_nuq[dq][q][i],
|
||||
av1_get_dequant_val_nuq(uvquant, i, deq->uv_dequant_val_nuq[dq][q][i],
|
||||
quants->uv_cuml_bins_nuq[dq][q][i], dq);
|
||||
}
|
||||
}
|
||||
|
|
@ -1650,7 +1648,7 @@ void av1_init_quantizer(AV1_COMP *cpi) {
|
|||
quants->y_quant_shift[q][i] = quants->y_quant_shift[q][1];
|
||||
quants->y_zbin[q][i] = quants->y_zbin[q][1];
|
||||
quants->y_round[q][i] = quants->y_round[q][1];
|
||||
cpi->y_dequant[q][i] = cpi->y_dequant[q][1];
|
||||
deq->y_dequant[q][i] = deq->y_dequant[q][1];
|
||||
|
||||
quants->uv_quant[q][i] = quants->uv_quant[q][1];
|
||||
quants->uv_quant_fp[q][i] = quants->uv_quant_fp[q][1];
|
||||
|
|
@ -1658,11 +1656,19 @@ void av1_init_quantizer(AV1_COMP *cpi) {
|
|||
quants->uv_quant_shift[q][i] = quants->uv_quant_shift[q][1];
|
||||
quants->uv_zbin[q][i] = quants->uv_zbin[q][1];
|
||||
quants->uv_round[q][i] = quants->uv_round[q][1];
|
||||
cpi->uv_dequant[q][i] = cpi->uv_dequant[q][1];
|
||||
deq->uv_dequant[q][i] = deq->uv_dequant[q][1];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void av1_init_quantizer(AV1_COMP *cpi) {
|
||||
AV1_COMMON *const cm = &cpi->common;
|
||||
QUANTS *const quants = &cpi->quants;
|
||||
Dequants *const dequants = &cpi->dequants;
|
||||
av1_build_quantizer(cm->bit_depth, cm->y_dc_delta_q, cm->uv_dc_delta_q,
|
||||
cm->uv_ac_delta_q, quants, dequants);
|
||||
}
|
||||
|
||||
void av1_init_plane_quantizers(const AV1_COMP *cpi, MACROBLOCK *x,
|
||||
int segment_id) {
|
||||
const AV1_COMMON *const cm = &cpi->common;
|
||||
|
|
@ -1712,11 +1718,12 @@ void av1_init_plane_quantizers(const AV1_COMP *cpi, MACROBLOCK *x,
|
|||
memcpy(&xd->plane[0].seg_iqmatrix[segment_id], cm->giqmatrix[qmlevel][0],
|
||||
sizeof(cm->giqmatrix[qmlevel][0]));
|
||||
#endif
|
||||
xd->plane[0].dequant = cpi->y_dequant[qindex];
|
||||
xd->plane[0].dequant = cpi->dequants.y_dequant[qindex];
|
||||
#if CONFIG_NEW_QUANT
|
||||
for (dq = 0; dq < QUANT_PROFILES; dq++) {
|
||||
x->plane[0].cuml_bins_nuq[dq] = quants->y_cuml_bins_nuq[dq][qindex];
|
||||
xd->plane[0].dequant_val_nuq[dq] = cpi->y_dequant_val_nuq[dq][qindex];
|
||||
xd->plane[0].dequant_val_nuq[dq] =
|
||||
cpi->dequants.y_dequant_val_nuq[dq][qindex];
|
||||
}
|
||||
#endif // CONFIG_NEW_QUANT
|
||||
|
||||
|
|
@ -1734,11 +1741,12 @@ void av1_init_plane_quantizers(const AV1_COMP *cpi, MACROBLOCK *x,
|
|||
memcpy(&xd->plane[i].seg_iqmatrix[segment_id], cm->giqmatrix[qmlevel][1],
|
||||
sizeof(cm->giqmatrix[qmlevel][1]));
|
||||
#endif
|
||||
xd->plane[i].dequant = cpi->uv_dequant[qindex];
|
||||
xd->plane[i].dequant = cpi->dequants.uv_dequant[qindex];
|
||||
#if CONFIG_NEW_QUANT
|
||||
for (dq = 0; dq < QUANT_PROFILES; dq++) {
|
||||
x->plane[i].cuml_bins_nuq[dq] = quants->uv_cuml_bins_nuq[dq][qindex];
|
||||
xd->plane[i].dequant_val_nuq[dq] = cpi->uv_dequant_val_nuq[dq][qindex];
|
||||
xd->plane[i].dequant_val_nuq[dq] =
|
||||
cpi->dequants.uv_dequant_val_nuq[dq][qindex];
|
||||
}
|
||||
#endif // CONFIG_NEW_QUANT
|
||||
}
|
||||
|
|
|
|||
15
third_party/aom/av1/encoder/av1_quantize.h
vendored
15
third_party/aom/av1/encoder/av1_quantize.h
vendored
|
|
@ -69,6 +69,17 @@ typedef struct {
|
|||
DECLARE_ALIGNED(16, int16_t, uv_round[QINDEX_RANGE][8]);
|
||||
} QUANTS;
|
||||
|
||||
typedef struct {
|
||||
DECLARE_ALIGNED(16, int16_t, y_dequant[QINDEX_RANGE][8]); // 8: SIMD width
|
||||
DECLARE_ALIGNED(16, int16_t, uv_dequant[QINDEX_RANGE][8]); // 8: SIMD width
|
||||
#if CONFIG_NEW_QUANT
|
||||
DECLARE_ALIGNED(16, dequant_val_type_nuq,
|
||||
y_dequant_val_nuq[QUANT_PROFILES][QINDEX_RANGE][COEF_BANDS]);
|
||||
DECLARE_ALIGNED(16, dequant_val_type_nuq,
|
||||
uv_dequant_val_nuq[QUANT_PROFILES][QINDEX_RANGE][COEF_BANDS]);
|
||||
#endif // CONFIG_NEW_QUANT
|
||||
} Dequants;
|
||||
|
||||
struct AV1_COMP;
|
||||
struct AV1Common;
|
||||
|
||||
|
|
@ -77,6 +88,10 @@ void av1_frame_init_quantizer(struct AV1_COMP *cpi);
|
|||
void av1_init_plane_quantizers(const struct AV1_COMP *cpi, MACROBLOCK *x,
|
||||
int segment_id);
|
||||
|
||||
void av1_build_quantizer(aom_bit_depth_t bit_depth, int y_dc_delta_q,
|
||||
int uv_dc_delta_q, int uv_ac_delta_q,
|
||||
QUANTS *const quants, Dequants *const deq);
|
||||
|
||||
void av1_init_quantizer(struct AV1_COMP *cpi);
|
||||
|
||||
void av1_set_quantizer(struct AV1Common *cm, int q);
|
||||
|
|
|
|||
1132
third_party/aom/av1/encoder/bitstream.c
vendored
1132
third_party/aom/av1/encoder/bitstream.c
vendored
File diff suppressed because it is too large
Load diff
28
third_party/aom/av1/encoder/block.h
vendored
28
third_party/aom/av1/encoder/block.h
vendored
|
|
@ -17,9 +17,7 @@
|
|||
#if CONFIG_PVQ
|
||||
#include "av1/encoder/encint.h"
|
||||
#endif
|
||||
#if CONFIG_REF_MV
|
||||
#include "av1/common/mvref_common.h"
|
||||
#endif
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
|
|
@ -79,13 +77,11 @@ typedef struct {
|
|||
int dc_sign_ctx[MAX_MB_PLANE]
|
||||
[MAX_SB_SQUARE / (TX_SIZE_W_MIN * TX_SIZE_H_MIN)];
|
||||
#endif
|
||||
#if CONFIG_REF_MV
|
||||
uint8_t ref_mv_count[MODE_CTX_REF_FRAMES];
|
||||
CANDIDATE_MV ref_mv_stack[MODE_CTX_REF_FRAMES][MAX_REF_MV_STACK_SIZE];
|
||||
#if CONFIG_EXT_INTER
|
||||
int16_t compound_mode_context[MODE_CTX_REF_FRAMES];
|
||||
#endif // CONFIG_EXT_INTER
|
||||
#endif
|
||||
} MB_MODE_INFO_EXT;
|
||||
|
||||
typedef struct {
|
||||
|
|
@ -141,27 +137,18 @@ struct macroblock {
|
|||
unsigned int pred_sse[TOTAL_REFS_PER_FRAME];
|
||||
int pred_mv_sad[TOTAL_REFS_PER_FRAME];
|
||||
|
||||
#if CONFIG_REF_MV
|
||||
int *nmvjointcost;
|
||||
int nmv_vec_cost[NMV_CONTEXTS][MV_JOINTS];
|
||||
int *nmvcost[NMV_CONTEXTS][2];
|
||||
int *nmvcost_hp[NMV_CONTEXTS][2];
|
||||
int **mv_cost_stack[NMV_CONTEXTS];
|
||||
int *nmvjointsadcost;
|
||||
#else
|
||||
int nmvjointcost[MV_JOINTS];
|
||||
int *nmvcost[2];
|
||||
int *nmvcost_hp[2];
|
||||
int nmvjointsadcost[MV_JOINTS];
|
||||
#endif
|
||||
|
||||
int **mvcost;
|
||||
int *nmvsadcost[2];
|
||||
int *nmvsadcost_hp[2];
|
||||
int **mvsadcost;
|
||||
|
||||
#if CONFIG_MOTION_VAR
|
||||
int32_t *wsrc_buf;
|
||||
int32_t *mask_buf;
|
||||
uint8_t *above_pred_buf;
|
||||
uint8_t *left_pred_buf;
|
||||
#endif // CONFIG_MOTION_VAR
|
||||
|
||||
#if CONFIG_PALETTE
|
||||
|
|
@ -174,9 +161,7 @@ struct macroblock {
|
|||
|
||||
#if CONFIG_VAR_TX
|
||||
uint8_t blk_skip[MAX_MB_PLANE][MAX_MIB_SIZE * MAX_MIB_SIZE * 8];
|
||||
#if CONFIG_REF_MV
|
||||
uint8_t blk_skip_drl[MAX_MB_PLANE][MAX_MIB_SIZE * MAX_MIB_SIZE * 8];
|
||||
#endif
|
||||
#endif
|
||||
|
||||
int skip;
|
||||
|
|
@ -226,8 +211,11 @@ struct macroblock {
|
|||
// This is needed when using the 8x8 Daala distortion metric during RDO,
|
||||
// because it evaluates distortion in a different order than the underlying
|
||||
// 4x4 blocks are coded.
|
||||
int rate_4x4[256];
|
||||
#endif
|
||||
int rate_4x4[MAX_SB_SQUARE / (TX_SIZE_W_MIN * TX_SIZE_H_MIN)];
|
||||
#if CONFIG_CB4X4
|
||||
DECLARE_ALIGNED(16, uint8_t, decoded_8x8[8 * 8]);
|
||||
#endif // CONFIG_CB4X4
|
||||
#endif // CONFIG_DAALA_DIST
|
||||
#if CONFIG_CFL
|
||||
// Whether luma needs to be stored during RDO.
|
||||
int cfl_store_y;
|
||||
|
|
|
|||
4
third_party/aom/av1/encoder/context_tree.h
vendored
4
third_party/aom/av1/encoder/context_tree.h
vendored
|
|
@ -34,7 +34,6 @@ typedef struct {
|
|||
uint8_t *blk_skip[MAX_MB_PLANE];
|
||||
#endif
|
||||
|
||||
// dual buffer pointers, 0: in use, 1: best in store
|
||||
tran_low_t *coeff[MAX_MB_PLANE];
|
||||
tran_low_t *qcoeff[MAX_MB_PLANE];
|
||||
tran_low_t *dqcoeff[MAX_MB_PLANE];
|
||||
|
|
@ -48,9 +47,8 @@ typedef struct {
|
|||
|
||||
int num_4x4_blk;
|
||||
int skip;
|
||||
int pred_pixel_ready;
|
||||
// For current partition, only if all Y, U, and V transform blocks'
|
||||
// coefficients are quantized to 0, skippable is set to 0.
|
||||
// coefficients are quantized to 0, skippable is set to 1.
|
||||
int skippable;
|
||||
int best_mode_index;
|
||||
int hybrid_pred_diff;
|
||||
|
|
|
|||
15
third_party/aom/av1/encoder/corner_match.c
vendored
15
third_party/aom/av1/encoder/corner_match.c
vendored
|
|
@ -9,16 +9,13 @@
|
|||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <memory.h>
|
||||
#include <math.h>
|
||||
|
||||
#include "./av1_rtcd.h"
|
||||
#include "av1/encoder/corner_match.h"
|
||||
|
||||
#define MATCH_SZ 13
|
||||
#define MATCH_SZ_BY2 ((MATCH_SZ - 1) / 2)
|
||||
#define MATCH_SZ_SQ (MATCH_SZ * MATCH_SZ)
|
||||
#define SEARCH_SZ 9
|
||||
#define SEARCH_SZ_BY2 ((SEARCH_SZ - 1) / 2)
|
||||
|
||||
|
|
@ -28,8 +25,8 @@
|
|||
centered at (x, y).
|
||||
*/
|
||||
static double compute_variance(unsigned char *im, int stride, int x, int y) {
|
||||
int sum = 0.0;
|
||||
int sumsq = 0.0;
|
||||
int sum = 0;
|
||||
int sumsq = 0;
|
||||
int var;
|
||||
int i, j;
|
||||
for (i = 0; i < MATCH_SZ; ++i)
|
||||
|
|
@ -46,9 +43,9 @@ static double compute_variance(unsigned char *im, int stride, int x, int y) {
|
|||
correlation/standard deviation are taken over MATCH_SZ by MATCH_SZ windows
|
||||
of each image, centered at (x1, y1) and (x2, y2) respectively.
|
||||
*/
|
||||
static double compute_cross_correlation(unsigned char *im1, int stride1, int x1,
|
||||
int y1, unsigned char *im2, int stride2,
|
||||
int x2, int y2) {
|
||||
double compute_cross_correlation_c(unsigned char *im1, int stride1, int x1,
|
||||
int y1, unsigned char *im2, int stride2,
|
||||
int x2, int y2) {
|
||||
int v1, v2;
|
||||
int sum1 = 0;
|
||||
int sum2 = 0;
|
||||
|
|
|
|||
4
third_party/aom/av1/encoder/corner_match.h
vendored
4
third_party/aom/av1/encoder/corner_match.h
vendored
|
|
@ -15,6 +15,10 @@
|
|||
#include <stdlib.h>
|
||||
#include <memory.h>
|
||||
|
||||
#define MATCH_SZ 13
|
||||
#define MATCH_SZ_BY2 ((MATCH_SZ - 1) / 2)
|
||||
#define MATCH_SZ_SQ (MATCH_SZ * MATCH_SZ)
|
||||
|
||||
typedef struct {
|
||||
int x, y;
|
||||
int rx, ry;
|
||||
|
|
|
|||
|
|
@ -12,19 +12,19 @@
|
|||
#include "encint.h"
|
||||
|
||||
void od_encode_checkpoint(const daala_enc_ctx *enc, od_rollback_buffer *rbuf) {
|
||||
#if CONFIG_DAALA_EC
|
||||
#if !CONFIG_ANS
|
||||
od_ec_enc_checkpoint(&rbuf->ec, &enc->w.ec);
|
||||
#else
|
||||
#error "CONFIG_PVQ currently requires CONFIG_DAALA_EC."
|
||||
#error "CONFIG_PVQ currently requires !CONFIG_ANS."
|
||||
#endif
|
||||
OD_COPY(&rbuf->adapt, enc->state.adapt, 1);
|
||||
}
|
||||
|
||||
void od_encode_rollback(daala_enc_ctx *enc, const od_rollback_buffer *rbuf) {
|
||||
#if CONFIG_DAALA_EC
|
||||
#if !CONFIG_ANS
|
||||
od_ec_enc_rollback(&enc->w.ec, &rbuf->ec);
|
||||
#else
|
||||
#error "CONFIG_PVQ currently requires CONFIG_DAALA_EC."
|
||||
#error "CONFIG_PVQ currently requires !CONFIG_ANS."
|
||||
#endif
|
||||
OD_COPY(enc->state.adapt, &rbuf->adapt, 1);
|
||||
}
|
||||
|
|
|
|||
57
third_party/aom/av1/encoder/dct.c
vendored
57
third_party/aom/av1/encoder/dct.c
vendored
|
|
@ -19,7 +19,7 @@
|
|||
#include "aom_ports/mem.h"
|
||||
#include "av1/common/blockd.h"
|
||||
#include "av1/common/av1_fwd_txfm1d.h"
|
||||
#include "av1/common/av1_fwd_txfm2d_cfg.h"
|
||||
#include "av1/common/av1_fwd_txfm1d_cfg.h"
|
||||
#include "av1/common/idct.h"
|
||||
|
||||
static INLINE void range_check(const tran_low_t *input, const int size,
|
||||
|
|
@ -1022,6 +1022,10 @@ static void fhalfright32(const tran_low_t *input, tran_low_t *output) {
|
|||
}
|
||||
|
||||
#if CONFIG_EXT_TX
|
||||
// TODO(sarahparker) these functions will be removed once the highbitdepth
|
||||
// codepath works properly for rectangular transforms. They have almost
|
||||
// identical versions in av1_fwd_txfm1d.c, but those are currently only
|
||||
// being used for square transforms.
|
||||
static void fidtx4(const tran_low_t *input, tran_low_t *output) {
|
||||
int i;
|
||||
for (i = 0; i < 4; ++i)
|
||||
|
|
@ -2133,8 +2137,7 @@ static void fdct64_col(const tran_low_t *input, tran_low_t *output) {
|
|||
int32_t in[64], out[64];
|
||||
int i;
|
||||
for (i = 0; i < 64; ++i) in[i] = (int32_t)input[i];
|
||||
av1_fdct64_new(in, out, fwd_cos_bit_col_dct_dct_64,
|
||||
fwd_stage_range_col_dct_dct_64);
|
||||
av1_fdct64_new(in, out, fwd_cos_bit_col_dct_64, fwd_stage_range_col_dct_64);
|
||||
for (i = 0; i < 64; ++i) output[i] = (tran_low_t)out[i];
|
||||
}
|
||||
|
||||
|
|
@ -2142,8 +2145,7 @@ static void fdct64_row(const tran_low_t *input, tran_low_t *output) {
|
|||
int32_t in[64], out[64];
|
||||
int i;
|
||||
for (i = 0; i < 64; ++i) in[i] = (int32_t)input[i];
|
||||
av1_fdct64_new(in, out, fwd_cos_bit_row_dct_dct_64,
|
||||
fwd_stage_range_row_dct_dct_64);
|
||||
av1_fdct64_new(in, out, fwd_cos_bit_row_dct_64, fwd_stage_range_row_dct_64);
|
||||
for (i = 0; i < 64; ++i) output[i] = (tran_low_t)out[i];
|
||||
}
|
||||
|
||||
|
|
@ -2225,4 +2227,49 @@ void av1_highbd_fht64x64_c(const int16_t *input, tran_low_t *output, int stride,
|
|||
}
|
||||
#endif // CONFIG_TX64X64
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
|
||||
#if CONFIG_DPCM_INTRA
|
||||
void av1_dpcm_ft4_c(const int16_t *input, int stride, TX_TYPE_1D tx_type,
|
||||
tran_low_t *output) {
|
||||
assert(tx_type < TX_TYPES_1D);
|
||||
static const transform_1d FHT[] = { fdct4, fadst4, fadst4, fidtx4 };
|
||||
const transform_1d ft = FHT[tx_type];
|
||||
tran_low_t temp_in[4];
|
||||
for (int i = 0; i < 4; ++i)
|
||||
temp_in[i] = (tran_low_t)fdct_round_shift(input[i * stride] * 4 * Sqrt2);
|
||||
ft(temp_in, output);
|
||||
}
|
||||
|
||||
void av1_dpcm_ft8_c(const int16_t *input, int stride, TX_TYPE_1D tx_type,
|
||||
tran_low_t *output) {
|
||||
assert(tx_type < TX_TYPES_1D);
|
||||
static const transform_1d FHT[] = { fdct8, fadst8, fadst8, fidtx8 };
|
||||
const transform_1d ft = FHT[tx_type];
|
||||
tran_low_t temp_in[8];
|
||||
for (int i = 0; i < 8; ++i) temp_in[i] = input[i * stride] * 4;
|
||||
ft(temp_in, output);
|
||||
}
|
||||
|
||||
void av1_dpcm_ft16_c(const int16_t *input, int stride, TX_TYPE_1D tx_type,
|
||||
tran_low_t *output) {
|
||||
assert(tx_type < TX_TYPES_1D);
|
||||
static const transform_1d FHT[] = { fdct16, fadst16, fadst16, fidtx16 };
|
||||
const transform_1d ft = FHT[tx_type];
|
||||
tran_low_t temp_in[16];
|
||||
for (int i = 0; i < 16; ++i)
|
||||
temp_in[i] = (tran_low_t)fdct_round_shift(input[i * stride] * 2 * Sqrt2);
|
||||
ft(temp_in, output);
|
||||
}
|
||||
|
||||
void av1_dpcm_ft32_c(const int16_t *input, int stride, TX_TYPE_1D tx_type,
|
||||
tran_low_t *output) {
|
||||
assert(tx_type < TX_TYPES_1D);
|
||||
static const transform_1d FHT[] = { fdct32, fhalfright32, fhalfright32,
|
||||
fidtx32 };
|
||||
const transform_1d ft = FHT[tx_type];
|
||||
tran_low_t temp_in[32];
|
||||
for (int i = 0; i < 32; ++i) temp_in[i] = input[i * stride];
|
||||
ft(temp_in, output);
|
||||
}
|
||||
#endif // CONFIG_DPCM_INTRA
|
||||
#endif // !AV1_DCT_GTEST
|
||||
|
|
|
|||
1335
third_party/aom/av1/encoder/encodeframe.c
vendored
1335
third_party/aom/av1/encoder/encodeframe.c
vendored
File diff suppressed because it is too large
Load diff
9
third_party/aom/av1/encoder/encodeframe.h
vendored
9
third_party/aom/av1/encoder/encodeframe.h
vendored
|
|
@ -25,13 +25,6 @@ struct yv12_buffer_config;
|
|||
struct AV1_COMP;
|
||||
struct ThreadData;
|
||||
|
||||
// Constants used in SOURCE_VAR_BASED_PARTITION
|
||||
#define VAR_HIST_MAX_BG_VAR 1000
|
||||
#define VAR_HIST_FACTOR 10
|
||||
#define VAR_HIST_BINS (VAR_HIST_MAX_BG_VAR / VAR_HIST_FACTOR + 1)
|
||||
#define VAR_HIST_LARGE_CUT_OFF 75
|
||||
#define VAR_HIST_SMALL_CUT_OFF 45
|
||||
|
||||
void av1_setup_src_planes(struct macroblock *x,
|
||||
const struct yv12_buffer_config *src, int mi_row,
|
||||
int mi_col);
|
||||
|
|
@ -42,8 +35,6 @@ void av1_init_tile_data(struct AV1_COMP *cpi);
|
|||
void av1_encode_tile(struct AV1_COMP *cpi, struct ThreadData *td, int tile_row,
|
||||
int tile_col);
|
||||
|
||||
void av1_set_variance_partition_thresholds(struct AV1_COMP *cpi, int q);
|
||||
|
||||
void av1_update_tx_type_count(const struct AV1Common *cm, MACROBLOCKD *xd,
|
||||
#if CONFIG_TXK_SEL
|
||||
int block, int plane,
|
||||
|
|
|
|||
703
third_party/aom/av1/encoder/encodemb.c
vendored
703
third_party/aom/av1/encoder/encodemb.c
vendored
|
|
@ -115,7 +115,7 @@ static const int plane_rd_mult[REF_TYPES][PLANE_TYPES] = {
|
|||
#if CONFIG_EC_ADAPT
|
||||
{ 10, 7 }, { 8, 5 },
|
||||
#else
|
||||
{ 10, 6 }, { 8, 5 },
|
||||
{ 10, 6 }, { 8, 6 },
|
||||
#endif
|
||||
};
|
||||
|
||||
|
|
@ -125,35 +125,31 @@ static const int plane_rd_mult[REF_TYPES][PLANE_TYPES] = {
|
|||
rd_cost1 = RDCOST(rdmult, rddiv, rate1, error1); \
|
||||
}
|
||||
|
||||
static INLINE int64_t
|
||||
get_token_bit_costs(unsigned int token_costs[2][COEFF_CONTEXTS][ENTROPY_TOKENS],
|
||||
int skip_eob, int ctx, int token) {
|
||||
#if CONFIG_NEW_TOKENSET
|
||||
static INLINE unsigned int get_token_bit_costs(
|
||||
unsigned int token_costs[2][COEFF_CONTEXTS][ENTROPY_TOKENS], int skip_eob,
|
||||
int ctx, int token) {
|
||||
(void)skip_eob;
|
||||
return token_costs[token == ZERO_TOKEN || token == EOB_TOKEN][ctx][token];
|
||||
#else
|
||||
return token_costs[skip_eob][ctx][token];
|
||||
#endif
|
||||
}
|
||||
|
||||
#if !CONFIG_LV_MAP
|
||||
#define USE_GREEDY_OPTIMIZE_B 0
|
||||
|
||||
#if USE_GREEDY_OPTIMIZE_B
|
||||
|
||||
typedef struct av1_token_state {
|
||||
typedef struct av1_token_state_greedy {
|
||||
int16_t token;
|
||||
tran_low_t qc;
|
||||
tran_low_t dqc;
|
||||
} av1_token_state;
|
||||
} av1_token_state_greedy;
|
||||
|
||||
int av1_optimize_b(const AV1_COMMON *cm, MACROBLOCK *mb, int plane, int block,
|
||||
TX_SIZE tx_size, int ctx) {
|
||||
#if !CONFIG_PVQ
|
||||
static int optimize_b_greedy(const AV1_COMMON *cm, MACROBLOCK *mb, int plane,
|
||||
int block, TX_SIZE tx_size, int ctx) {
|
||||
MACROBLOCKD *const xd = &mb->e_mbd;
|
||||
struct macroblock_plane *const p = &mb->plane[plane];
|
||||
struct macroblockd_plane *const pd = &xd->plane[plane];
|
||||
const int ref = is_inter_block(&xd->mi[0]->mbmi);
|
||||
av1_token_state tokens[MAX_TX_SQUARE + 1][2];
|
||||
av1_token_state_greedy tokens[MAX_TX_SQUARE + 1][2];
|
||||
uint8_t token_cache[MAX_TX_SQUARE];
|
||||
const tran_low_t *const coeff = BLOCK_OFFSET(p->coeff, block);
|
||||
tran_low_t *const qcoeff = BLOCK_OFFSET(p->qcoeff, block);
|
||||
|
|
@ -176,38 +172,23 @@ int av1_optimize_b(const AV1_COMMON *cm, MACROBLOCK *mb, int plane, int block,
|
|||
#if CONFIG_NEW_QUANT
|
||||
int dq = get_dq_profile_from_ctx(mb->qindex, ctx, ref, plane_type);
|
||||
const dequant_val_type_nuq *dequant_val = pd->dequant_val_nuq[dq];
|
||||
#elif !CONFIG_AOM_QM
|
||||
const int dq_step[2] = { dequant_ptr[0] >> shift, dequant_ptr[1] >> shift };
|
||||
#endif // CONFIG_NEW_QUANT
|
||||
int sz = 0;
|
||||
const int64_t rddiv = mb->rddiv;
|
||||
int64_t rd_cost0, rd_cost1;
|
||||
int16_t t0, t1;
|
||||
int i, final_eob;
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
const int cat6_bits = av1_get_cat6_extrabits_size(tx_size, xd->bd);
|
||||
#else
|
||||
const int cat6_bits = av1_get_cat6_extrabits_size(tx_size, 8);
|
||||
#endif
|
||||
unsigned int(*token_costs)[2][COEFF_CONTEXTS][ENTROPY_TOKENS] =
|
||||
mb->token_costs[txsize_sqr_map[tx_size]][plane_type][ref];
|
||||
const int default_eob = tx_size_2d[tx_size];
|
||||
|
||||
assert((mb->qindex == 0) ^ (xd->lossless[xd->mi[0]->mbmi.segment_id] == 0));
|
||||
assert(mb->qindex > 0);
|
||||
|
||||
assert((!plane_type && !plane) || (plane_type && plane));
|
||||
assert(eob <= default_eob);
|
||||
|
||||
int64_t rdmult = (mb->rdmult * plane_rd_mult[ref][plane_type]) >> 1;
|
||||
/* CpuSpeedTest uses "--min-q=0 --max-q=0" and expects 100dB psnr
|
||||
* This creates conflict with search for a better EOB position
|
||||
* The line below is to make sure EOB search is disabled at this corner case.
|
||||
*/
|
||||
#if !CONFIG_NEW_QUANT && !CONFIG_AOM_QM
|
||||
if (dq_step[1] <= 4) {
|
||||
rdmult = 1;
|
||||
}
|
||||
#endif
|
||||
|
||||
int64_t rate0, rate1;
|
||||
for (i = 0; i < eob; i++) {
|
||||
|
|
@ -402,22 +383,10 @@ int av1_optimize_b(const AV1_COMMON *cm, MACROBLOCK *mb, int plane, int block,
|
|||
dqc_a = shift ? ROUND_POWER_OF_TWO(dqc_a, shift) : dqc_a;
|
||||
if (sz) dqc_a = -dqc_a;
|
||||
#else
|
||||
// The 32x32 transform coefficient uses half quantization step size.
|
||||
// Account for the rounding difference in the dequantized coefficeint
|
||||
// value when the quantization index is dropped from an even number
|
||||
// to an odd number.
|
||||
|
||||
#if CONFIG_AOM_QM
|
||||
tran_low_t offset = dqv >> shift;
|
||||
#else
|
||||
tran_low_t offset = dq_step[rc != 0];
|
||||
#endif
|
||||
if (shift & x_a) offset += (dqv & 0x01);
|
||||
|
||||
if (sz == 0)
|
||||
dqc_a = dqcoeff[rc] - offset;
|
||||
if (x_a < 0)
|
||||
dqc_a = -((-x_a * dqv) >> shift);
|
||||
else
|
||||
dqc_a = dqcoeff[rc] + offset;
|
||||
dqc_a = (x_a * dqv) >> shift;
|
||||
#endif // CONFIG_NEW_QUANT
|
||||
} else {
|
||||
dqc_a = 0;
|
||||
|
|
@ -483,19 +452,11 @@ int av1_optimize_b(const AV1_COMMON *cm, MACROBLOCK *mb, int plane, int block,
|
|||
|
||||
mb->plane[plane].eobs[block] = final_eob;
|
||||
return final_eob;
|
||||
|
||||
#else // !CONFIG_PVQ
|
||||
(void)cm;
|
||||
(void)tx_size;
|
||||
(void)ctx;
|
||||
struct macroblock_plane *const p = &mb->plane[plane];
|
||||
return p->eobs[block];
|
||||
#endif // !CONFIG_PVQ
|
||||
}
|
||||
|
||||
#else // USE_GREEDY_OPTIMIZE_B
|
||||
|
||||
typedef struct av1_token_state {
|
||||
typedef struct av1_token_state_org {
|
||||
int64_t error;
|
||||
int rate;
|
||||
int16_t next;
|
||||
|
|
@ -503,16 +464,15 @@ typedef struct av1_token_state {
|
|||
tran_low_t qc;
|
||||
tran_low_t dqc;
|
||||
uint8_t best_index;
|
||||
} av1_token_state;
|
||||
} av1_token_state_org;
|
||||
|
||||
int av1_optimize_b(const AV1_COMMON *cm, MACROBLOCK *mb, int plane, int block,
|
||||
TX_SIZE tx_size, int ctx) {
|
||||
#if !CONFIG_PVQ
|
||||
static int optimize_b_org(const AV1_COMMON *cm, MACROBLOCK *mb, int plane,
|
||||
int block, TX_SIZE tx_size, int ctx) {
|
||||
MACROBLOCKD *const xd = &mb->e_mbd;
|
||||
struct macroblock_plane *const p = &mb->plane[plane];
|
||||
struct macroblockd_plane *const pd = &xd->plane[plane];
|
||||
const int ref = is_inter_block(&xd->mi[0]->mbmi);
|
||||
av1_token_state tokens[MAX_TX_SQUARE + 1][2];
|
||||
av1_token_state_org tokens[MAX_TX_SQUARE + 1][2];
|
||||
uint8_t token_cache[MAX_TX_SQUARE];
|
||||
const tran_low_t *const coeff = BLOCK_OFFSET(p->coeff, block);
|
||||
tran_low_t *const qcoeff = BLOCK_OFFSET(p->qcoeff, block);
|
||||
|
|
@ -536,8 +496,6 @@ int av1_optimize_b(const AV1_COMMON *cm, MACROBLOCK *mb, int plane, int block,
|
|||
#if CONFIG_NEW_QUANT
|
||||
int dq = get_dq_profile_from_ctx(mb->qindex, ctx, ref, plane_type);
|
||||
const dequant_val_type_nuq *dequant_val = pd->dequant_val_nuq[dq];
|
||||
#elif !CONFIG_AOM_QM
|
||||
const int dq_step[2] = { dequant_ptr[0] >> shift, dequant_ptr[1] >> shift };
|
||||
#endif // CONFIG_NEW_QUANT
|
||||
int next = eob, sz = 0;
|
||||
const int64_t rdmult = (mb->rdmult * plane_rd_mult[ref][plane_type]) >> 1;
|
||||
|
|
@ -549,11 +507,7 @@ int av1_optimize_b(const AV1_COMMON *cm, MACROBLOCK *mb, int plane, int block,
|
|||
int best, band = (eob < default_eob) ? band_translate[eob]
|
||||
: band_translate[eob - 1];
|
||||
int pt, i, final_eob;
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
const int cat6_bits = av1_get_cat6_extrabits_size(tx_size, xd->bd);
|
||||
#else
|
||||
const int cat6_bits = av1_get_cat6_extrabits_size(tx_size, 8);
|
||||
#endif
|
||||
unsigned int(*token_costs)[2][COEFF_CONTEXTS][ENTROPY_TOKENS] =
|
||||
mb->token_costs[txsize_sqr_map[tx_size]][plane_type][ref];
|
||||
const uint16_t *band_counts = &band_count_table[tx_size][band];
|
||||
|
|
@ -566,11 +520,10 @@ int av1_optimize_b(const AV1_COMMON *cm, MACROBLOCK *mb, int plane, int block,
|
|||
? av1_get_qindex(&cm->seg, xd->mi[0]->mbmi.segment_id,
|
||||
cm->base_qindex)
|
||||
: cm->base_qindex;
|
||||
if (qindex == 0) {
|
||||
assert((qindex == 0) ^ (xd->lossless[xd->mi[0]->mbmi.segment_id] == 0));
|
||||
}
|
||||
assert(qindex > 0);
|
||||
(void)qindex;
|
||||
#else
|
||||
assert((mb->qindex == 0) ^ (xd->lossless[xd->mi[0]->mbmi.segment_id] == 0));
|
||||
assert(mb->qindex > 0);
|
||||
#endif
|
||||
|
||||
token_costs += band;
|
||||
|
|
@ -777,22 +730,10 @@ int av1_optimize_b(const AV1_COMMON *cm, MACROBLOCK *mb, int plane, int block,
|
|||
: tokens[i][1].dqc;
|
||||
if (sz) tokens[i][1].dqc = -tokens[i][1].dqc;
|
||||
#else
|
||||
// The 32x32 transform coefficient uses half quantization step size.
|
||||
// Account for the rounding difference in the dequantized coefficeint
|
||||
// value when the quantization index is dropped from an even number
|
||||
// to an odd number.
|
||||
|
||||
#if CONFIG_AOM_QM
|
||||
tran_low_t offset = dqv >> shift;
|
||||
#else
|
||||
tran_low_t offset = dq_step[rc != 0];
|
||||
#endif
|
||||
if (shift & x) offset += (dqv & 0x01);
|
||||
|
||||
if (sz == 0)
|
||||
tokens[i][1].dqc = dqcoeff[rc] - offset;
|
||||
if (x < 0)
|
||||
tokens[i][1].dqc = -((-x * dqv) >> shift);
|
||||
else
|
||||
tokens[i][1].dqc = dqcoeff[rc] + offset;
|
||||
tokens[i][1].dqc = (x * dqv) >> shift;
|
||||
#endif // CONFIG_NEW_QUANT
|
||||
} else {
|
||||
tokens[i][1].dqc = 0;
|
||||
|
|
@ -858,16 +799,47 @@ int av1_optimize_b(const AV1_COMMON *cm, MACROBLOCK *mb, int plane, int block,
|
|||
mb->plane[plane].eobs[block] = final_eob;
|
||||
assert(final_eob <= default_eob);
|
||||
return final_eob;
|
||||
#else // !CONFIG_PVQ
|
||||
(void)cm;
|
||||
(void)tx_size;
|
||||
(void)ctx;
|
||||
struct macroblock_plane *const p = &mb->plane[plane];
|
||||
return p->eobs[block];
|
||||
#endif // !CONFIG_PVQ
|
||||
}
|
||||
|
||||
#endif // USE_GREEDY_OPTIMIZE_B
|
||||
#endif // !CONFIG_LV_MAP
|
||||
|
||||
int av1_optimize_b(const AV1_COMMON *cm, MACROBLOCK *mb, int plane, int block,
|
||||
BLOCK_SIZE plane_bsize, TX_SIZE tx_size,
|
||||
const ENTROPY_CONTEXT *a, const ENTROPY_CONTEXT *l) {
|
||||
MACROBLOCKD *const xd = &mb->e_mbd;
|
||||
struct macroblock_plane *const p = &mb->plane[plane];
|
||||
const int eob = p->eobs[block];
|
||||
assert((mb->qindex == 0) ^ (xd->lossless[xd->mi[0]->mbmi.segment_id] == 0));
|
||||
if (eob == 0) return eob;
|
||||
if (xd->lossless[xd->mi[0]->mbmi.segment_id]) return eob;
|
||||
#if CONFIG_PVQ
|
||||
(void)cm;
|
||||
(void)tx_size;
|
||||
(void)a;
|
||||
(void)l;
|
||||
return eob;
|
||||
#endif
|
||||
|
||||
#if !CONFIG_LV_MAP
|
||||
(void)plane_bsize;
|
||||
#if CONFIG_VAR_TX
|
||||
int ctx = get_entropy_context(tx_size, a, l);
|
||||
#else
|
||||
int ctx = combine_entropy_contexts(*a, *l);
|
||||
#endif
|
||||
|
||||
#if USE_GREEDY_OPTIMIZE_B
|
||||
return optimize_b_greedy(cm, mb, plane, block, tx_size, ctx);
|
||||
#else // USE_GREEDY_OPTIMIZE_B
|
||||
return optimize_b_org(cm, mb, plane, block, tx_size, ctx);
|
||||
#endif // USE_GREEDY_OPTIMIZE_B
|
||||
#else // !CONFIG_LV_MAP
|
||||
TXB_CTX txb_ctx;
|
||||
get_txb_ctx(plane_bsize, tx_size, plane, a, l, &txb_ctx);
|
||||
return av1_optimize_txb(cm, mb, plane, block, tx_size, &txb_ctx);
|
||||
#endif // !CONFIG_LV_MAP
|
||||
}
|
||||
|
||||
#if !CONFIG_PVQ
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
|
|
@ -1158,8 +1130,7 @@ static void encode_block(int plane, int block, int blk_row, int blk_col,
|
|||
#endif
|
||||
|
||||
#if !CONFIG_PVQ
|
||||
if (p->eobs[block] && !xd->lossless[xd->mi[0]->mbmi.segment_id])
|
||||
av1_optimize_b(cm, x, plane, block, tx_size, ctx);
|
||||
av1_optimize_b(cm, x, plane, block, plane_bsize, tx_size, a, l);
|
||||
|
||||
av1_set_txb_context(x, plane, block, tx_size, a, l);
|
||||
|
||||
|
|
@ -1202,12 +1173,13 @@ static void encode_block_inter(int plane, int block, int blk_row, int blk_col,
|
|||
if (tx_size == plane_tx_size) {
|
||||
encode_block(plane, block, blk_row, blk_col, plane_bsize, tx_size, arg);
|
||||
} else {
|
||||
assert(tx_size < TX_SIZES_ALL);
|
||||
const TX_SIZE sub_txs = sub_tx_size_map[tx_size];
|
||||
assert(sub_txs < tx_size);
|
||||
// This is the square transform block partition entry point.
|
||||
int bsl = tx_size_wide_unit[sub_txs];
|
||||
int i;
|
||||
assert(bsl > 0);
|
||||
assert(tx_size < TX_SIZES_ALL);
|
||||
|
||||
for (i = 0; i < 4; ++i) {
|
||||
const int offsetr = blk_row + ((i >> 1) * bsl);
|
||||
|
|
@ -1301,8 +1273,8 @@ void av1_encode_sby_pass1(AV1_COMMON *cm, MACROBLOCK *x, BLOCK_SIZE bsize) {
|
|||
encode_block_pass1, &args);
|
||||
}
|
||||
|
||||
void av1_encode_sb(AV1_COMMON *cm, MACROBLOCK *x, BLOCK_SIZE bsize,
|
||||
const int mi_row, const int mi_col) {
|
||||
void av1_encode_sb(AV1_COMMON *cm, MACROBLOCK *x, BLOCK_SIZE bsize, int mi_row,
|
||||
int mi_col) {
|
||||
MACROBLOCKD *const xd = &x->e_mbd;
|
||||
struct optimize_ctx ctx;
|
||||
MB_MODE_INFO *mbmi = &xd->mi[0]->mbmi;
|
||||
|
|
@ -1433,6 +1405,301 @@ static void encode_block_intra_and_set_context(int plane, int block,
|
|||
#endif
|
||||
}
|
||||
|
||||
#if CONFIG_DPCM_INTRA
|
||||
static int get_eob(const tran_low_t *qcoeff, intptr_t n_coeffs,
|
||||
const int16_t *scan) {
|
||||
int eob = -1;
|
||||
for (int i = (int)n_coeffs - 1; i >= 0; i--) {
|
||||
const int rc = scan[i];
|
||||
if (qcoeff[rc]) {
|
||||
eob = i;
|
||||
break;
|
||||
}
|
||||
}
|
||||
return eob + 1;
|
||||
}
|
||||
|
||||
static void quantize_scaler(int coeff, int16_t zbin, int16_t round_value,
|
||||
int16_t quant, int16_t quant_shift, int16_t dequant,
|
||||
int log_scale, tran_low_t *const qcoeff,
|
||||
tran_low_t *const dqcoeff) {
|
||||
zbin = ROUND_POWER_OF_TWO(zbin, log_scale);
|
||||
round_value = ROUND_POWER_OF_TWO(round_value, log_scale);
|
||||
const int coeff_sign = (coeff >> 31);
|
||||
const int abs_coeff = (coeff ^ coeff_sign) - coeff_sign;
|
||||
if (abs_coeff >= zbin) {
|
||||
int tmp = clamp(abs_coeff + round_value, INT16_MIN, INT16_MAX);
|
||||
tmp = ((((tmp * quant) >> 16) + tmp) * quant_shift) >> (16 - log_scale);
|
||||
*qcoeff = (tmp ^ coeff_sign) - coeff_sign;
|
||||
*dqcoeff = (*qcoeff * dequant) / (1 << log_scale);
|
||||
}
|
||||
}
|
||||
|
||||
typedef void (*dpcm_fwd_tx_func)(const int16_t *input, int stride,
|
||||
TX_TYPE_1D tx_type, tran_low_t *output);
|
||||
|
||||
static dpcm_fwd_tx_func get_dpcm_fwd_tx_func(int tx_length) {
|
||||
switch (tx_length) {
|
||||
case 4: return av1_dpcm_ft4_c;
|
||||
case 8: return av1_dpcm_ft8_c;
|
||||
case 16: return av1_dpcm_ft16_c;
|
||||
case 32:
|
||||
return av1_dpcm_ft32_c;
|
||||
// TODO(huisu): add support for TX_64X64.
|
||||
default: assert(0); return NULL;
|
||||
}
|
||||
}
|
||||
|
||||
static void process_block_dpcm_vert(TX_SIZE tx_size, TX_TYPE_1D tx_type_1d,
|
||||
struct macroblockd_plane *const pd,
|
||||
struct macroblock_plane *const p,
|
||||
uint8_t *src, int src_stride, uint8_t *dst,
|
||||
int dst_stride, int16_t *src_diff,
|
||||
int diff_stride, tran_low_t *coeff,
|
||||
tran_low_t *qcoeff, tran_low_t *dqcoeff) {
|
||||
const int tx1d_width = tx_size_wide[tx_size];
|
||||
dpcm_fwd_tx_func forward_tx = get_dpcm_fwd_tx_func(tx1d_width);
|
||||
dpcm_inv_txfm_add_func inverse_tx =
|
||||
av1_get_dpcm_inv_txfm_add_func(tx1d_width);
|
||||
const int tx1d_height = tx_size_high[tx_size];
|
||||
const int log_scale = av1_get_tx_scale(tx_size);
|
||||
int q_idx = 0;
|
||||
for (int r = 0; r < tx1d_height; ++r) {
|
||||
// Update prediction.
|
||||
if (r > 0) memcpy(dst, dst - dst_stride, tx1d_width * sizeof(dst[0]));
|
||||
// Subtraction.
|
||||
for (int c = 0; c < tx1d_width; ++c) src_diff[c] = src[c] - dst[c];
|
||||
// Forward transform.
|
||||
forward_tx(src_diff, 1, tx_type_1d, coeff);
|
||||
// Quantization.
|
||||
for (int c = 0; c < tx1d_width; ++c) {
|
||||
quantize_scaler(coeff[c], p->zbin[q_idx], p->round[q_idx],
|
||||
p->quant[q_idx], p->quant_shift[q_idx],
|
||||
pd->dequant[q_idx], log_scale, &qcoeff[c], &dqcoeff[c]);
|
||||
q_idx = 1;
|
||||
}
|
||||
// Inverse transform.
|
||||
inverse_tx(dqcoeff, 1, tx_type_1d, dst);
|
||||
// Move to the next row.
|
||||
coeff += tx1d_width;
|
||||
qcoeff += tx1d_width;
|
||||
dqcoeff += tx1d_width;
|
||||
src_diff += diff_stride;
|
||||
dst += dst_stride;
|
||||
src += src_stride;
|
||||
}
|
||||
}
|
||||
|
||||
static void process_block_dpcm_horz(TX_SIZE tx_size, TX_TYPE_1D tx_type_1d,
|
||||
struct macroblockd_plane *const pd,
|
||||
struct macroblock_plane *const p,
|
||||
uint8_t *src, int src_stride, uint8_t *dst,
|
||||
int dst_stride, int16_t *src_diff,
|
||||
int diff_stride, tran_low_t *coeff,
|
||||
tran_low_t *qcoeff, tran_low_t *dqcoeff) {
|
||||
const int tx1d_height = tx_size_high[tx_size];
|
||||
dpcm_fwd_tx_func forward_tx = get_dpcm_fwd_tx_func(tx1d_height);
|
||||
dpcm_inv_txfm_add_func inverse_tx =
|
||||
av1_get_dpcm_inv_txfm_add_func(tx1d_height);
|
||||
const int tx1d_width = tx_size_wide[tx_size];
|
||||
const int log_scale = av1_get_tx_scale(tx_size);
|
||||
int q_idx = 0;
|
||||
for (int c = 0; c < tx1d_width; ++c) {
|
||||
for (int r = 0; r < tx1d_height; ++r) {
|
||||
// Update prediction.
|
||||
if (c > 0) dst[r * dst_stride] = dst[r * dst_stride - 1];
|
||||
// Subtraction.
|
||||
src_diff[r * diff_stride] = src[r * src_stride] - dst[r * dst_stride];
|
||||
}
|
||||
// Forward transform.
|
||||
tran_low_t tx_buff[64];
|
||||
forward_tx(src_diff, diff_stride, tx_type_1d, tx_buff);
|
||||
for (int r = 0; r < tx1d_height; ++r) coeff[r * tx1d_width] = tx_buff[r];
|
||||
// Quantization.
|
||||
for (int r = 0; r < tx1d_height; ++r) {
|
||||
quantize_scaler(coeff[r * tx1d_width], p->zbin[q_idx], p->round[q_idx],
|
||||
p->quant[q_idx], p->quant_shift[q_idx],
|
||||
pd->dequant[q_idx], log_scale, &qcoeff[r * tx1d_width],
|
||||
&dqcoeff[r * tx1d_width]);
|
||||
q_idx = 1;
|
||||
}
|
||||
// Inverse transform.
|
||||
for (int r = 0; r < tx1d_height; ++r) tx_buff[r] = dqcoeff[r * tx1d_width];
|
||||
inverse_tx(tx_buff, dst_stride, tx_type_1d, dst);
|
||||
// Move to the next column.
|
||||
++coeff, ++qcoeff, ++dqcoeff, ++src_diff, ++dst, ++src;
|
||||
}
|
||||
}
|
||||
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
static void hbd_process_block_dpcm_vert(
|
||||
TX_SIZE tx_size, TX_TYPE_1D tx_type_1d, int bd,
|
||||
struct macroblockd_plane *const pd, struct macroblock_plane *const p,
|
||||
uint8_t *src8, int src_stride, uint8_t *dst8, int dst_stride,
|
||||
int16_t *src_diff, int diff_stride, tran_low_t *coeff, tran_low_t *qcoeff,
|
||||
tran_low_t *dqcoeff) {
|
||||
const int tx1d_width = tx_size_wide[tx_size];
|
||||
dpcm_fwd_tx_func forward_tx = get_dpcm_fwd_tx_func(tx1d_width);
|
||||
hbd_dpcm_inv_txfm_add_func inverse_tx =
|
||||
av1_get_hbd_dpcm_inv_txfm_add_func(tx1d_width);
|
||||
uint16_t *src = CONVERT_TO_SHORTPTR(src8);
|
||||
uint16_t *dst = CONVERT_TO_SHORTPTR(dst8);
|
||||
const int tx1d_height = tx_size_high[tx_size];
|
||||
const int log_scale = av1_get_tx_scale(tx_size);
|
||||
int q_idx = 0;
|
||||
for (int r = 0; r < tx1d_height; ++r) {
|
||||
// Update prediction.
|
||||
if (r > 0) memcpy(dst, dst - dst_stride, tx1d_width * sizeof(dst[0]));
|
||||
// Subtraction.
|
||||
for (int c = 0; c < tx1d_width; ++c) src_diff[c] = src[c] - dst[c];
|
||||
// Forward transform.
|
||||
forward_tx(src_diff, 1, tx_type_1d, coeff);
|
||||
// Quantization.
|
||||
for (int c = 0; c < tx1d_width; ++c) {
|
||||
quantize_scaler(coeff[c], p->zbin[q_idx], p->round[q_idx],
|
||||
p->quant[q_idx], p->quant_shift[q_idx],
|
||||
pd->dequant[q_idx], log_scale, &qcoeff[c], &dqcoeff[c]);
|
||||
q_idx = 1;
|
||||
}
|
||||
// Inverse transform.
|
||||
inverse_tx(dqcoeff, 1, tx_type_1d, bd, dst);
|
||||
// Move to the next row.
|
||||
coeff += tx1d_width;
|
||||
qcoeff += tx1d_width;
|
||||
dqcoeff += tx1d_width;
|
||||
src_diff += diff_stride;
|
||||
dst += dst_stride;
|
||||
src += src_stride;
|
||||
}
|
||||
}
|
||||
|
||||
static void hbd_process_block_dpcm_horz(
|
||||
TX_SIZE tx_size, TX_TYPE_1D tx_type_1d, int bd,
|
||||
struct macroblockd_plane *const pd, struct macroblock_plane *const p,
|
||||
uint8_t *src8, int src_stride, uint8_t *dst8, int dst_stride,
|
||||
int16_t *src_diff, int diff_stride, tran_low_t *coeff, tran_low_t *qcoeff,
|
||||
tran_low_t *dqcoeff) {
|
||||
const int tx1d_height = tx_size_high[tx_size];
|
||||
dpcm_fwd_tx_func forward_tx = get_dpcm_fwd_tx_func(tx1d_height);
|
||||
hbd_dpcm_inv_txfm_add_func inverse_tx =
|
||||
av1_get_hbd_dpcm_inv_txfm_add_func(tx1d_height);
|
||||
uint16_t *src = CONVERT_TO_SHORTPTR(src8);
|
||||
uint16_t *dst = CONVERT_TO_SHORTPTR(dst8);
|
||||
const int tx1d_width = tx_size_wide[tx_size];
|
||||
const int log_scale = av1_get_tx_scale(tx_size);
|
||||
int q_idx = 0;
|
||||
for (int c = 0; c < tx1d_width; ++c) {
|
||||
for (int r = 0; r < tx1d_height; ++r) {
|
||||
// Update prediction.
|
||||
if (c > 0) dst[r * dst_stride] = dst[r * dst_stride - 1];
|
||||
// Subtraction.
|
||||
src_diff[r * diff_stride] = src[r * src_stride] - dst[r * dst_stride];
|
||||
}
|
||||
// Forward transform.
|
||||
tran_low_t tx_buff[64];
|
||||
forward_tx(src_diff, diff_stride, tx_type_1d, tx_buff);
|
||||
for (int r = 0; r < tx1d_height; ++r) coeff[r * tx1d_width] = tx_buff[r];
|
||||
// Quantization.
|
||||
for (int r = 0; r < tx1d_height; ++r) {
|
||||
quantize_scaler(coeff[r * tx1d_width], p->zbin[q_idx], p->round[q_idx],
|
||||
p->quant[q_idx], p->quant_shift[q_idx],
|
||||
pd->dequant[q_idx], log_scale, &qcoeff[r * tx1d_width],
|
||||
&dqcoeff[r * tx1d_width]);
|
||||
q_idx = 1;
|
||||
}
|
||||
// Inverse transform.
|
||||
for (int r = 0; r < tx1d_height; ++r) tx_buff[r] = dqcoeff[r * tx1d_width];
|
||||
inverse_tx(tx_buff, dst_stride, tx_type_1d, bd, dst);
|
||||
// Move to the next column.
|
||||
++coeff, ++qcoeff, ++dqcoeff, ++src_diff, ++dst, ++src;
|
||||
}
|
||||
}
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
|
||||
void av1_encode_block_intra_dpcm(const AV1_COMMON *cm, MACROBLOCK *x,
|
||||
PREDICTION_MODE mode, int plane, int block,
|
||||
int blk_row, int blk_col,
|
||||
BLOCK_SIZE plane_bsize, TX_SIZE tx_size,
|
||||
TX_TYPE tx_type, ENTROPY_CONTEXT *ta,
|
||||
ENTROPY_CONTEXT *tl, int8_t *skip) {
|
||||
MACROBLOCKD *const xd = &x->e_mbd;
|
||||
struct macroblock_plane *const p = &x->plane[plane];
|
||||
struct macroblockd_plane *const pd = &xd->plane[plane];
|
||||
tran_low_t *dqcoeff = BLOCK_OFFSET(pd->dqcoeff, block);
|
||||
const int diff_stride = block_size_wide[plane_bsize];
|
||||
const int src_stride = p->src.stride;
|
||||
const int dst_stride = pd->dst.stride;
|
||||
const int tx1d_width = tx_size_wide[tx_size];
|
||||
const int tx1d_height = tx_size_high[tx_size];
|
||||
const SCAN_ORDER *const scan_order = get_scan(cm, tx_size, tx_type, 0);
|
||||
tran_low_t *coeff = BLOCK_OFFSET(p->coeff, block);
|
||||
tran_low_t *qcoeff = BLOCK_OFFSET(p->qcoeff, block);
|
||||
uint8_t *dst =
|
||||
&pd->dst.buf[(blk_row * dst_stride + blk_col) << tx_size_wide_log2[0]];
|
||||
uint8_t *src =
|
||||
&p->src.buf[(blk_row * src_stride + blk_col) << tx_size_wide_log2[0]];
|
||||
int16_t *src_diff =
|
||||
&p->src_diff[(blk_row * diff_stride + blk_col) << tx_size_wide_log2[0]];
|
||||
uint16_t *eob = &p->eobs[block];
|
||||
*eob = 0;
|
||||
memset(qcoeff, 0, tx1d_height * tx1d_width * sizeof(*qcoeff));
|
||||
memset(dqcoeff, 0, tx1d_height * tx1d_width * sizeof(*dqcoeff));
|
||||
|
||||
if (LIKELY(!x->skip_block)) {
|
||||
TX_TYPE_1D tx_type_1d = DCT_1D;
|
||||
switch (tx_type) {
|
||||
case IDTX: tx_type_1d = IDTX_1D; break;
|
||||
case V_DCT:
|
||||
assert(mode == H_PRED);
|
||||
tx_type_1d = DCT_1D;
|
||||
break;
|
||||
case H_DCT:
|
||||
assert(mode == V_PRED);
|
||||
tx_type_1d = DCT_1D;
|
||||
break;
|
||||
default: assert(0);
|
||||
}
|
||||
switch (mode) {
|
||||
case V_PRED:
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
if (xd->cur_buf->flags & YV12_FLAG_HIGHBITDEPTH) {
|
||||
hbd_process_block_dpcm_vert(tx_size, tx_type_1d, xd->bd, pd, p, src,
|
||||
src_stride, dst, dst_stride, src_diff,
|
||||
diff_stride, coeff, qcoeff, dqcoeff);
|
||||
} else {
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
process_block_dpcm_vert(tx_size, tx_type_1d, pd, p, src, src_stride,
|
||||
dst, dst_stride, src_diff, diff_stride, coeff,
|
||||
qcoeff, dqcoeff);
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
}
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
break;
|
||||
case H_PRED:
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
if (xd->cur_buf->flags & YV12_FLAG_HIGHBITDEPTH) {
|
||||
hbd_process_block_dpcm_horz(tx_size, tx_type_1d, xd->bd, pd, p, src,
|
||||
src_stride, dst, dst_stride, src_diff,
|
||||
diff_stride, coeff, qcoeff, dqcoeff);
|
||||
} else {
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
process_block_dpcm_horz(tx_size, tx_type_1d, pd, p, src, src_stride,
|
||||
dst, dst_stride, src_diff, diff_stride, coeff,
|
||||
qcoeff, dqcoeff);
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
}
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
break;
|
||||
default: assert(0);
|
||||
}
|
||||
*eob = get_eob(qcoeff, tx1d_height * tx1d_width, scan_order->scan);
|
||||
}
|
||||
|
||||
ta[blk_col] = tl[blk_row] = *eob > 0;
|
||||
if (*eob) *skip = 0;
|
||||
}
|
||||
#endif // CONFIG_DPCM_INTRA
|
||||
|
||||
void av1_encode_block_intra(int plane, int block, int blk_row, int blk_col,
|
||||
BLOCK_SIZE plane_bsize, TX_SIZE tx_size,
|
||||
void *arg) {
|
||||
|
|
@ -1449,7 +1716,33 @@ void av1_encode_block_intra(int plane, int block, int blk_row, int blk_col,
|
|||
const int dst_stride = pd->dst.stride;
|
||||
uint8_t *dst =
|
||||
&pd->dst.buf[(blk_row * dst_stride + blk_col) << tx_size_wide_log2[0]];
|
||||
#if CONFIG_CFL
|
||||
|
||||
#if CONFIG_EC_ADAPT
|
||||
FRAME_CONTEXT *const ec_ctx = xd->tile_ctx;
|
||||
#else
|
||||
FRAME_CONTEXT *const ec_ctx = cm->fc;
|
||||
#endif // CONFIG_EC_ADAPT
|
||||
|
||||
av1_predict_intra_block_encoder_facade(x, ec_ctx, plane, block, blk_col,
|
||||
blk_row, tx_size, plane_bsize);
|
||||
#else
|
||||
av1_predict_intra_block_facade(xd, plane, block, blk_col, blk_row, tx_size);
|
||||
#endif
|
||||
|
||||
#if CONFIG_DPCM_INTRA
|
||||
const int block_raster_idx = av1_block_index_to_raster_order(tx_size, block);
|
||||
const MB_MODE_INFO *const mbmi = &xd->mi[0]->mbmi;
|
||||
const PREDICTION_MODE mode =
|
||||
(plane == 0) ? get_y_mode(xd->mi[0], block_raster_idx) : mbmi->uv_mode;
|
||||
if (av1_use_dpcm_intra(plane, mode, tx_type, mbmi)) {
|
||||
av1_encode_block_intra_dpcm(cm, x, mode, plane, block, blk_row, blk_col,
|
||||
plane_bsize, tx_size, tx_type, args->ta,
|
||||
args->tl, args->skip);
|
||||
return;
|
||||
}
|
||||
#endif // CONFIG_DPCM_INTRA
|
||||
|
||||
av1_subtract_txb(x, plane, plane_bsize, blk_col, blk_row, tx_size);
|
||||
|
||||
const ENTROPY_CONTEXT *a = &args->ta[blk_col];
|
||||
|
|
@ -1458,9 +1751,7 @@ void av1_encode_block_intra(int plane, int block, int blk_row, int blk_col,
|
|||
if (args->enable_optimize_b) {
|
||||
av1_xform_quant(cm, x, plane, block, blk_row, blk_col, plane_bsize, tx_size,
|
||||
ctx, AV1_XFORM_QUANT_FP);
|
||||
if (p->eobs[block]) {
|
||||
av1_optimize_b(cm, x, plane, block, tx_size, ctx);
|
||||
}
|
||||
av1_optimize_b(cm, x, plane, block, plane_bsize, tx_size, a, l);
|
||||
} else {
|
||||
av1_xform_quant(cm, x, plane, block, blk_row, blk_col, plane_bsize, tx_size,
|
||||
ctx, AV1_XFORM_QUANT_B);
|
||||
|
|
@ -1480,16 +1771,216 @@ void av1_encode_block_intra(int plane, int block, int blk_row, int blk_col,
|
|||
// Note : *(args->skip) == mbmi->skip
|
||||
#endif
|
||||
#if CONFIG_CFL
|
||||
MB_MODE_INFO *mbmi = &xd->mi[0]->mbmi;
|
||||
if (plane == AOM_PLANE_Y && x->cfl_store_y) {
|
||||
cfl_store(xd->cfl, dst, dst_stride, blk_row, blk_col, tx_size);
|
||||
}
|
||||
|
||||
if (mbmi->uv_mode == DC_PRED) {
|
||||
// TODO(ltrudeau) find a cleaner way to detect last transform block
|
||||
if (plane == AOM_PLANE_U) {
|
||||
xd->cfl->num_tx_blk[CFL_PRED_U] =
|
||||
(blk_row == 0 && blk_col == 0) ? 1
|
||||
: xd->cfl->num_tx_blk[CFL_PRED_U] + 1;
|
||||
}
|
||||
|
||||
if (plane == AOM_PLANE_V) {
|
||||
xd->cfl->num_tx_blk[CFL_PRED_V] =
|
||||
(blk_row == 0 && blk_col == 0) ? 1
|
||||
: xd->cfl->num_tx_blk[CFL_PRED_V] + 1;
|
||||
|
||||
if (mbmi->skip &&
|
||||
xd->cfl->num_tx_blk[CFL_PRED_U] == xd->cfl->num_tx_blk[CFL_PRED_V]) {
|
||||
assert(plane_bsize != BLOCK_INVALID);
|
||||
const int block_width = block_size_wide[plane_bsize];
|
||||
const int block_height = block_size_high[plane_bsize];
|
||||
|
||||
// if SKIP is chosen at the block level, and ind != 0, we must change
|
||||
// the prediction
|
||||
if (mbmi->cfl_alpha_idx != 0) {
|
||||
const struct macroblockd_plane *const pd_cb = &xd->plane[AOM_PLANE_U];
|
||||
uint8_t *const dst_cb = pd_cb->dst.buf;
|
||||
const int dst_stride_cb = pd_cb->dst.stride;
|
||||
uint8_t *const dst_cr = pd->dst.buf;
|
||||
const int dst_stride_cr = pd->dst.stride;
|
||||
for (int j = 0; j < block_height; j++) {
|
||||
for (int i = 0; i < block_width; i++) {
|
||||
dst_cb[dst_stride_cb * j + i] =
|
||||
(uint8_t)(xd->cfl->dc_pred[CFL_PRED_U] + 0.5);
|
||||
dst_cr[dst_stride_cr * j + i] =
|
||||
(uint8_t)(xd->cfl->dc_pred[CFL_PRED_V] + 0.5);
|
||||
}
|
||||
}
|
||||
mbmi->cfl_alpha_idx = 0;
|
||||
mbmi->cfl_alpha_signs[CFL_PRED_U] = CFL_SIGN_POS;
|
||||
mbmi->cfl_alpha_signs[CFL_PRED_V] = CFL_SIGN_POS;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
#if CONFIG_CFL
|
||||
static int cfl_alpha_dist(const uint8_t *y_pix, int y_stride, double y_avg,
|
||||
const uint8_t *src, int src_stride, int blk_width,
|
||||
int blk_height, double dc_pred, double alpha,
|
||||
int *dist_neg_out) {
|
||||
const double dc_pred_bias = dc_pred + 0.5;
|
||||
int dist = 0;
|
||||
int diff;
|
||||
|
||||
if (alpha == 0.0) {
|
||||
const int dc_pred_i = (int)dc_pred_bias;
|
||||
for (int j = 0; j < blk_height; j++) {
|
||||
for (int i = 0; i < blk_width; i++) {
|
||||
diff = src[i] - dc_pred_i;
|
||||
dist += diff * diff;
|
||||
}
|
||||
src += src_stride;
|
||||
}
|
||||
|
||||
if (dist_neg_out) *dist_neg_out = dist;
|
||||
|
||||
return dist;
|
||||
}
|
||||
|
||||
int dist_neg = 0;
|
||||
for (int j = 0; j < blk_height; j++) {
|
||||
for (int i = 0; i < blk_width; i++) {
|
||||
const double scaled_luma = alpha * (y_pix[i] - y_avg);
|
||||
const int uv = src[i];
|
||||
diff = uv - (int)(scaled_luma + dc_pred_bias);
|
||||
dist += diff * diff;
|
||||
diff = uv + (int)(scaled_luma - dc_pred_bias);
|
||||
dist_neg += diff * diff;
|
||||
}
|
||||
y_pix += y_stride;
|
||||
src += src_stride;
|
||||
}
|
||||
|
||||
if (dist_neg_out) *dist_neg_out = dist_neg;
|
||||
|
||||
return dist;
|
||||
}
|
||||
|
||||
static int cfl_compute_alpha_ind(MACROBLOCK *const x, const CFL_CTX *const cfl,
|
||||
BLOCK_SIZE bsize,
|
||||
CFL_SIGN_TYPE signs_out[CFL_SIGNS]) {
|
||||
const struct macroblock_plane *const p_u = &x->plane[AOM_PLANE_U];
|
||||
const struct macroblock_plane *const p_v = &x->plane[AOM_PLANE_V];
|
||||
const uint8_t *const src_u = p_u->src.buf;
|
||||
const uint8_t *const src_v = p_v->src.buf;
|
||||
const int src_stride_u = p_u->src.stride;
|
||||
const int src_stride_v = p_v->src.stride;
|
||||
const int block_width = block_size_wide[bsize];
|
||||
const int block_height = block_size_high[bsize];
|
||||
const double dc_pred_u = cfl->dc_pred[CFL_PRED_U];
|
||||
const double dc_pred_v = cfl->dc_pred[CFL_PRED_V];
|
||||
|
||||
// Temporary pixel buffer used to store the CfL prediction when we compute the
|
||||
// alpha index.
|
||||
uint8_t tmp_pix[MAX_SB_SQUARE];
|
||||
// Load CfL Prediction over the entire block
|
||||
const double y_avg =
|
||||
cfl_load(cfl, tmp_pix, MAX_SB_SIZE, 0, 0, block_width, block_height);
|
||||
|
||||
int sse[CFL_PRED_PLANES][CFL_MAGS_SIZE];
|
||||
sse[CFL_PRED_U][0] =
|
||||
cfl_alpha_dist(tmp_pix, MAX_SB_SIZE, y_avg, src_u, src_stride_u,
|
||||
block_width, block_height, dc_pred_u, 0, NULL);
|
||||
sse[CFL_PRED_V][0] =
|
||||
cfl_alpha_dist(tmp_pix, MAX_SB_SIZE, y_avg, src_v, src_stride_v,
|
||||
block_width, block_height, dc_pred_v, 0, NULL);
|
||||
for (int m = 1; m < CFL_MAGS_SIZE; m += 2) {
|
||||
assert(cfl_alpha_mags[m + 1] == -cfl_alpha_mags[m]);
|
||||
sse[CFL_PRED_U][m] = cfl_alpha_dist(
|
||||
tmp_pix, MAX_SB_SIZE, y_avg, src_u, src_stride_u, block_width,
|
||||
block_height, dc_pred_u, cfl_alpha_mags[m], &sse[CFL_PRED_U][m + 1]);
|
||||
sse[CFL_PRED_V][m] = cfl_alpha_dist(
|
||||
tmp_pix, MAX_SB_SIZE, y_avg, src_v, src_stride_v, block_width,
|
||||
block_height, dc_pred_v, cfl_alpha_mags[m], &sse[CFL_PRED_V][m + 1]);
|
||||
}
|
||||
|
||||
int dist;
|
||||
int64_t cost;
|
||||
int64_t best_cost;
|
||||
|
||||
// Compute least squares parameter of the entire block
|
||||
// IMPORTANT: We assume that the first code is 0,0
|
||||
int ind = 0;
|
||||
signs_out[CFL_PRED_U] = CFL_SIGN_POS;
|
||||
signs_out[CFL_PRED_V] = CFL_SIGN_POS;
|
||||
|
||||
dist = sse[CFL_PRED_U][0] + sse[CFL_PRED_V][0];
|
||||
dist *= 16;
|
||||
best_cost = RDCOST(x->rdmult, x->rddiv, cfl->costs[0], dist);
|
||||
|
||||
for (int c = 1; c < CFL_ALPHABET_SIZE; c++) {
|
||||
const int idx_u = cfl_alpha_codes[c][CFL_PRED_U];
|
||||
const int idx_v = cfl_alpha_codes[c][CFL_PRED_V];
|
||||
for (CFL_SIGN_TYPE sign_u = idx_u == 0; sign_u < CFL_SIGNS; sign_u++) {
|
||||
for (CFL_SIGN_TYPE sign_v = idx_v == 0; sign_v < CFL_SIGNS; sign_v++) {
|
||||
dist = sse[CFL_PRED_U][idx_u + (sign_u == CFL_SIGN_NEG)] +
|
||||
sse[CFL_PRED_V][idx_v + (sign_v == CFL_SIGN_NEG)];
|
||||
dist *= 16;
|
||||
cost = RDCOST(x->rdmult, x->rddiv, cfl->costs[c], dist);
|
||||
if (cost < best_cost) {
|
||||
best_cost = cost;
|
||||
ind = c;
|
||||
signs_out[CFL_PRED_U] = sign_u;
|
||||
signs_out[CFL_PRED_V] = sign_v;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return ind;
|
||||
}
|
||||
|
||||
static inline void cfl_update_costs(CFL_CTX *cfl, FRAME_CONTEXT *ec_ctx) {
|
||||
assert(ec_ctx->cfl_alpha_cdf[CFL_ALPHABET_SIZE - 1] ==
|
||||
AOM_ICDF(CDF_PROB_TOP));
|
||||
const int prob_den = CDF_PROB_TOP;
|
||||
|
||||
int prob_num = AOM_ICDF(ec_ctx->cfl_alpha_cdf[0]);
|
||||
cfl->costs[0] = av1_cost_zero(get_prob(prob_num, prob_den));
|
||||
|
||||
for (int c = 1; c < CFL_ALPHABET_SIZE; c++) {
|
||||
int sign_bit_cost = (cfl_alpha_codes[c][CFL_PRED_U] != 0) +
|
||||
(cfl_alpha_codes[c][CFL_PRED_V] != 0);
|
||||
prob_num = AOM_ICDF(ec_ctx->cfl_alpha_cdf[c]) -
|
||||
AOM_ICDF(ec_ctx->cfl_alpha_cdf[c - 1]);
|
||||
cfl->costs[c] = av1_cost_zero(get_prob(prob_num, prob_den)) +
|
||||
av1_cost_literal(sign_bit_cost);
|
||||
}
|
||||
}
|
||||
|
||||
void av1_predict_intra_block_encoder_facade(MACROBLOCK *x,
|
||||
FRAME_CONTEXT *ec_ctx, int plane,
|
||||
int block_idx, int blk_col,
|
||||
int blk_row, TX_SIZE tx_size,
|
||||
BLOCK_SIZE plane_bsize) {
|
||||
MACROBLOCKD *const xd = &x->e_mbd;
|
||||
MB_MODE_INFO *mbmi = &xd->mi[0]->mbmi;
|
||||
if (plane != AOM_PLANE_Y && mbmi->uv_mode == DC_PRED) {
|
||||
if (blk_col == 0 && blk_row == 0 && plane == AOM_PLANE_U) {
|
||||
CFL_CTX *const cfl = xd->cfl;
|
||||
cfl_update_costs(cfl, ec_ctx);
|
||||
cfl_dc_pred(xd, plane_bsize, tx_size);
|
||||
mbmi->cfl_alpha_idx =
|
||||
cfl_compute_alpha_ind(x, cfl, plane_bsize, mbmi->cfl_alpha_signs);
|
||||
}
|
||||
}
|
||||
av1_predict_intra_block_facade(xd, plane, block_idx, blk_col, blk_row,
|
||||
tx_size);
|
||||
}
|
||||
#endif
|
||||
|
||||
void av1_encode_intra_block_plane(AV1_COMMON *cm, MACROBLOCK *x,
|
||||
BLOCK_SIZE bsize, int plane,
|
||||
int enable_optimize_b, const int mi_row,
|
||||
const int mi_col) {
|
||||
int enable_optimize_b, int mi_row,
|
||||
int mi_col) {
|
||||
const MACROBLOCKD *const xd = &x->e_mbd;
|
||||
ENTROPY_CONTEXT ta[2 * MAX_MIB_SIZE] = { 0 };
|
||||
ENTROPY_CONTEXT tl[2 * MAX_MIB_SIZE] = { 0 };
|
||||
|
|
@ -1545,9 +2036,7 @@ PVQ_SKIP_TYPE av1_pvq_encode_helper(MACROBLOCK *x, tran_low_t *const coeff,
|
|||
DECLARE_ALIGNED(16, int32_t, ref_int32[OD_TXSIZE_MAX * OD_TXSIZE_MAX]);
|
||||
DECLARE_ALIGNED(16, int32_t, out_int32[OD_TXSIZE_MAX * OD_TXSIZE_MAX]);
|
||||
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
hbd_downshift = x->e_mbd.bd - 8;
|
||||
#endif
|
||||
|
||||
assert(OD_COEFF_SHIFT >= 4);
|
||||
// DC quantizer for PVQ
|
||||
|
|
@ -1563,10 +2052,10 @@ PVQ_SKIP_TYPE av1_pvq_encode_helper(MACROBLOCK *x, tran_low_t *const coeff,
|
|||
|
||||
*eob = 0;
|
||||
|
||||
#if CONFIG_DAALA_EC
|
||||
#if !CONFIG_ANS
|
||||
tell = od_ec_enc_tell_frac(&daala_enc->w.ec);
|
||||
#else
|
||||
#error "CONFIG_PVQ currently requires CONFIG_DAALA_EC."
|
||||
#error "CONFIG_PVQ currently requires !CONFIG_ANS."
|
||||
#endif
|
||||
|
||||
// Change coefficient ordering for pvq encoding.
|
||||
|
|
@ -1635,11 +2124,11 @@ PVQ_SKIP_TYPE av1_pvq_encode_helper(MACROBLOCK *x, tran_low_t *const coeff,
|
|||
|
||||
*eob = tx_blk_size * tx_blk_size;
|
||||
|
||||
#if CONFIG_DAALA_EC
|
||||
#if !CONFIG_ANS
|
||||
*rate = (od_ec_enc_tell_frac(&daala_enc->w.ec) - tell)
|
||||
<< (AV1_PROB_COST_SHIFT - OD_BITRES);
|
||||
#else
|
||||
#error "CONFIG_PVQ currently requires CONFIG_DAALA_EC."
|
||||
#error "CONFIG_PVQ currently requires !CONFIG_ANS."
|
||||
#endif
|
||||
assert(*rate >= 0);
|
||||
|
||||
|
|
|
|||
20
third_party/aom/av1/encoder/encodemb.h
vendored
20
third_party/aom/av1/encoder/encodemb.h
vendored
|
|
@ -54,7 +54,8 @@ void av1_xform_quant(const AV1_COMMON *cm, MACROBLOCK *x, int plane, int block,
|
|||
TX_SIZE tx_size, int ctx, AV1_XFORM_QUANT xform_quant_idx);
|
||||
|
||||
int av1_optimize_b(const AV1_COMMON *cm, MACROBLOCK *mb, int plane, int block,
|
||||
TX_SIZE tx_size, int ctx);
|
||||
BLOCK_SIZE plane_bsize, TX_SIZE tx_size,
|
||||
const ENTROPY_CONTEXT *a, const ENTROPY_CONTEXT *l);
|
||||
|
||||
void av1_subtract_txb(MACROBLOCK *x, int plane, BLOCK_SIZE plane_bsize,
|
||||
int blk_col, int blk_row, TX_SIZE tx_size);
|
||||
|
|
@ -85,6 +86,23 @@ void av1_store_pvq_enc_info(PVQ_INFO *pvq_info, int *qg, int *theta, int *k,
|
|||
int *size, int skip_rest, int skip_dir, int bs);
|
||||
#endif
|
||||
|
||||
#if CONFIG_CFL
|
||||
void av1_predict_intra_block_encoder_facade(MACROBLOCK *x,
|
||||
FRAME_CONTEXT *ec_ctx, int plane,
|
||||
int block_idx, int blk_col,
|
||||
int blk_row, TX_SIZE tx_size,
|
||||
BLOCK_SIZE plane_bsize);
|
||||
#endif
|
||||
|
||||
#if CONFIG_DPCM_INTRA
|
||||
void av1_encode_block_intra_dpcm(const AV1_COMMON *cm, MACROBLOCK *x,
|
||||
PREDICTION_MODE mode, int plane, int block,
|
||||
int blk_row, int blk_col,
|
||||
BLOCK_SIZE plane_bsize, TX_SIZE tx_size,
|
||||
TX_TYPE tx_type, ENTROPY_CONTEXT *ta,
|
||||
ENTROPY_CONTEXT *tl, int8_t *skip);
|
||||
#endif // CONFIG_DPCM_INTRA
|
||||
|
||||
#ifdef __cplusplus
|
||||
} // extern "C"
|
||||
#endif
|
||||
|
|
|
|||
124
third_party/aom/av1/encoder/encodemv.c
vendored
124
third_party/aom/av1/encoder/encodemv.c
vendored
|
|
@ -45,13 +45,8 @@ static void encode_mv_component(aom_writer *w, int comp, nmv_component *mvcomp,
|
|||
// Sign
|
||||
aom_write(w, sign, mvcomp->sign);
|
||||
|
||||
// Class
|
||||
#if CONFIG_EC_MULTISYMBOL
|
||||
// Class
|
||||
aom_write_symbol(w, mv_class, mvcomp->class_cdf, MV_CLASSES);
|
||||
#else
|
||||
av1_write_token(w, av1_mv_class_tree, mvcomp->classes,
|
||||
&mv_class_encodings[mv_class]);
|
||||
#endif
|
||||
|
||||
// Integer bits
|
||||
if (mv_class == MV_CLASS_0) {
|
||||
|
|
@ -62,16 +57,10 @@ static void encode_mv_component(aom_writer *w, int comp, nmv_component *mvcomp,
|
|||
for (i = 0; i < n; ++i) aom_write(w, (d >> i) & 1, mvcomp->bits[i]);
|
||||
}
|
||||
|
||||
// Fractional bits
|
||||
#if CONFIG_EC_MULTISYMBOL
|
||||
// Fractional bits
|
||||
aom_write_symbol(
|
||||
w, fr, mv_class == MV_CLASS_0 ? mvcomp->class0_fp_cdf[d] : mvcomp->fp_cdf,
|
||||
MV_FP_SIZE);
|
||||
#else
|
||||
av1_write_token(w, av1_mv_fp_tree,
|
||||
mv_class == MV_CLASS_0 ? mvcomp->class0_fp[d] : mvcomp->fp,
|
||||
&mv_fp_encodings[fr]);
|
||||
#endif
|
||||
|
||||
// High precision bit
|
||||
if (usehp)
|
||||
|
|
@ -171,7 +160,6 @@ static void write_mv_update(const aom_tree_index *tree,
|
|||
void av1_write_nmv_probs(AV1_COMMON *cm, int usehp, aom_writer *w,
|
||||
nmv_context_counts *const nmv_counts) {
|
||||
int i;
|
||||
#if CONFIG_REF_MV
|
||||
int nmv_ctx = 0;
|
||||
for (nmv_ctx = 0; nmv_ctx < NMV_CONTEXTS; ++nmv_ctx) {
|
||||
nmv_context *const mvc = &cm->fc->nmvc[nmv_ctx];
|
||||
|
|
@ -213,57 +201,13 @@ void av1_write_nmv_probs(AV1_COMMON *cm, int usehp, aom_writer *w,
|
|||
}
|
||||
}
|
||||
}
|
||||
#else
|
||||
nmv_context *const mvc = &cm->fc->nmvc;
|
||||
nmv_context_counts *const counts = nmv_counts;
|
||||
|
||||
#if !CONFIG_EC_ADAPT
|
||||
write_mv_update(av1_mv_joint_tree, mvc->joints, counts->joints, MV_JOINTS, w);
|
||||
|
||||
for (i = 0; i < 2; ++i) {
|
||||
int j;
|
||||
nmv_component *comp = &mvc->comps[i];
|
||||
nmv_component_counts *comp_counts = &counts->comps[i];
|
||||
|
||||
update_mv(w, comp_counts->sign, &comp->sign, MV_UPDATE_PROB);
|
||||
write_mv_update(av1_mv_class_tree, comp->classes, comp_counts->classes,
|
||||
MV_CLASSES, w);
|
||||
write_mv_update(av1_mv_class0_tree, comp->class0, comp_counts->class0,
|
||||
CLASS0_SIZE, w);
|
||||
for (j = 0; j < MV_OFFSET_BITS; ++j)
|
||||
update_mv(w, comp_counts->bits[j], &comp->bits[j], MV_UPDATE_PROB);
|
||||
}
|
||||
|
||||
for (i = 0; i < 2; ++i) {
|
||||
int j;
|
||||
for (j = 0; j < CLASS0_SIZE; ++j) {
|
||||
write_mv_update(av1_mv_fp_tree, mvc->comps[i].class0_fp[j],
|
||||
counts->comps[i].class0_fp[j], MV_FP_SIZE, w);
|
||||
}
|
||||
write_mv_update(av1_mv_fp_tree, mvc->comps[i].fp, counts->comps[i].fp,
|
||||
MV_FP_SIZE, w);
|
||||
}
|
||||
#endif // !CONFIG_EC_ADAPT
|
||||
|
||||
if (usehp) {
|
||||
for (i = 0; i < 2; ++i) {
|
||||
update_mv(w, counts->comps[i].class0_hp, &mvc->comps[i].class0_hp,
|
||||
MV_UPDATE_PROB);
|
||||
update_mv(w, counts->comps[i].hp, &mvc->comps[i].hp, MV_UPDATE_PROB);
|
||||
}
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
void av1_encode_mv(AV1_COMP *cpi, aom_writer *w, const MV *mv, const MV *ref,
|
||||
nmv_context *mvctx, int usehp) {
|
||||
const MV diff = { mv->row - ref->row, mv->col - ref->col };
|
||||
const MV_JOINT_TYPE j = av1_get_mv_joint(&diff);
|
||||
#if CONFIG_EC_MULTISYMBOL
|
||||
aom_write_symbol(w, j, mvctx->joint_cdf, MV_JOINTS);
|
||||
#else
|
||||
av1_write_token(w, av1_mv_joint_tree, mvctx->joints, &mv_joint_encodings[j]);
|
||||
#endif
|
||||
if (mv_joint_vertical(j))
|
||||
encode_mv_component(w, diff.row, &mvctx->comps[0], usehp);
|
||||
|
||||
|
|
@ -284,11 +228,7 @@ void av1_encode_dv(aom_writer *w, const MV *mv, const MV *ref,
|
|||
const MV diff = { mv->row - ref->row, mv->col - ref->col };
|
||||
const MV_JOINT_TYPE j = av1_get_mv_joint(&diff);
|
||||
|
||||
#if CONFIG_EC_MULTISYMBOL
|
||||
aom_write_symbol(w, j, mvctx->joint_cdf, MV_JOINTS);
|
||||
#else
|
||||
av1_write_token(w, av1_mv_joint_tree, mvctx->joints, &mv_joint_encodings[j]);
|
||||
#endif
|
||||
if (mv_joint_vertical(j))
|
||||
encode_mv_component(w, diff.row, &mvctx->comps[0], 0);
|
||||
|
||||
|
|
@ -306,135 +246,101 @@ void av1_build_nmv_cost_table(int *mvjoint, int *mvcost[2],
|
|||
|
||||
#if CONFIG_EXT_INTER
|
||||
static void inc_mvs(const MB_MODE_INFO *mbmi, const MB_MODE_INFO_EXT *mbmi_ext,
|
||||
const int_mv mvs[2],
|
||||
#if CONFIG_REF_MV
|
||||
const int_mv pred_mvs[2],
|
||||
#endif
|
||||
const int_mv mvs[2], const int_mv pred_mvs[2],
|
||||
nmv_context_counts *nmv_counts) {
|
||||
int i;
|
||||
PREDICTION_MODE mode = mbmi->mode;
|
||||
#if !CONFIG_REF_MV
|
||||
nmv_context_counts *counts = nmv_counts;
|
||||
#endif
|
||||
|
||||
if (mode == NEWMV || mode == NEW_NEWMV) {
|
||||
for (i = 0; i < 1 + has_second_ref(mbmi); ++i) {
|
||||
const MV *ref = &mbmi_ext->ref_mvs[mbmi->ref_frame[i]][0].as_mv;
|
||||
const MV diff = { mvs[i].as_mv.row - ref->row,
|
||||
mvs[i].as_mv.col - ref->col };
|
||||
#if CONFIG_REF_MV
|
||||
int8_t rf_type = av1_ref_frame_type(mbmi->ref_frame);
|
||||
int nmv_ctx =
|
||||
av1_nmv_ctx(mbmi_ext->ref_mv_count[rf_type],
|
||||
mbmi_ext->ref_mv_stack[rf_type], i, mbmi->ref_mv_idx);
|
||||
nmv_context_counts *counts = &nmv_counts[nmv_ctx];
|
||||
(void)pred_mvs;
|
||||
#endif
|
||||
av1_inc_mv(&diff, counts, 1);
|
||||
}
|
||||
} else if (mode == NEAREST_NEWMV || mode == NEAR_NEWMV) {
|
||||
const MV *ref = &mbmi_ext->ref_mvs[mbmi->ref_frame[1]][0].as_mv;
|
||||
const MV diff = { mvs[1].as_mv.row - ref->row,
|
||||
mvs[1].as_mv.col - ref->col };
|
||||
#if CONFIG_REF_MV
|
||||
int8_t rf_type = av1_ref_frame_type(mbmi->ref_frame);
|
||||
int nmv_ctx =
|
||||
av1_nmv_ctx(mbmi_ext->ref_mv_count[rf_type],
|
||||
mbmi_ext->ref_mv_stack[rf_type], 1, mbmi->ref_mv_idx);
|
||||
nmv_context_counts *counts = &nmv_counts[nmv_ctx];
|
||||
#endif
|
||||
av1_inc_mv(&diff, counts, 1);
|
||||
} else if (mode == NEW_NEARESTMV || mode == NEW_NEARMV) {
|
||||
const MV *ref = &mbmi_ext->ref_mvs[mbmi->ref_frame[0]][0].as_mv;
|
||||
const MV diff = { mvs[0].as_mv.row - ref->row,
|
||||
mvs[0].as_mv.col - ref->col };
|
||||
#if CONFIG_REF_MV
|
||||
int8_t rf_type = av1_ref_frame_type(mbmi->ref_frame);
|
||||
int nmv_ctx =
|
||||
av1_nmv_ctx(mbmi_ext->ref_mv_count[rf_type],
|
||||
mbmi_ext->ref_mv_stack[rf_type], 0, mbmi->ref_mv_idx);
|
||||
nmv_context_counts *counts = &nmv_counts[nmv_ctx];
|
||||
#endif
|
||||
av1_inc_mv(&diff, counts, 1);
|
||||
}
|
||||
}
|
||||
|
||||
static void inc_mvs_sub8x8(const MODE_INFO *mi, int block, const int_mv mvs[2],
|
||||
#if CONFIG_REF_MV
|
||||
const MB_MODE_INFO_EXT *mbmi_ext,
|
||||
#endif
|
||||
nmv_context_counts *nmv_counts) {
|
||||
int i;
|
||||
PREDICTION_MODE mode = mi->bmi[block].as_mode;
|
||||
#if CONFIG_REF_MV
|
||||
const MB_MODE_INFO *mbmi = &mi->mbmi;
|
||||
#else
|
||||
nmv_context_counts *counts = nmv_counts;
|
||||
#endif
|
||||
|
||||
if (mode == NEWMV || mode == NEW_NEWMV) {
|
||||
for (i = 0; i < 1 + has_second_ref(&mi->mbmi); ++i) {
|
||||
const MV *ref = &mi->bmi[block].ref_mv[i].as_mv;
|
||||
const MV diff = { mvs[i].as_mv.row - ref->row,
|
||||
mvs[i].as_mv.col - ref->col };
|
||||
#if CONFIG_REF_MV
|
||||
int8_t rf_type = av1_ref_frame_type(mbmi->ref_frame);
|
||||
int nmv_ctx =
|
||||
av1_nmv_ctx(mbmi_ext->ref_mv_count[rf_type],
|
||||
mbmi_ext->ref_mv_stack[rf_type], i, mbmi->ref_mv_idx);
|
||||
nmv_context_counts *counts = &nmv_counts[nmv_ctx];
|
||||
#endif
|
||||
av1_inc_mv(&diff, counts, 1);
|
||||
}
|
||||
} else if (mode == NEAREST_NEWMV || mode == NEAR_NEWMV) {
|
||||
const MV *ref = &mi->bmi[block].ref_mv[1].as_mv;
|
||||
const MV diff = { mvs[1].as_mv.row - ref->row,
|
||||
mvs[1].as_mv.col - ref->col };
|
||||
#if CONFIG_REF_MV
|
||||
int8_t rf_type = av1_ref_frame_type(mbmi->ref_frame);
|
||||
int nmv_ctx =
|
||||
av1_nmv_ctx(mbmi_ext->ref_mv_count[rf_type],
|
||||
mbmi_ext->ref_mv_stack[rf_type], 1, mbmi->ref_mv_idx);
|
||||
nmv_context_counts *counts = &nmv_counts[nmv_ctx];
|
||||
#endif
|
||||
av1_inc_mv(&diff, counts, 1);
|
||||
} else if (mode == NEW_NEARESTMV || mode == NEW_NEARMV) {
|
||||
const MV *ref = &mi->bmi[block].ref_mv[0].as_mv;
|
||||
const MV diff = { mvs[0].as_mv.row - ref->row,
|
||||
mvs[0].as_mv.col - ref->col };
|
||||
#if CONFIG_REF_MV
|
||||
int8_t rf_type = av1_ref_frame_type(mbmi->ref_frame);
|
||||
int nmv_ctx =
|
||||
av1_nmv_ctx(mbmi_ext->ref_mv_count[rf_type],
|
||||
mbmi_ext->ref_mv_stack[rf_type], 0, mbmi->ref_mv_idx);
|
||||
nmv_context_counts *counts = &nmv_counts[nmv_ctx];
|
||||
#endif
|
||||
av1_inc_mv(&diff, counts, 1);
|
||||
}
|
||||
}
|
||||
#else
|
||||
static void inc_mvs(const MB_MODE_INFO *mbmi, const MB_MODE_INFO_EXT *mbmi_ext,
|
||||
const int_mv mvs[2],
|
||||
#if CONFIG_REF_MV
|
||||
const int_mv pred_mvs[2],
|
||||
#endif
|
||||
const int_mv mvs[2], const int_mv pred_mvs[2],
|
||||
nmv_context_counts *nmv_counts) {
|
||||
int i;
|
||||
#if !CONFIG_REF_MV
|
||||
nmv_context_counts *counts = nmv_counts;
|
||||
#endif
|
||||
|
||||
for (i = 0; i < 1 + has_second_ref(mbmi); ++i) {
|
||||
#if CONFIG_REF_MV
|
||||
int8_t rf_type = av1_ref_frame_type(mbmi->ref_frame);
|
||||
int nmv_ctx =
|
||||
av1_nmv_ctx(mbmi_ext->ref_mv_count[rf_type],
|
||||
mbmi_ext->ref_mv_stack[rf_type], i, mbmi->ref_mv_idx);
|
||||
nmv_context_counts *counts = &nmv_counts[nmv_ctx];
|
||||
const MV *ref = &pred_mvs[i].as_mv;
|
||||
#else
|
||||
const MV *ref = &mbmi_ext->ref_mvs[mbmi->ref_frame[i]][0].as_mv;
|
||||
#endif
|
||||
const MV diff = { mvs[i].as_mv.row - ref->row,
|
||||
mvs[i].as_mv.col - ref->col };
|
||||
av1_inc_mv(&diff, counts, 1);
|
||||
|
|
@ -464,20 +370,11 @@ void av1_update_mv_count(ThreadData *td) {
|
|||
|
||||
#if CONFIG_EXT_INTER
|
||||
if (have_newmv_in_inter_mode(mi->bmi[i].as_mode))
|
||||
inc_mvs_sub8x8(mi, i, mi->bmi[i].as_mv,
|
||||
#if CONFIG_REF_MV
|
||||
mbmi_ext, td->counts->mv);
|
||||
#else
|
||||
&td->counts->mv);
|
||||
#endif
|
||||
inc_mvs_sub8x8(mi, i, mi->bmi[i].as_mv, mbmi_ext, td->counts->mv);
|
||||
#else
|
||||
if (mi->bmi[i].as_mode == NEWMV)
|
||||
inc_mvs(mbmi, mbmi_ext, mi->bmi[i].as_mv,
|
||||
#if CONFIG_REF_MV
|
||||
mi->bmi[i].pred_mv, td->counts->mv);
|
||||
#else
|
||||
&td->counts->mv);
|
||||
#endif
|
||||
inc_mvs(mbmi, mbmi_ext, mi->bmi[i].as_mv, mi->bmi[i].pred_mv,
|
||||
td->counts->mv);
|
||||
#endif // CONFIG_EXT_INTER
|
||||
}
|
||||
}
|
||||
|
|
@ -487,11 +384,6 @@ void av1_update_mv_count(ThreadData *td) {
|
|||
#else
|
||||
if (mbmi->mode == NEWMV)
|
||||
#endif // CONFIG_EXT_INTER
|
||||
inc_mvs(mbmi, mbmi_ext, mbmi->mv,
|
||||
#if CONFIG_REF_MV
|
||||
mbmi->pred_mv, td->counts->mv);
|
||||
#else
|
||||
&td->counts->mv);
|
||||
#endif
|
||||
inc_mvs(mbmi, mbmi_ext, mbmi->mv, mbmi->pred_mv, td->counts->mv);
|
||||
}
|
||||
}
|
||||
|
|
|
|||
727
third_party/aom/av1/encoder/encoder.c
vendored
727
third_party/aom/av1/encoder/encoder.c
vendored
File diff suppressed because it is too large
Load diff
114
third_party/aom/av1/encoder/encoder.h
vendored
114
third_party/aom/av1/encoder/encoder.h
vendored
|
|
@ -37,7 +37,6 @@
|
|||
#include "av1/encoder/rd.h"
|
||||
#include "av1/encoder/speed_features.h"
|
||||
#include "av1/encoder/tokenize.h"
|
||||
#include "av1/encoder/variance_tree.h"
|
||||
#if CONFIG_XIPHRC
|
||||
#include "av1/encoder/ratectrl_xiph.h"
|
||||
#endif
|
||||
|
|
@ -54,15 +53,9 @@ extern "C" {
|
|||
#endif
|
||||
|
||||
typedef struct {
|
||||
int nmvjointcost[MV_JOINTS];
|
||||
int nmvcosts[2][MV_VALS];
|
||||
int nmvcosts_hp[2][MV_VALS];
|
||||
|
||||
#if CONFIG_REF_MV
|
||||
int nmv_vec_cost[NMV_CONTEXTS][MV_JOINTS];
|
||||
int nmv_costs[NMV_CONTEXTS][2][MV_VALS];
|
||||
int nmv_costs_hp[NMV_CONTEXTS][2][MV_VALS];
|
||||
#endif
|
||||
|
||||
// 0 = Intra, Last, GF, ARF
|
||||
signed char last_ref_lf_deltas[TOTAL_REFS_PER_FRAME];
|
||||
|
|
@ -210,6 +203,11 @@ typedef struct AV1EncoderConfig {
|
|||
int scaled_frame_width;
|
||||
int scaled_frame_height;
|
||||
|
||||
#if CONFIG_FRAME_SUPERRES
|
||||
// Frame Super-Resolution size scaling
|
||||
int superres_enabled;
|
||||
#endif // CONFIG_FRAME_SUPERRES
|
||||
|
||||
// Enable feature to reduce the frame quantization every x frames.
|
||||
int frame_periodic_boost;
|
||||
|
||||
|
|
@ -323,9 +321,16 @@ typedef struct ThreadData {
|
|||
PICK_MODE_CONTEXT *leaf_tree;
|
||||
PC_TREE *pc_tree;
|
||||
PC_TREE *pc_root[MAX_MIB_SIZE_LOG2 - MIN_MIB_SIZE_LOG2 + 1];
|
||||
#if CONFIG_MOTION_VAR
|
||||
int32_t *wsrc_buf;
|
||||
int32_t *mask_buf;
|
||||
uint8_t *above_pred_buf;
|
||||
uint8_t *left_pred_buf;
|
||||
#endif
|
||||
|
||||
VAR_TREE *var_tree;
|
||||
VAR_TREE *var_root[MAX_MIB_SIZE_LOG2 - MIN_MIB_SIZE_LOG2 + 1];
|
||||
#if CONFIG_PALETTE
|
||||
PALETTE_BUFFER *palette_buffer;
|
||||
#endif // CONFIG_PALETTE
|
||||
} ThreadData;
|
||||
|
||||
struct EncWorkerData;
|
||||
|
|
@ -350,16 +355,6 @@ typedef struct {
|
|||
YV12_BUFFER_CONFIG buf;
|
||||
} EncRefCntBuffer;
|
||||
|
||||
#if CONFIG_SUBFRAME_PROB_UPDATE
|
||||
typedef struct SUBFRAME_STATS {
|
||||
av1_coeff_probs_model coef_probs_buf[COEF_PROBS_BUFS][TX_SIZES][PLANE_TYPES];
|
||||
av1_coeff_count coef_counts_buf[COEF_PROBS_BUFS][TX_SIZES][PLANE_TYPES];
|
||||
unsigned int eob_counts_buf[COEF_PROBS_BUFS][TX_SIZES][PLANE_TYPES][REF_TYPES]
|
||||
[COEF_BANDS][COEFF_CONTEXTS];
|
||||
av1_coeff_probs_model enc_starting_coef_probs[TX_SIZES][PLANE_TYPES];
|
||||
} SUBFRAME_STATS;
|
||||
#endif // CONFIG_SUBFRAME_PROB_UPDATE
|
||||
|
||||
typedef struct TileBufferEnc {
|
||||
uint8_t *data;
|
||||
size_t size;
|
||||
|
|
@ -369,14 +364,7 @@ typedef struct AV1_COMP {
|
|||
QUANTS quants;
|
||||
ThreadData td;
|
||||
MB_MODE_INFO_EXT *mbmi_ext_base;
|
||||
DECLARE_ALIGNED(16, int16_t, y_dequant[QINDEX_RANGE][8]); // 8: SIMD width
|
||||
DECLARE_ALIGNED(16, int16_t, uv_dequant[QINDEX_RANGE][8]); // 8: SIMD width
|
||||
#if CONFIG_NEW_QUANT
|
||||
DECLARE_ALIGNED(16, dequant_val_type_nuq,
|
||||
y_dequant_val_nuq[QUANT_PROFILES][QINDEX_RANGE][COEF_BANDS]);
|
||||
DECLARE_ALIGNED(16, dequant_val_type_nuq,
|
||||
uv_dequant_val_nuq[QUANT_PROFILES][QINDEX_RANGE][COEF_BANDS]);
|
||||
#endif // CONFIG_NEW_QUANT
|
||||
Dequants dequants;
|
||||
AV1_COMMON common;
|
||||
AV1EncoderConfig oxcf;
|
||||
struct lookahead_ctx *lookahead;
|
||||
|
|
@ -443,15 +431,8 @@ typedef struct AV1_COMP {
|
|||
|
||||
CODING_CONTEXT coding_context;
|
||||
|
||||
#if CONFIG_REF_MV
|
||||
int nmv_costs[NMV_CONTEXTS][2][MV_VALS];
|
||||
int nmv_costs_hp[NMV_CONTEXTS][2][MV_VALS];
|
||||
#endif
|
||||
|
||||
int nmvcosts[2][MV_VALS];
|
||||
int nmvcosts_hp[2][MV_VALS];
|
||||
int nmvsadcosts[2][MV_VALS];
|
||||
int nmvsadcosts_hp[2][MV_VALS];
|
||||
|
||||
int64_t last_time_stamp_seen;
|
||||
int64_t last_end_time_stamp_seen;
|
||||
|
|
@ -543,29 +524,23 @@ typedef struct AV1_COMP {
|
|||
// number of MBs in the current frame when the frame is
|
||||
// scaled.
|
||||
|
||||
// Store frame variance info in SOURCE_VAR_BASED_PARTITION search type.
|
||||
DIFF *source_diff_var;
|
||||
// The threshold used in SOURCE_VAR_BASED_PARTITION search type.
|
||||
unsigned int source_var_thresh;
|
||||
int frames_till_next_var_check;
|
||||
|
||||
int frame_flags;
|
||||
|
||||
search_site_config ss_cfg;
|
||||
|
||||
int mbmode_cost[BLOCK_SIZE_GROUPS][INTRA_MODES];
|
||||
#if CONFIG_REF_MV
|
||||
int newmv_mode_cost[NEWMV_MODE_CONTEXTS][2];
|
||||
int zeromv_mode_cost[ZEROMV_MODE_CONTEXTS][2];
|
||||
int refmv_mode_cost[REFMV_MODE_CONTEXTS][2];
|
||||
int drl_mode_cost0[DRL_MODE_CONTEXTS][2];
|
||||
#endif
|
||||
|
||||
unsigned int inter_mode_cost[INTER_MODE_CONTEXTS][INTER_MODES];
|
||||
#if CONFIG_EXT_INTER
|
||||
unsigned int inter_compound_mode_cost[INTER_MODE_CONTEXTS]
|
||||
[INTER_COMPOUND_MODES];
|
||||
#if CONFIG_INTERINTRA
|
||||
unsigned int interintra_mode_cost[BLOCK_SIZE_GROUPS][INTERINTRA_MODES];
|
||||
#endif // CONFIG_INTERINTRA
|
||||
#endif // CONFIG_EXT_INTER
|
||||
#if CONFIG_MOTION_VAR || CONFIG_WARPED_MOTION
|
||||
int motion_mode_cost[BLOCK_SIZES][MOTION_MODES];
|
||||
|
|
@ -625,24 +600,18 @@ typedef struct AV1_COMP {
|
|||
|
||||
TileBufferEnc tile_buffers[MAX_TILE_ROWS][MAX_TILE_COLS];
|
||||
|
||||
int resize_pending;
|
||||
int resize_state;
|
||||
int resize_scale_num;
|
||||
int resize_scale_den;
|
||||
int resize_next_scale_num;
|
||||
int resize_next_scale_den;
|
||||
int resize_avg_qp;
|
||||
int resize_buffer_underflow;
|
||||
int resize_count;
|
||||
|
||||
// VAR_BASED_PARTITION thresholds
|
||||
// 0 - threshold_128x128;
|
||||
// 1 - threshold_64x64;
|
||||
// 2 - threshold_32x32;
|
||||
// 3 - threshold_16x16;
|
||||
// 4 - threshold_8x8;
|
||||
int64_t vbp_thresholds[5];
|
||||
int64_t vbp_threshold_minmax;
|
||||
int64_t vbp_threshold_sad;
|
||||
BLOCK_SIZE vbp_bsize_min;
|
||||
#if CONFIG_FRAME_SUPERRES
|
||||
int superres_pending;
|
||||
#endif // CONFIG_FRAME_SUPERRES
|
||||
|
||||
// VARIANCE_AQ segment map refresh
|
||||
int vaq_refresh;
|
||||
|
|
@ -652,12 +621,6 @@ typedef struct AV1_COMP {
|
|||
AVxWorker *workers;
|
||||
struct EncWorkerData *tile_thr_data;
|
||||
AV1LfSync lf_row_sync;
|
||||
#if CONFIG_SUBFRAME_PROB_UPDATE
|
||||
SUBFRAME_STATS subframe_stats;
|
||||
// TODO(yaowu): minimize the size of count buffers
|
||||
SUBFRAME_STATS wholeframe_stats;
|
||||
av1_coeff_stats branch_ct_buf[COEF_PROBS_BUFS][TX_SIZES][PLANE_TYPES];
|
||||
#endif // CONFIG_SUBFRAME_PROB_UPDATE
|
||||
#if CONFIG_ANS
|
||||
struct BufAnsCoder buf_ans;
|
||||
#endif
|
||||
|
|
@ -720,8 +683,8 @@ int av1_get_active_map(AV1_COMP *cpi, unsigned char *map, int rows, int cols);
|
|||
int av1_set_internal_size(AV1_COMP *cpi, AOM_SCALING horiz_mode,
|
||||
AOM_SCALING vert_mode);
|
||||
|
||||
int av1_set_size_literal(AV1_COMP *cpi, unsigned int width,
|
||||
unsigned int height);
|
||||
// Returns 1 if the assigned width or height was <= 0.
|
||||
int av1_set_size_literal(AV1_COMP *cpi, int width, int height);
|
||||
|
||||
int av1_get_quantizer(struct AV1_COMP *cpi);
|
||||
|
||||
|
|
@ -774,7 +737,7 @@ static INLINE const YV12_BUFFER_CONFIG *get_upsampled_ref(
|
|||
return &cpi->upsampled_ref_bufs[buf_idx].buf;
|
||||
}
|
||||
|
||||
#if CONFIG_EXT_REFS
|
||||
#if CONFIG_EXT_REFS || CONFIG_TEMPMV_SIGNALING
|
||||
static INLINE int enc_is_ref_frame_buf(AV1_COMP *cpi, RefCntBuffer *frame_buf) {
|
||||
MV_REFERENCE_FRAME ref_frame;
|
||||
AV1_COMMON *const cm = &cpi->common;
|
||||
|
|
@ -819,14 +782,6 @@ void av1_set_high_precision_mv(AV1_COMP *cpi, int allow_high_precision_mv);
|
|||
void av1_set_temporal_mv_prediction(AV1_COMP *cpi, int allow_tempmv_prediction);
|
||||
#endif
|
||||
|
||||
YV12_BUFFER_CONFIG *av1_scale_if_required_fast(AV1_COMMON *cm,
|
||||
YV12_BUFFER_CONFIG *unscaled,
|
||||
YV12_BUFFER_CONFIG *scaled);
|
||||
|
||||
YV12_BUFFER_CONFIG *av1_scale_if_required(AV1_COMMON *cm,
|
||||
YV12_BUFFER_CONFIG *unscaled,
|
||||
YV12_BUFFER_CONFIG *scaled);
|
||||
|
||||
void av1_apply_encoding_flags(AV1_COMP *cpi, aom_enc_frame_flags_t flags);
|
||||
|
||||
static INLINE int is_altref_enabled(const AV1_COMP *const cpi) {
|
||||
|
|
@ -876,6 +831,25 @@ static INLINE void uref_cnt_fb(EncRefCntBuffer *ubufs, int *uidx,
|
|||
ubufs[new_uidx].ref_count++;
|
||||
}
|
||||
|
||||
// Returns 1 if a resize is pending and 0 otherwise.
|
||||
static INLINE int av1_resize_pending(const struct AV1_COMP *cpi) {
|
||||
return cpi->resize_scale_num != cpi->resize_next_scale_num ||
|
||||
cpi->resize_scale_den != cpi->resize_next_scale_den;
|
||||
}
|
||||
|
||||
// Returns 1 if a frame is unscaled and 0 otherwise.
|
||||
static INLINE int av1_resize_unscaled(const struct AV1_COMP *cpi) {
|
||||
return cpi->resize_scale_num == cpi->resize_scale_den;
|
||||
}
|
||||
|
||||
// Moves resizing to the next state. This is just setting the numerator and
|
||||
// denominator to the next numerator and denominator, causing
|
||||
// av1_resize_pending to subsequently return false.
|
||||
static INLINE void av1_resize_step(struct AV1_COMP *cpi) {
|
||||
cpi->resize_scale_num = cpi->resize_next_scale_num;
|
||||
cpi->resize_scale_den = cpi->resize_next_scale_den;
|
||||
}
|
||||
|
||||
#ifdef __cplusplus
|
||||
} // extern "C"
|
||||
#endif
|
||||
|
|
|
|||
1149
third_party/aom/av1/encoder/encodetxb.c
vendored
1149
third_party/aom/av1/encoder/encodetxb.c
vendored
File diff suppressed because it is too large
Load diff
51
third_party/aom/av1/encoder/encodetxb.h
vendored
51
third_party/aom/av1/encoder/encodetxb.h
vendored
|
|
@ -22,6 +22,47 @@
|
|||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
|
||||
typedef struct TxbInfo {
|
||||
tran_low_t *qcoeff;
|
||||
tran_low_t *dqcoeff;
|
||||
const tran_low_t *tcoeff;
|
||||
const int16_t *dequant;
|
||||
int shift;
|
||||
TX_SIZE tx_size;
|
||||
int bwl;
|
||||
int stride;
|
||||
int eob;
|
||||
int seg_eob;
|
||||
const SCAN_ORDER *scan_order;
|
||||
TXB_CTX *txb_ctx;
|
||||
int64_t rdmult;
|
||||
int64_t rddiv;
|
||||
} TxbInfo;
|
||||
|
||||
typedef struct TxbCache {
|
||||
int nz_count_arr[MAX_TX_SQUARE];
|
||||
int nz_ctx_arr[MAX_TX_SQUARE][2];
|
||||
int base_count_arr[NUM_BASE_LEVELS][MAX_TX_SQUARE];
|
||||
int base_mag_arr[MAX_TX_SQUARE]
|
||||
[2]; // [0]: max magnitude [1]: num of max magnitude
|
||||
int base_ctx_arr[NUM_BASE_LEVELS][MAX_TX_SQUARE][2]; // [1]: not used
|
||||
|
||||
int br_count_arr[MAX_TX_SQUARE];
|
||||
int br_mag_arr[MAX_TX_SQUARE]
|
||||
[2]; // [0]: max magnitude [1]: num of max magnitude
|
||||
int br_ctx_arr[MAX_TX_SQUARE][2]; // [1]: not used
|
||||
} TxbCache;
|
||||
|
||||
typedef struct TxbProbs {
|
||||
const aom_prob *dc_sign_prob;
|
||||
const aom_prob *nz_map;
|
||||
aom_prob (*coeff_base)[COEFF_BASE_CONTEXTS];
|
||||
const aom_prob *coeff_lps;
|
||||
const aom_prob *eob_flag;
|
||||
const aom_prob *txb_skip;
|
||||
} TxbProbs;
|
||||
|
||||
void av1_alloc_txb_buf(AV1_COMP *cpi);
|
||||
void av1_free_txb_buf(AV1_COMP *cpi);
|
||||
int av1_cost_coeffs_txb(const AV1_COMP *const cpi, MACROBLOCK *x, int plane,
|
||||
|
|
@ -39,6 +80,14 @@ void av1_update_txb_context(const AV1_COMP *cpi, ThreadData *td,
|
|||
const int mi_row, const int mi_col);
|
||||
void av1_write_txb_probs(AV1_COMP *cpi, aom_writer *w);
|
||||
|
||||
void av1_update_txb_context_b(int plane, int block, int blk_row, int blk_col,
|
||||
BLOCK_SIZE plane_bsize, TX_SIZE tx_size,
|
||||
void *arg);
|
||||
|
||||
void av1_update_and_record_txb_context(int plane, int block, int blk_row,
|
||||
int blk_col, BLOCK_SIZE plane_bsize,
|
||||
TX_SIZE tx_size, void *arg);
|
||||
|
||||
#if CONFIG_TXK_SEL
|
||||
int64_t av1_search_txk_type(const AV1_COMP *cpi, MACROBLOCK *x, int plane,
|
||||
int block, int blk_row, int blk_col,
|
||||
|
|
@ -46,6 +95,8 @@ int64_t av1_search_txk_type(const AV1_COMP *cpi, MACROBLOCK *x, int plane,
|
|||
const ENTROPY_CONTEXT *a, const ENTROPY_CONTEXT *l,
|
||||
int use_fast_coef_costing, RD_STATS *rd_stats);
|
||||
#endif
|
||||
int av1_optimize_txb(const AV1_COMMON *cm, MACROBLOCK *x, int plane, int block,
|
||||
TX_SIZE tx_size, TXB_CTX *txb_ctx);
|
||||
#ifdef __cplusplus
|
||||
}
|
||||
#endif
|
||||
|
|
|
|||
53
third_party/aom/av1/encoder/ethread.c
vendored
53
third_party/aom/av1/encoder/ethread.c
vendored
|
|
@ -93,14 +93,42 @@ void av1_encode_tiles_mt(AV1_COMP *cpi) {
|
|||
thread_data->td->pc_tree = NULL;
|
||||
av1_setup_pc_tree(cm, thread_data->td);
|
||||
|
||||
// Set up variance tree if needed.
|
||||
if (cpi->sf.partition_search_type == VAR_BASED_PARTITION)
|
||||
av1_setup_var_tree(cm, thread_data->td);
|
||||
|
||||
#if CONFIG_MOTION_VAR
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
int buf_scaler = 2;
|
||||
#else
|
||||
int buf_scaler = 1;
|
||||
#endif
|
||||
CHECK_MEM_ERROR(cm, thread_data->td->above_pred_buf,
|
||||
(uint8_t *)aom_memalign(
|
||||
16, buf_scaler * MAX_MB_PLANE * MAX_SB_SQUARE *
|
||||
sizeof(*thread_data->td->above_pred_buf)));
|
||||
CHECK_MEM_ERROR(cm, thread_data->td->left_pred_buf,
|
||||
(uint8_t *)aom_memalign(
|
||||
16, buf_scaler * MAX_MB_PLANE * MAX_SB_SQUARE *
|
||||
sizeof(*thread_data->td->left_pred_buf)));
|
||||
CHECK_MEM_ERROR(
|
||||
cm, thread_data->td->wsrc_buf,
|
||||
(int32_t *)aom_memalign(
|
||||
16, MAX_SB_SQUARE * sizeof(*thread_data->td->wsrc_buf)));
|
||||
CHECK_MEM_ERROR(
|
||||
cm, thread_data->td->mask_buf,
|
||||
(int32_t *)aom_memalign(
|
||||
16, MAX_SB_SQUARE * sizeof(*thread_data->td->mask_buf)));
|
||||
#endif
|
||||
// Allocate frame counters in thread data.
|
||||
CHECK_MEM_ERROR(cm, thread_data->td->counts,
|
||||
aom_calloc(1, sizeof(*thread_data->td->counts)));
|
||||
|
||||
#if CONFIG_PALETTE
|
||||
// Allocate buffers used by palette coding mode.
|
||||
if (cpi->common.allow_screen_content_tools) {
|
||||
CHECK_MEM_ERROR(
|
||||
cm, thread_data->td->palette_buffer,
|
||||
aom_memalign(16, sizeof(*thread_data->td->palette_buffer)));
|
||||
}
|
||||
#endif // CONFIG_PALETTE
|
||||
|
||||
// Create threads
|
||||
if (!winterface->reset(worker))
|
||||
aom_internal_error(&cm->error, AOM_CODEC_ERROR,
|
||||
|
|
@ -127,6 +155,12 @@ void av1_encode_tiles_mt(AV1_COMP *cpi) {
|
|||
if (thread_data->td != &cpi->td) {
|
||||
thread_data->td->mb = cpi->td.mb;
|
||||
thread_data->td->rd_counts = cpi->td.rd_counts;
|
||||
#if CONFIG_MOTION_VAR
|
||||
thread_data->td->mb.above_pred_buf = thread_data->td->above_pred_buf;
|
||||
thread_data->td->mb.left_pred_buf = thread_data->td->left_pred_buf;
|
||||
thread_data->td->mb.wsrc_buf = thread_data->td->wsrc_buf;
|
||||
thread_data->td->mb.mask_buf = thread_data->td->mask_buf;
|
||||
#endif
|
||||
}
|
||||
if (thread_data->td->counts != &cpi->common.counts) {
|
||||
memcpy(thread_data->td->counts, &cpi->common.counts,
|
||||
|
|
@ -134,12 +168,8 @@ void av1_encode_tiles_mt(AV1_COMP *cpi) {
|
|||
}
|
||||
|
||||
#if CONFIG_PALETTE
|
||||
// Allocate buffers used by palette coding mode.
|
||||
if (cpi->common.allow_screen_content_tools && i < num_workers - 1) {
|
||||
MACROBLOCK *x = &thread_data->td->mb;
|
||||
CHECK_MEM_ERROR(cm, x->palette_buffer,
|
||||
aom_memalign(16, sizeof(*x->palette_buffer)));
|
||||
}
|
||||
if (cpi->common.allow_screen_content_tools && i < num_workers - 1)
|
||||
thread_data->td->mb.palette_buffer = thread_data->td->palette_buffer;
|
||||
#endif // CONFIG_PALETTE
|
||||
}
|
||||
|
||||
|
|
@ -171,6 +201,9 @@ void av1_encode_tiles_mt(AV1_COMP *cpi) {
|
|||
if (i < cpi->num_workers - 1) {
|
||||
av1_accumulate_frame_counts(&cm->counts, thread_data->td->counts);
|
||||
accumulate_rd_opt(&cpi->td, thread_data->td);
|
||||
#if CONFIG_VAR_TX
|
||||
cpi->td.mb.txb_split_count += thread_data->td->mb.txb_split_count;
|
||||
#endif
|
||||
}
|
||||
}
|
||||
}
|
||||
|
|
|
|||
60
third_party/aom/av1/encoder/firstpass.c
vendored
60
third_party/aom/av1/encoder/firstpass.c
vendored
|
|
@ -568,16 +568,11 @@ void av1_first_pass(AV1_COMP *cpi, const struct lookahead_entry *source) {
|
|||
|
||||
od_init_qm(x->daala_enc.state.qm, x->daala_enc.state.qm_inv,
|
||||
x->daala_enc.qm == OD_HVS_QM ? OD_QM8_Q4_HVS : OD_QM8_Q4_FLAT);
|
||||
#if CONFIG_DAALA_EC
|
||||
#if !CONFIG_ANS
|
||||
od_ec_enc_init(&x->daala_enc.w.ec, 65025);
|
||||
#else
|
||||
#error "CONFIG_PVQ currently requires CONFIG_DAALA_EC."
|
||||
#endif
|
||||
|
||||
#if CONFIG_DAALA_EC
|
||||
od_ec_enc_reset(&x->daala_enc.w.ec);
|
||||
#else
|
||||
#error "CONFIG_PVQ currently requires CONFIG_DAALA_EC."
|
||||
#error "CONFIG_PVQ currently requires !CONFIG_ANS."
|
||||
#endif
|
||||
}
|
||||
#endif
|
||||
|
|
@ -598,6 +593,7 @@ void av1_first_pass(AV1_COMP *cpi, const struct lookahead_entry *source) {
|
|||
av1_init_mv_probs(cm);
|
||||
#if CONFIG_ADAPT_SCAN
|
||||
av1_init_scan_order(cm);
|
||||
av1_deliver_eob_threshold(cm, xd);
|
||||
#endif
|
||||
av1_convolve_init(cm);
|
||||
#if CONFIG_PVQ
|
||||
|
|
@ -884,7 +880,7 @@ void av1_first_pass(AV1_COMP *cpi, const struct lookahead_entry *source) {
|
|||
xd->mi[0]->mbmi.tx_size = TX_4X4;
|
||||
xd->mi[0]->mbmi.ref_frame[0] = LAST_FRAME;
|
||||
xd->mi[0]->mbmi.ref_frame[1] = NONE_FRAME;
|
||||
av1_build_inter_predictors_sby(xd, mb_row * mb_scale,
|
||||
av1_build_inter_predictors_sby(cm, xd, mb_row * mb_scale,
|
||||
mb_col * mb_scale, NULL, bsize);
|
||||
av1_encode_sby_pass1(cm, x, bsize);
|
||||
sum_mvr += mv.row;
|
||||
|
|
@ -997,10 +993,10 @@ void av1_first_pass(AV1_COMP *cpi, const struct lookahead_entry *source) {
|
|||
}
|
||||
|
||||
#if CONFIG_PVQ
|
||||
#if CONFIG_DAALA_EC
|
||||
#if !CONFIG_ANS
|
||||
od_ec_enc_clear(&x->daala_enc.w.ec);
|
||||
#else
|
||||
#error "CONFIG_PVQ currently requires CONFIG_DAALA_EC."
|
||||
#error "CONFIG_PVQ currently requires !CONFIG_ANS."
|
||||
#endif
|
||||
|
||||
x->pvq_q->last_pos = x->pvq_q->curr_pos;
|
||||
|
|
@ -1235,28 +1231,26 @@ static void setup_rf_level_maxq(AV1_COMP *cpi) {
|
|||
}
|
||||
}
|
||||
|
||||
void av1_init_subsampling(AV1_COMP *cpi) {
|
||||
const AV1_COMMON *const cm = &cpi->common;
|
||||
RATE_CONTROL *const rc = &cpi->rc;
|
||||
const int w = cm->width;
|
||||
const int h = cm->height;
|
||||
int i;
|
||||
|
||||
for (i = 0; i < FRAME_SCALE_STEPS; ++i) {
|
||||
// Note: Frames with odd-sized dimensions may result from this scaling.
|
||||
rc->frame_width[i] = (w * 16) / frame_scale_factor[i];
|
||||
rc->frame_height[i] = (h * 16) / frame_scale_factor[i];
|
||||
}
|
||||
|
||||
setup_rf_level_maxq(cpi);
|
||||
void av1_calculate_next_scaled_size(const AV1_COMP *cpi,
|
||||
int *scaled_frame_width,
|
||||
int *scaled_frame_height) {
|
||||
*scaled_frame_width =
|
||||
cpi->oxcf.width * cpi->resize_next_scale_num / cpi->resize_next_scale_den;
|
||||
*scaled_frame_height = cpi->oxcf.height * cpi->resize_next_scale_num /
|
||||
cpi->resize_next_scale_den;
|
||||
}
|
||||
|
||||
void av1_calculate_coded_size(AV1_COMP *cpi, int *scaled_frame_width,
|
||||
int *scaled_frame_height) {
|
||||
RATE_CONTROL *const rc = &cpi->rc;
|
||||
*scaled_frame_width = rc->frame_width[rc->frame_size_selector];
|
||||
*scaled_frame_height = rc->frame_height[rc->frame_size_selector];
|
||||
#if CONFIG_FRAME_SUPERRES
|
||||
void av1_calculate_superres_size(const AV1_COMP *cpi, int *encoded_width,
|
||||
int *encoded_height) {
|
||||
*encoded_width = cpi->oxcf.scaled_frame_width *
|
||||
cpi->common.superres_scale_numerator /
|
||||
SUPERRES_SCALE_DENOMINATOR;
|
||||
*encoded_height = cpi->oxcf.scaled_frame_height *
|
||||
cpi->common.superres_scale_numerator /
|
||||
SUPERRES_SCALE_DENOMINATOR;
|
||||
}
|
||||
#endif // CONFIG_FRAME_SUPERRES
|
||||
|
||||
void av1_init_second_pass(AV1_COMP *cpi) {
|
||||
const AV1EncoderConfig *const oxcf = &cpi->oxcf;
|
||||
|
|
@ -1316,7 +1310,7 @@ void av1_init_second_pass(AV1_COMP *cpi) {
|
|||
twopass->last_kfgroup_zeromotion_pct = 100;
|
||||
|
||||
if (oxcf->resize_mode != RESIZE_NONE) {
|
||||
av1_init_subsampling(cpi);
|
||||
setup_rf_level_maxq(cpi);
|
||||
}
|
||||
}
|
||||
|
||||
|
|
@ -2300,7 +2294,8 @@ static void define_gf_group(AV1_COMP *cpi, FIRSTPASS_STATS *this_frame) {
|
|||
|
||||
if (oxcf->resize_mode == RESIZE_DYNAMIC) {
|
||||
// Default to starting GF groups at normal frame size.
|
||||
cpi->rc.next_frame_size_selector = UNSCALED;
|
||||
// TODO(afergs): Make a function for this
|
||||
cpi->resize_next_scale_num = cpi->resize_next_scale_den;
|
||||
}
|
||||
}
|
||||
|
||||
|
|
@ -2646,7 +2641,8 @@ static void find_next_key_frame(AV1_COMP *cpi, FIRSTPASS_STATS *this_frame) {
|
|||
|
||||
if (oxcf->resize_mode == RESIZE_DYNAMIC) {
|
||||
// Default to normal-sized frame on keyframes.
|
||||
cpi->rc.next_frame_size_selector = UNSCALED;
|
||||
// TODO(afergs): Make a function for this
|
||||
cpi->resize_next_scale_num = cpi->resize_next_scale_den;
|
||||
}
|
||||
}
|
||||
|
||||
|
|
|
|||
13
third_party/aom/av1/encoder/firstpass.h
vendored
13
third_party/aom/av1/encoder/firstpass.h
vendored
|
|
@ -177,10 +177,17 @@ void av1_twopass_postencode_update(struct AV1_COMP *cpi);
|
|||
// Post encode update of the rate control parameters for 2-pass
|
||||
void av1_twopass_postencode_update(struct AV1_COMP *cpi);
|
||||
|
||||
void av1_init_subsampling(struct AV1_COMP *cpi);
|
||||
void av1_calculate_next_scaled_size(const struct AV1_COMP *cpi,
|
||||
int *scaled_frame_width,
|
||||
int *scaled_frame_height);
|
||||
|
||||
void av1_calculate_coded_size(struct AV1_COMP *cpi, int *scaled_frame_width,
|
||||
int *scaled_frame_height);
|
||||
#if CONFIG_FRAME_SUPERRES
|
||||
// This is the size after superress scaling, which could be 1:1.
|
||||
// Superres scaling happens after regular downscaling.
|
||||
// TODO(afergs): Limit overall reduction to 1/2 of the original size
|
||||
void av1_calculate_superres_size(const struct AV1_COMP *cpi, int *encoded_width,
|
||||
int *encoded_height);
|
||||
#endif // CONFIG_FRAME_SUPERRES
|
||||
|
||||
#if CONFIG_EXT_REFS
|
||||
static INLINE int get_number_of_extra_arfs(int interval, int arf_pending) {
|
||||
|
|
|
|||
34
third_party/aom/av1/encoder/global_motion.c
vendored
34
third_party/aom/av1/encoder/global_motion.c
vendored
|
|
@ -124,14 +124,15 @@ static void force_wmtype(WarpedMotionParams *wm, TransformationType wmtype) {
|
|||
wm->wmtype = wmtype;
|
||||
}
|
||||
|
||||
double refine_integerized_param(WarpedMotionParams *wm,
|
||||
TransformationType wmtype,
|
||||
int64_t refine_integerized_param(WarpedMotionParams *wm,
|
||||
TransformationType wmtype,
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
int use_hbd, int bd,
|
||||
int use_hbd, int bd,
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
uint8_t *ref, int r_width, int r_height,
|
||||
int r_stride, uint8_t *dst, int d_width,
|
||||
int d_height, int d_stride, int n_refinements) {
|
||||
uint8_t *ref, int r_width, int r_height,
|
||||
int r_stride, uint8_t *dst, int d_width,
|
||||
int d_height, int d_stride,
|
||||
int n_refinements) {
|
||||
static const int max_trans_model_params[TRANS_TYPES] = {
|
||||
0, 2, 4, 6, 8, 8, 8
|
||||
};
|
||||
|
|
@ -139,22 +140,21 @@ double refine_integerized_param(WarpedMotionParams *wm,
|
|||
int i = 0, p;
|
||||
int n_params = max_trans_model_params[wmtype];
|
||||
int32_t *param_mat = wm->wmmat;
|
||||
double step_error;
|
||||
int64_t step_error, best_error;
|
||||
int32_t step;
|
||||
int32_t *param;
|
||||
int32_t curr_param;
|
||||
int32_t best_param;
|
||||
double best_error;
|
||||
|
||||
force_wmtype(wm, wmtype);
|
||||
best_error = av1_warp_erroradv(wm,
|
||||
best_error = av1_warp_error(wm,
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
use_hbd, bd,
|
||||
use_hbd, bd,
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
ref, r_width, r_height, r_stride,
|
||||
dst + border * d_stride + border, border,
|
||||
border, d_width - 2 * border,
|
||||
d_height - 2 * border, d_stride, 0, 0, 16, 16);
|
||||
ref, r_width, r_height, r_stride,
|
||||
dst + border * d_stride + border, border, border,
|
||||
d_width - 2 * border, d_height - 2 * border,
|
||||
d_stride, 0, 0, 16, 16);
|
||||
step = 1 << (n_refinements + 1);
|
||||
for (i = 0; i < n_refinements; i++, step >>= 1) {
|
||||
for (p = 0; p < n_params; ++p) {
|
||||
|
|
@ -167,7 +167,7 @@ double refine_integerized_param(WarpedMotionParams *wm,
|
|||
best_param = curr_param;
|
||||
// look to the left
|
||||
*param = add_param_offset(p, curr_param, -step);
|
||||
step_error = av1_warp_erroradv(
|
||||
step_error = av1_warp_error(
|
||||
wm,
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
use_hbd, bd,
|
||||
|
|
@ -183,7 +183,7 @@ double refine_integerized_param(WarpedMotionParams *wm,
|
|||
|
||||
// look to the right
|
||||
*param = add_param_offset(p, curr_param, step);
|
||||
step_error = av1_warp_erroradv(
|
||||
step_error = av1_warp_error(
|
||||
wm,
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
use_hbd, bd,
|
||||
|
|
@ -202,7 +202,7 @@ double refine_integerized_param(WarpedMotionParams *wm,
|
|||
// for the biggest step size
|
||||
while (step_dir) {
|
||||
*param = add_param_offset(p, best_param, step * step_dir);
|
||||
step_error = av1_warp_erroradv(
|
||||
step_error = av1_warp_error(
|
||||
wm,
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
use_hbd, bd,
|
||||
|
|
|
|||
15
third_party/aom/av1/encoder/global_motion.h
vendored
15
third_party/aom/av1/encoder/global_motion.h
vendored
|
|
@ -26,14 +26,17 @@ void convert_model_to_params(const double *params, WarpedMotionParams *model);
|
|||
|
||||
int is_enough_erroradvantage(double erroradv, int params_cost);
|
||||
|
||||
double refine_integerized_param(WarpedMotionParams *wm,
|
||||
TransformationType wmtype,
|
||||
// Returns the av1_warp_error between "dst" and the result of applying the
|
||||
// motion params that result from fine-tuning "wm" to "ref". Note that "wm" is
|
||||
// modified in place.
|
||||
int64_t refine_integerized_param(WarpedMotionParams *wm,
|
||||
TransformationType wmtype,
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
int use_hbd, int bd,
|
||||
int use_hbd, int bd,
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
uint8_t *ref, int r_width, int r_height,
|
||||
int r_stride, uint8_t *dst, int d_width,
|
||||
int d_height, int d_stride, int n_refinements);
|
||||
uint8_t *ref, int r_width, int r_height,
|
||||
int r_stride, uint8_t *dst, int d_width,
|
||||
int d_height, int d_stride, int n_refinements);
|
||||
|
||||
/*
|
||||
Computes "num_motions" candidate global motion parameters between two frames.
|
||||
|
|
|
|||
52
third_party/aom/av1/encoder/hybrid_fwd_txfm.c
vendored
52
third_party/aom/av1/encoder/hybrid_fwd_txfm.c
vendored
|
|
@ -16,7 +16,7 @@
|
|||
#include "av1/common/idct.h"
|
||||
#include "av1/encoder/hybrid_fwd_txfm.h"
|
||||
|
||||
#if CONFIG_CB4X4
|
||||
#if CONFIG_CHROMA_2X2
|
||||
static void fwd_txfm_2x2(const int16_t *src_diff, tran_low_t *coeff,
|
||||
int diff_stride, TX_TYPE tx_type, int lossless) {
|
||||
tran_high_t a1 = src_diff[0];
|
||||
|
|
@ -132,8 +132,38 @@ static void fwd_txfm_64x64(const int16_t *src_diff, tran_low_t *coeff,
|
|||
}
|
||||
#endif // CONFIG_TX64X64
|
||||
|
||||
#if CONFIG_EXT_TX && CONFIG_RECT_TX && CONFIG_RECT_TX_EXT
|
||||
static void fwd_txfm_16x4(const int16_t *src_diff, tran_low_t *coeff,
|
||||
int diff_stride, TX_TYPE tx_type,
|
||||
FWD_TXFM_OPT fwd_txfm_opt) {
|
||||
(void)fwd_txfm_opt;
|
||||
av1_fht16x4(src_diff, coeff, diff_stride, tx_type);
|
||||
}
|
||||
|
||||
static void fwd_txfm_4x16(const int16_t *src_diff, tran_low_t *coeff,
|
||||
int diff_stride, TX_TYPE tx_type,
|
||||
FWD_TXFM_OPT fwd_txfm_opt) {
|
||||
(void)fwd_txfm_opt;
|
||||
av1_fht4x16(src_diff, coeff, diff_stride, tx_type);
|
||||
}
|
||||
|
||||
static void fwd_txfm_32x8(const int16_t *src_diff, tran_low_t *coeff,
|
||||
int diff_stride, TX_TYPE tx_type,
|
||||
FWD_TXFM_OPT fwd_txfm_opt) {
|
||||
(void)fwd_txfm_opt;
|
||||
av1_fht32x8(src_diff, coeff, diff_stride, tx_type);
|
||||
}
|
||||
|
||||
static void fwd_txfm_8x32(const int16_t *src_diff, tran_low_t *coeff,
|
||||
int diff_stride, TX_TYPE tx_type,
|
||||
FWD_TXFM_OPT fwd_txfm_opt) {
|
||||
(void)fwd_txfm_opt;
|
||||
av1_fht8x32(src_diff, coeff, diff_stride, tx_type);
|
||||
}
|
||||
#endif // CONFIG_EXT_TX && CONFIG_RECT_TX && CONFIG_RECT_TX_EXT
|
||||
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
#if CONFIG_CB4X4
|
||||
#if CONFIG_CHROMA_2X2
|
||||
static void highbd_fwd_txfm_2x2(const int16_t *src_diff, tran_low_t *coeff,
|
||||
int diff_stride, TX_TYPE tx_type, int lossless,
|
||||
const int bd) {
|
||||
|
|
@ -425,11 +455,25 @@ void av1_fwd_txfm(const int16_t *src_diff, tran_low_t *coeff, int diff_stride,
|
|||
case TX_4X4:
|
||||
fwd_txfm_4x4(src_diff, coeff, diff_stride, tx_type, lossless);
|
||||
break;
|
||||
#if CONFIG_CB4X4
|
||||
#if CONFIG_CHROMA_2X2
|
||||
case TX_2X2:
|
||||
fwd_txfm_2x2(src_diff, coeff, diff_stride, tx_type, lossless);
|
||||
break;
|
||||
#endif
|
||||
#if CONFIG_EXT_TX && CONFIG_RECT_TX && CONFIG_RECT_TX_EXT
|
||||
case TX_4X16:
|
||||
fwd_txfm_4x16(src_diff, coeff, diff_stride, tx_type, fwd_txfm_opt);
|
||||
break;
|
||||
case TX_16X4:
|
||||
fwd_txfm_16x4(src_diff, coeff, diff_stride, tx_type, fwd_txfm_opt);
|
||||
break;
|
||||
case TX_8X32:
|
||||
fwd_txfm_8x32(src_diff, coeff, diff_stride, tx_type, fwd_txfm_opt);
|
||||
break;
|
||||
case TX_32X8:
|
||||
fwd_txfm_32x8(src_diff, coeff, diff_stride, tx_type, fwd_txfm_opt);
|
||||
break;
|
||||
#endif // CONFIG_EXT_TX && CONFIG_RECT_TX && CONFIG_RECT_TX_EXT
|
||||
default: assert(0); break;
|
||||
}
|
||||
}
|
||||
|
|
@ -488,7 +532,7 @@ void av1_highbd_fwd_txfm(const int16_t *src_diff, tran_low_t *coeff,
|
|||
case TX_4X4:
|
||||
highbd_fwd_txfm_4x4(src_diff, coeff, diff_stride, tx_type, lossless, bd);
|
||||
break;
|
||||
#if CONFIG_CB4X4
|
||||
#if CONFIG_CHROMA_2X2
|
||||
case TX_2X2:
|
||||
highbd_fwd_txfm_2x2(src_diff, coeff, diff_stride, tx_type, lossless, bd);
|
||||
break;
|
||||
|
|
|
|||
354
third_party/aom/av1/encoder/mathutils.h
vendored
Normal file
354
third_party/aom/av1/encoder/mathutils.h
vendored
Normal file
|
|
@ -0,0 +1,354 @@
|
|||
/*
|
||||
* Copyright (c) 2017, Alliance for Open Media. All rights reserved
|
||||
*
|
||||
* This source code is subject to the terms of the BSD 2 Clause License and
|
||||
* the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
|
||||
* was not distributed with this source code in the LICENSE file, you can
|
||||
* obtain it at www.aomedia.org/license/software. If the Alliance for Open
|
||||
* Media Patent License 1.0 was not distributed with this source code in the
|
||||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#include <memory.h>
|
||||
#include <math.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <assert.h>
|
||||
|
||||
static const double TINY_NEAR_ZERO = 1.0E-16;
|
||||
|
||||
// Solves Ax = b, where x and b are column vectors of size nx1 and A is nxn
|
||||
static INLINE int linsolve(int n, double *A, int stride, double *b, double *x) {
|
||||
int i, j, k;
|
||||
double c;
|
||||
// Forward elimination
|
||||
for (k = 0; k < n - 1; k++) {
|
||||
// Bring the largest magitude to the diagonal position
|
||||
for (i = n - 1; i > k; i--) {
|
||||
if (fabs(A[(i - 1) * stride + k]) < fabs(A[i * stride + k])) {
|
||||
for (j = 0; j < n; j++) {
|
||||
c = A[i * stride + j];
|
||||
A[i * stride + j] = A[(i - 1) * stride + j];
|
||||
A[(i - 1) * stride + j] = c;
|
||||
}
|
||||
c = b[i];
|
||||
b[i] = b[i - 1];
|
||||
b[i - 1] = c;
|
||||
}
|
||||
}
|
||||
for (i = k; i < n - 1; i++) {
|
||||
if (fabs(A[k * stride + k]) < TINY_NEAR_ZERO) return 0;
|
||||
c = A[(i + 1) * stride + k] / A[k * stride + k];
|
||||
for (j = 0; j < n; j++) A[(i + 1) * stride + j] -= c * A[k * stride + j];
|
||||
b[i + 1] -= c * b[k];
|
||||
}
|
||||
}
|
||||
// Backward substitution
|
||||
for (i = n - 1; i >= 0; i--) {
|
||||
if (fabs(A[i * stride + i]) < TINY_NEAR_ZERO) return 0;
|
||||
c = 0;
|
||||
for (j = i + 1; j <= n - 1; j++) c += A[i * stride + j] * x[j];
|
||||
x[i] = (b[i] - c) / A[i * stride + i];
|
||||
}
|
||||
|
||||
return 1;
|
||||
}
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////
|
||||
// Least-squares
|
||||
// Solves for n-dim x in a least squares sense to minimize |Ax - b|^2
|
||||
// The solution is simply x = (A'A)^-1 A'b or simply the solution for
|
||||
// the system: A'A x = A'b
|
||||
static INLINE int least_squares(int n, double *A, int rows, int stride,
|
||||
double *b, double *scratch, double *x) {
|
||||
int i, j, k;
|
||||
double *scratch_ = NULL;
|
||||
double *AtA, *Atb;
|
||||
if (!scratch) {
|
||||
scratch_ = (double *)aom_malloc(sizeof(*scratch) * n * (n + 1));
|
||||
scratch = scratch_;
|
||||
}
|
||||
AtA = scratch;
|
||||
Atb = scratch + n * n;
|
||||
|
||||
for (i = 0; i < n; ++i) {
|
||||
for (j = i; j < n; ++j) {
|
||||
AtA[i * n + j] = 0.0;
|
||||
for (k = 0; k < rows; ++k)
|
||||
AtA[i * n + j] += A[k * stride + i] * A[k * stride + j];
|
||||
AtA[j * n + i] = AtA[i * n + j];
|
||||
}
|
||||
Atb[i] = 0;
|
||||
for (k = 0; k < rows; ++k) Atb[i] += A[k * stride + i] * b[k];
|
||||
}
|
||||
int ret = linsolve(n, AtA, n, Atb, x);
|
||||
if (scratch_) aom_free(scratch_);
|
||||
return ret;
|
||||
}
|
||||
|
||||
// Matrix multiply
|
||||
static INLINE void multiply_mat(const double *m1, const double *m2, double *res,
|
||||
const int m1_rows, const int inner_dim,
|
||||
const int m2_cols) {
|
||||
double sum;
|
||||
|
||||
int row, col, inner;
|
||||
for (row = 0; row < m1_rows; ++row) {
|
||||
for (col = 0; col < m2_cols; ++col) {
|
||||
sum = 0;
|
||||
for (inner = 0; inner < inner_dim; ++inner)
|
||||
sum += m1[row * inner_dim + inner] * m2[inner * m2_cols + col];
|
||||
*(res++) = sum;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
//
|
||||
// The functions below are needed only for homography computation
|
||||
// Remove if the homography models are not used.
|
||||
//
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
// svdcmp
|
||||
// Adopted from Numerical Recipes in C
|
||||
|
||||
static INLINE double sign(double a, double b) {
|
||||
return ((b) >= 0 ? fabs(a) : -fabs(a));
|
||||
}
|
||||
|
||||
static INLINE double pythag(double a, double b) {
|
||||
double ct;
|
||||
const double absa = fabs(a);
|
||||
const double absb = fabs(b);
|
||||
|
||||
if (absa > absb) {
|
||||
ct = absb / absa;
|
||||
return absa * sqrt(1.0 + ct * ct);
|
||||
} else {
|
||||
ct = absa / absb;
|
||||
return (absb == 0) ? 0 : absb * sqrt(1.0 + ct * ct);
|
||||
}
|
||||
}
|
||||
|
||||
static INLINE int svdcmp(double **u, int m, int n, double w[], double **v) {
|
||||
const int max_its = 30;
|
||||
int flag, i, its, j, jj, k, l, nm;
|
||||
double anorm, c, f, g, h, s, scale, x, y, z;
|
||||
double *rv1 = (double *)aom_malloc(sizeof(*rv1) * (n + 1));
|
||||
g = scale = anorm = 0.0;
|
||||
for (i = 0; i < n; i++) {
|
||||
l = i + 1;
|
||||
rv1[i] = scale * g;
|
||||
g = s = scale = 0.0;
|
||||
if (i < m) {
|
||||
for (k = i; k < m; k++) scale += fabs(u[k][i]);
|
||||
if (scale != 0.) {
|
||||
for (k = i; k < m; k++) {
|
||||
u[k][i] /= scale;
|
||||
s += u[k][i] * u[k][i];
|
||||
}
|
||||
f = u[i][i];
|
||||
g = -sign(sqrt(s), f);
|
||||
h = f * g - s;
|
||||
u[i][i] = f - g;
|
||||
for (j = l; j < n; j++) {
|
||||
for (s = 0.0, k = i; k < m; k++) s += u[k][i] * u[k][j];
|
||||
f = s / h;
|
||||
for (k = i; k < m; k++) u[k][j] += f * u[k][i];
|
||||
}
|
||||
for (k = i; k < m; k++) u[k][i] *= scale;
|
||||
}
|
||||
}
|
||||
w[i] = scale * g;
|
||||
g = s = scale = 0.0;
|
||||
if (i < m && i != n - 1) {
|
||||
for (k = l; k < n; k++) scale += fabs(u[i][k]);
|
||||
if (scale != 0.) {
|
||||
for (k = l; k < n; k++) {
|
||||
u[i][k] /= scale;
|
||||
s += u[i][k] * u[i][k];
|
||||
}
|
||||
f = u[i][l];
|
||||
g = -sign(sqrt(s), f);
|
||||
h = f * g - s;
|
||||
u[i][l] = f - g;
|
||||
for (k = l; k < n; k++) rv1[k] = u[i][k] / h;
|
||||
for (j = l; j < m; j++) {
|
||||
for (s = 0.0, k = l; k < n; k++) s += u[j][k] * u[i][k];
|
||||
for (k = l; k < n; k++) u[j][k] += s * rv1[k];
|
||||
}
|
||||
for (k = l; k < n; k++) u[i][k] *= scale;
|
||||
}
|
||||
}
|
||||
anorm = fmax(anorm, (fabs(w[i]) + fabs(rv1[i])));
|
||||
}
|
||||
|
||||
for (i = n - 1; i >= 0; i--) {
|
||||
if (i < n - 1) {
|
||||
if (g != 0.) {
|
||||
for (j = l; j < n; j++) v[j][i] = (u[i][j] / u[i][l]) / g;
|
||||
for (j = l; j < n; j++) {
|
||||
for (s = 0.0, k = l; k < n; k++) s += u[i][k] * v[k][j];
|
||||
for (k = l; k < n; k++) v[k][j] += s * v[k][i];
|
||||
}
|
||||
}
|
||||
for (j = l; j < n; j++) v[i][j] = v[j][i] = 0.0;
|
||||
}
|
||||
v[i][i] = 1.0;
|
||||
g = rv1[i];
|
||||
l = i;
|
||||
}
|
||||
for (i = AOMMIN(m, n) - 1; i >= 0; i--) {
|
||||
l = i + 1;
|
||||
g = w[i];
|
||||
for (j = l; j < n; j++) u[i][j] = 0.0;
|
||||
if (g != 0.) {
|
||||
g = 1.0 / g;
|
||||
for (j = l; j < n; j++) {
|
||||
for (s = 0.0, k = l; k < m; k++) s += u[k][i] * u[k][j];
|
||||
f = (s / u[i][i]) * g;
|
||||
for (k = i; k < m; k++) u[k][j] += f * u[k][i];
|
||||
}
|
||||
for (j = i; j < m; j++) u[j][i] *= g;
|
||||
} else {
|
||||
for (j = i; j < m; j++) u[j][i] = 0.0;
|
||||
}
|
||||
++u[i][i];
|
||||
}
|
||||
for (k = n - 1; k >= 0; k--) {
|
||||
for (its = 0; its < max_its; its++) {
|
||||
flag = 1;
|
||||
for (l = k; l >= 0; l--) {
|
||||
nm = l - 1;
|
||||
if ((double)(fabs(rv1[l]) + anorm) == anorm || nm < 0) {
|
||||
flag = 0;
|
||||
break;
|
||||
}
|
||||
if ((double)(fabs(w[nm]) + anorm) == anorm) break;
|
||||
}
|
||||
if (flag) {
|
||||
c = 0.0;
|
||||
s = 1.0;
|
||||
for (i = l; i <= k; i++) {
|
||||
f = s * rv1[i];
|
||||
rv1[i] = c * rv1[i];
|
||||
if ((double)(fabs(f) + anorm) == anorm) break;
|
||||
g = w[i];
|
||||
h = pythag(f, g);
|
||||
w[i] = h;
|
||||
h = 1.0 / h;
|
||||
c = g * h;
|
||||
s = -f * h;
|
||||
for (j = 0; j < m; j++) {
|
||||
y = u[j][nm];
|
||||
z = u[j][i];
|
||||
u[j][nm] = y * c + z * s;
|
||||
u[j][i] = z * c - y * s;
|
||||
}
|
||||
}
|
||||
}
|
||||
z = w[k];
|
||||
if (l == k) {
|
||||
if (z < 0.0) {
|
||||
w[k] = -z;
|
||||
for (j = 0; j < n; j++) v[j][k] = -v[j][k];
|
||||
}
|
||||
break;
|
||||
}
|
||||
if (its == max_its - 1) {
|
||||
aom_free(rv1);
|
||||
return 1;
|
||||
}
|
||||
assert(k > 0);
|
||||
x = w[l];
|
||||
nm = k - 1;
|
||||
y = w[nm];
|
||||
g = rv1[nm];
|
||||
h = rv1[k];
|
||||
f = ((y - z) * (y + z) + (g - h) * (g + h)) / (2.0 * h * y);
|
||||
g = pythag(f, 1.0);
|
||||
f = ((x - z) * (x + z) + h * ((y / (f + sign(g, f))) - h)) / x;
|
||||
c = s = 1.0;
|
||||
for (j = l; j <= nm; j++) {
|
||||
i = j + 1;
|
||||
g = rv1[i];
|
||||
y = w[i];
|
||||
h = s * g;
|
||||
g = c * g;
|
||||
z = pythag(f, h);
|
||||
rv1[j] = z;
|
||||
c = f / z;
|
||||
s = h / z;
|
||||
f = x * c + g * s;
|
||||
g = g * c - x * s;
|
||||
h = y * s;
|
||||
y *= c;
|
||||
for (jj = 0; jj < n; jj++) {
|
||||
x = v[jj][j];
|
||||
z = v[jj][i];
|
||||
v[jj][j] = x * c + z * s;
|
||||
v[jj][i] = z * c - x * s;
|
||||
}
|
||||
z = pythag(f, h);
|
||||
w[j] = z;
|
||||
if (z != 0.) {
|
||||
z = 1.0 / z;
|
||||
c = f * z;
|
||||
s = h * z;
|
||||
}
|
||||
f = c * g + s * y;
|
||||
x = c * y - s * g;
|
||||
for (jj = 0; jj < m; jj++) {
|
||||
y = u[jj][j];
|
||||
z = u[jj][i];
|
||||
u[jj][j] = y * c + z * s;
|
||||
u[jj][i] = z * c - y * s;
|
||||
}
|
||||
}
|
||||
rv1[l] = 0.0;
|
||||
rv1[k] = f;
|
||||
w[k] = x;
|
||||
}
|
||||
}
|
||||
aom_free(rv1);
|
||||
return 0;
|
||||
}
|
||||
|
||||
static INLINE int SVD(double *U, double *W, double *V, double *matx, int M,
|
||||
int N) {
|
||||
// Assumes allocation for U is MxN
|
||||
double **nrU = (double **)aom_malloc((M) * sizeof(*nrU));
|
||||
double **nrV = (double **)aom_malloc((N) * sizeof(*nrV));
|
||||
int problem, i;
|
||||
|
||||
problem = !(nrU && nrV);
|
||||
if (!problem) {
|
||||
for (i = 0; i < M; i++) {
|
||||
nrU[i] = &U[i * N];
|
||||
}
|
||||
for (i = 0; i < N; i++) {
|
||||
nrV[i] = &V[i * N];
|
||||
}
|
||||
} else {
|
||||
if (nrU) aom_free(nrU);
|
||||
if (nrV) aom_free(nrV);
|
||||
return 1;
|
||||
}
|
||||
|
||||
/* copy from given matx into nrU */
|
||||
for (i = 0; i < M; i++) {
|
||||
memcpy(&(nrU[i][0]), matx + N * i, N * sizeof(*matx));
|
||||
}
|
||||
|
||||
/* HERE IT IS: do SVD */
|
||||
if (svdcmp(nrU, M, N, W, nrV)) {
|
||||
aom_free(nrU);
|
||||
aom_free(nrV);
|
||||
return 1;
|
||||
}
|
||||
|
||||
/* aom_free Numerical Recipes arrays */
|
||||
aom_free(nrU);
|
||||
aom_free(nrV);
|
||||
|
||||
return 0;
|
||||
}
|
||||
16
third_party/aom/av1/encoder/mbgraph.c
vendored
16
third_party/aom/av1/encoder/mbgraph.c
vendored
|
|
@ -52,11 +52,14 @@ static unsigned int do_16x16_motion_iteration(AV1_COMP *cpi, const MV *ref_mv,
|
|||
{
|
||||
int distortion;
|
||||
unsigned int sse;
|
||||
cpi->find_fractional_mv_step(x, ref_mv, cpi->common.allow_high_precision_mv,
|
||||
x->errorperbit, &v_fn_ptr, 0,
|
||||
mv_sf->subpel_iters_per_step,
|
||||
cond_cost_list(cpi, cost_list), NULL, NULL,
|
||||
&distortion, &sse, NULL, 0, 0, 0);
|
||||
cpi->find_fractional_mv_step(
|
||||
x, ref_mv, cpi->common.allow_high_precision_mv, x->errorperbit,
|
||||
&v_fn_ptr, 0, mv_sf->subpel_iters_per_step,
|
||||
cond_cost_list(cpi, cost_list), NULL, NULL, &distortion, &sse, NULL,
|
||||
#if CONFIG_EXT_INTER
|
||||
NULL, 0, 0,
|
||||
#endif
|
||||
0, 0, 0);
|
||||
}
|
||||
|
||||
#if CONFIG_EXT_INTER
|
||||
|
|
@ -71,7 +74,8 @@ static unsigned int do_16x16_motion_iteration(AV1_COMP *cpi, const MV *ref_mv,
|
|||
xd->mi[0]->mbmi.ref_frame[1] = NONE_FRAME;
|
||||
#endif // CONFIG_EXT_INTER
|
||||
|
||||
av1_build_inter_predictors_sby(xd, mb_row, mb_col, NULL, BLOCK_16X16);
|
||||
av1_build_inter_predictors_sby(&cpi->common, xd, mb_row, mb_col, NULL,
|
||||
BLOCK_16X16);
|
||||
|
||||
/* restore UMV window */
|
||||
x->mv_limits = tmp_mv_limits;
|
||||
|
|
|
|||
1067
third_party/aom/av1/encoder/mcomp.c
vendored
1067
third_party/aom/av1/encoder/mcomp.c
vendored
File diff suppressed because it is too large
Load diff
49
third_party/aom/av1/encoder/mcomp.h
vendored
49
third_party/aom/av1/encoder/mcomp.h
vendored
|
|
@ -58,6 +58,13 @@ int av1_get_mvpred_var(const MACROBLOCK *x, const MV *best_mv,
|
|||
int av1_get_mvpred_av_var(const MACROBLOCK *x, const MV *best_mv,
|
||||
const MV *center_mv, const uint8_t *second_pred,
|
||||
const aom_variance_fn_ptr_t *vfp, int use_mvcost);
|
||||
#if CONFIG_EXT_INTER
|
||||
int av1_get_mvpred_mask_var(const MACROBLOCK *x, const MV *best_mv,
|
||||
const MV *center_mv, const uint8_t *second_pred,
|
||||
const uint8_t *mask, int mask_stride,
|
||||
int invert_mask, const aom_variance_fn_ptr_t *vfp,
|
||||
int use_mvcost);
|
||||
#endif
|
||||
|
||||
struct AV1_COMP;
|
||||
struct SPEED_FEATURES;
|
||||
|
|
@ -91,8 +98,11 @@ typedef int(fractional_mv_step_fp)(
|
|||
const aom_variance_fn_ptr_t *vfp,
|
||||
int forced_stop, // 0 - full, 1 - qtr only, 2 - half only
|
||||
int iters_per_step, int *cost_list, int *mvjcost, int *mvcost[2],
|
||||
int *distortion, unsigned int *sse1, const uint8_t *second_pred, int w,
|
||||
int h, int use_upsampled_ref);
|
||||
int *distortion, unsigned int *sse1, const uint8_t *second_pred,
|
||||
#if CONFIG_EXT_INTER
|
||||
const uint8_t *mask, int mask_stride, int invert_mask,
|
||||
#endif
|
||||
int w, int h, int use_upsampled_ref);
|
||||
|
||||
extern fractional_mv_step_fp av1_find_best_sub_pixel_tree;
|
||||
extern fractional_mv_step_fp av1_find_best_sub_pixel_tree_pruned;
|
||||
|
|
@ -113,6 +123,10 @@ typedef int (*av1_diamond_search_fn_t)(
|
|||
|
||||
int av1_refining_search_8p_c(MACROBLOCK *x, int error_per_bit, int search_range,
|
||||
const aom_variance_fn_ptr_t *fn_ptr,
|
||||
#if CONFIG_EXT_INTER
|
||||
const uint8_t *mask, int mask_stride,
|
||||
int invert_mask,
|
||||
#endif
|
||||
const MV *center_mv, const uint8_t *second_pred);
|
||||
|
||||
struct AV1_COMP;
|
||||
|
|
@ -122,27 +136,6 @@ int av1_full_pixel_search(const struct AV1_COMP *cpi, MACROBLOCK *x,
|
|||
int error_per_bit, int *cost_list, const MV *ref_mv,
|
||||
int var_max, int rd);
|
||||
|
||||
#if CONFIG_EXT_INTER
|
||||
int av1_find_best_masked_sub_pixel_tree(
|
||||
const MACROBLOCK *x, const uint8_t *mask, int mask_stride, MV *bestmv,
|
||||
const MV *ref_mv, int allow_hp, int error_per_bit,
|
||||
const aom_variance_fn_ptr_t *vfp, int forced_stop, int iters_per_step,
|
||||
int *mvjcost, int *mvcost[2], int *distortion, unsigned int *sse1,
|
||||
int is_second);
|
||||
int av1_find_best_masked_sub_pixel_tree_up(
|
||||
const struct AV1_COMP *cpi, MACROBLOCK *x, const uint8_t *mask,
|
||||
int mask_stride, int mi_row, int mi_col, MV *bestmv, const MV *ref_mv,
|
||||
int allow_hp, int error_per_bit, const aom_variance_fn_ptr_t *vfp,
|
||||
int forced_stop, int iters_per_step, int *mvjcost, int *mvcost[2],
|
||||
int *distortion, unsigned int *sse1, int is_second, int use_upsampled_ref);
|
||||
int av1_masked_full_pixel_diamond(const struct AV1_COMP *cpi, MACROBLOCK *x,
|
||||
const uint8_t *mask, int mask_stride,
|
||||
MV *mvp_full, int step_param, int sadpb,
|
||||
int further_steps, int do_refine,
|
||||
const aom_variance_fn_ptr_t *fn_ptr,
|
||||
const MV *ref_mv, MV *dst_mv, int is_second);
|
||||
#endif // CONFIG_EXT_INTER
|
||||
|
||||
#if CONFIG_MOTION_VAR
|
||||
int av1_obmc_full_pixel_diamond(const struct AV1_COMP *cpi, MACROBLOCK *x,
|
||||
MV *mvp_full, int step_param, int sadpb,
|
||||
|
|
@ -160,4 +153,14 @@ int av1_find_best_obmc_sub_pixel_tree_up(
|
|||
} // extern "C"
|
||||
#endif
|
||||
|
||||
#if CONFIG_WARPED_MOTION
|
||||
unsigned int av1_compute_motion_cost(const struct AV1_COMP *cpi,
|
||||
MACROBLOCK *const x, BLOCK_SIZE bsize,
|
||||
int mi_row, int mi_col, const MV *this_mv);
|
||||
unsigned int av1_refine_warped_mv(const struct AV1_COMP *cpi,
|
||||
MACROBLOCK *const x, BLOCK_SIZE bsize,
|
||||
int mi_row, int mi_col, int *pts,
|
||||
int *pts_inref);
|
||||
#endif // CONFIG_WARPED_MOTION
|
||||
|
||||
#endif // AV1_ENCODER_MCOMP_H_
|
||||
|
|
|
|||
107
third_party/aom/av1/encoder/palette.c
vendored
107
third_party/aom/av1/encoder/palette.c
vendored
|
|
@ -167,31 +167,58 @@ int av1_count_colors(const uint8_t *src, int stride, int rows, int cols) {
|
|||
}
|
||||
|
||||
#if CONFIG_PALETTE_DELTA_ENCODING
|
||||
int av1_get_palette_delta_bits_y(const PALETTE_MODE_INFO *const pmi,
|
||||
int bit_depth, int *min_bits) {
|
||||
const int n = pmi->palette_size[0];
|
||||
int max_d = 0, i;
|
||||
*min_bits = bit_depth - 3;
|
||||
for (i = 1; i < n; ++i) {
|
||||
const int delta = pmi->palette_colors[i] - pmi->palette_colors[i - 1];
|
||||
assert(delta > 0);
|
||||
if (delta > max_d) max_d = delta;
|
||||
static int delta_encode_cost(const int *colors, int num, int bit_depth,
|
||||
int min_val) {
|
||||
if (num <= 0) return 0;
|
||||
int bits_cost = bit_depth;
|
||||
if (num == 1) return bits_cost;
|
||||
bits_cost += 2;
|
||||
int max_delta = 0;
|
||||
int deltas[PALETTE_MAX_SIZE];
|
||||
const int min_bits = bit_depth - 3;
|
||||
for (int i = 1; i < num; ++i) {
|
||||
const int delta = colors[i] - colors[i - 1];
|
||||
deltas[i - 1] = delta;
|
||||
assert(delta >= min_val);
|
||||
if (delta > max_delta) max_delta = delta;
|
||||
}
|
||||
return AOMMAX(av1_ceil_log2(max_d), *min_bits);
|
||||
int bits_per_delta = AOMMAX(av1_ceil_log2(max_delta + 1 - min_val), min_bits);
|
||||
assert(bits_per_delta <= bit_depth);
|
||||
int range = (1 << bit_depth) - colors[0] - min_val;
|
||||
for (int i = 0; i < num - 1; ++i) {
|
||||
bits_cost += bits_per_delta;
|
||||
range -= deltas[i];
|
||||
bits_per_delta = AOMMIN(bits_per_delta, av1_ceil_log2(range));
|
||||
}
|
||||
return bits_cost;
|
||||
}
|
||||
|
||||
int av1_get_palette_delta_bits_u(const PALETTE_MODE_INFO *const pmi,
|
||||
int bit_depth, int *min_bits) {
|
||||
const int n = pmi->palette_size[1];
|
||||
int max_d = 0, i;
|
||||
*min_bits = bit_depth - 3;
|
||||
for (i = 1; i < n; ++i) {
|
||||
const int delta = pmi->palette_colors[PALETTE_MAX_SIZE + i] -
|
||||
pmi->palette_colors[PALETTE_MAX_SIZE + i - 1];
|
||||
assert(delta >= 0);
|
||||
if (delta > max_d) max_d = delta;
|
||||
int av1_index_color_cache(const uint16_t *color_cache, int n_cache,
|
||||
const uint16_t *colors, int n_colors,
|
||||
uint8_t *cache_color_found, int *out_cache_colors) {
|
||||
if (n_cache <= 0) {
|
||||
for (int i = 0; i < n_colors; ++i) out_cache_colors[i] = colors[i];
|
||||
return n_colors;
|
||||
}
|
||||
return AOMMAX(av1_ceil_log2(max_d + 1), *min_bits);
|
||||
memset(cache_color_found, 0, n_cache * sizeof(*cache_color_found));
|
||||
int n_in_cache = 0;
|
||||
int in_cache_flags[PALETTE_MAX_SIZE];
|
||||
memset(in_cache_flags, 0, sizeof(in_cache_flags));
|
||||
for (int i = 0; i < n_cache && n_in_cache < n_colors; ++i) {
|
||||
for (int j = 0; j < n_colors; ++j) {
|
||||
if (colors[j] == color_cache[i]) {
|
||||
in_cache_flags[j] = 1;
|
||||
cache_color_found[i] = 1;
|
||||
++n_in_cache;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
int j = 0;
|
||||
for (int i = 0; i < n_colors; ++i)
|
||||
if (!in_cache_flags[i]) out_cache_colors[j++] = colors[i];
|
||||
assert(j == n_colors - n_in_cache);
|
||||
return j;
|
||||
}
|
||||
|
||||
int av1_get_palette_delta_bits_v(const PALETTE_MODE_INFO *const pmi,
|
||||
|
|
@ -199,10 +226,10 @@ int av1_get_palette_delta_bits_v(const PALETTE_MODE_INFO *const pmi,
|
|||
int *min_bits) {
|
||||
const int n = pmi->palette_size[1];
|
||||
const int max_val = 1 << bit_depth;
|
||||
int max_d = 0, i;
|
||||
int max_d = 0;
|
||||
*min_bits = bit_depth - 4;
|
||||
*zero_count = 0;
|
||||
for (i = 1; i < n; ++i) {
|
||||
for (int i = 1; i < n; ++i) {
|
||||
const int delta = pmi->palette_colors[2 * PALETTE_MAX_SIZE + i] -
|
||||
pmi->palette_colors[2 * PALETTE_MAX_SIZE + i - 1];
|
||||
const int v = abs(delta);
|
||||
|
|
@ -215,26 +242,42 @@ int av1_get_palette_delta_bits_v(const PALETTE_MODE_INFO *const pmi,
|
|||
#endif // CONFIG_PALETTE_DELTA_ENCODING
|
||||
|
||||
int av1_palette_color_cost_y(const PALETTE_MODE_INFO *const pmi,
|
||||
#if CONFIG_PALETTE_DELTA_ENCODING
|
||||
uint16_t *color_cache, int n_cache,
|
||||
#endif // CONFIG_PALETTE_DELTA_ENCODING
|
||||
int bit_depth) {
|
||||
const int n = pmi->palette_size[0];
|
||||
#if CONFIG_PALETTE_DELTA_ENCODING
|
||||
int min_bits = 0;
|
||||
const int bits = av1_get_palette_delta_bits_y(pmi, bit_depth, &min_bits);
|
||||
return av1_cost_bit(128, 0) * (2 + bit_depth + bits * (n - 1));
|
||||
int out_cache_colors[PALETTE_MAX_SIZE];
|
||||
uint8_t cache_color_found[2 * PALETTE_MAX_SIZE];
|
||||
const int n_out_cache =
|
||||
av1_index_color_cache(color_cache, n_cache, pmi->palette_colors, n,
|
||||
cache_color_found, out_cache_colors);
|
||||
const int total_bits =
|
||||
n_cache + delta_encode_cost(out_cache_colors, n_out_cache, bit_depth, 1);
|
||||
return total_bits * av1_cost_bit(128, 0);
|
||||
#else
|
||||
return bit_depth * n * av1_cost_bit(128, 0);
|
||||
#endif // CONFIG_PALETTE_DELTA_ENCODING
|
||||
}
|
||||
|
||||
int av1_palette_color_cost_uv(const PALETTE_MODE_INFO *const pmi,
|
||||
#if CONFIG_PALETTE_DELTA_ENCODING
|
||||
uint16_t *color_cache, int n_cache,
|
||||
#endif // CONFIG_PALETTE_DELTA_ENCODING
|
||||
int bit_depth) {
|
||||
const int n = pmi->palette_size[1];
|
||||
#if CONFIG_PALETTE_DELTA_ENCODING
|
||||
int cost = 0;
|
||||
int total_bits = 0;
|
||||
// U channel palette color cost.
|
||||
int min_bits_u = 0;
|
||||
const int bits_u = av1_get_palette_delta_bits_u(pmi, bit_depth, &min_bits_u);
|
||||
cost += av1_cost_bit(128, 0) * (2 + bit_depth + bits_u * (n - 1));
|
||||
int out_cache_colors[PALETTE_MAX_SIZE];
|
||||
uint8_t cache_color_found[2 * PALETTE_MAX_SIZE];
|
||||
const int n_out_cache = av1_index_color_cache(
|
||||
color_cache, n_cache, pmi->palette_colors + PALETTE_MAX_SIZE, n,
|
||||
cache_color_found, out_cache_colors);
|
||||
total_bits +=
|
||||
n_cache + delta_encode_cost(out_cache_colors, n_out_cache, bit_depth, 0);
|
||||
|
||||
// V channel palette color cost.
|
||||
int zero_count = 0, min_bits_v = 0;
|
||||
const int bits_v =
|
||||
|
|
@ -242,8 +285,8 @@ int av1_palette_color_cost_uv(const PALETTE_MODE_INFO *const pmi,
|
|||
const int bits_using_delta =
|
||||
2 + bit_depth + (bits_v + 1) * (n - 1) - zero_count;
|
||||
const int bits_using_raw = bit_depth * n;
|
||||
cost += av1_cost_bit(128, 0) * (1 + AOMMIN(bits_using_delta, bits_using_raw));
|
||||
return cost;
|
||||
total_bits += 1 + AOMMIN(bits_using_delta, bits_using_raw);
|
||||
return total_bits * av1_cost_bit(128, 0);
|
||||
#else
|
||||
return 2 * bit_depth * n * av1_cost_bit(128, 0);
|
||||
#endif // CONFIG_PALETTE_DELTA_ENCODING
|
||||
|
|
|
|||
22
third_party/aom/av1/encoder/palette.h
vendored
22
third_party/aom/av1/encoder/palette.h
vendored
|
|
@ -45,13 +45,12 @@ int av1_count_colors_highbd(const uint8_t *src8, int stride, int rows, int cols,
|
|||
#endif // CONFIG_HIGHBITDEPTH
|
||||
|
||||
#if CONFIG_PALETTE_DELTA_ENCODING
|
||||
// Return the number of bits used to transmit each luma palette color delta.
|
||||
int av1_get_palette_delta_bits_y(const PALETTE_MODE_INFO *const pmi,
|
||||
int bit_depth, int *min_bits);
|
||||
|
||||
// Return the number of bits used to transmit each U palette color delta.
|
||||
int av1_get_palette_delta_bits_u(const PALETTE_MODE_INFO *const pmi,
|
||||
int bit_depth, int *min_bits);
|
||||
// Given a color cache and a set of base colors, find if each cache color is
|
||||
// present in the base colors, record the binary results in "cache_color_found".
|
||||
// Record the colors that are not in the color cache in "out_cache_colors".
|
||||
int av1_index_color_cache(const uint16_t *color_cache, int n_cache,
|
||||
const uint16_t *colors, int n_colors,
|
||||
uint8_t *cache_color_found, int *out_cache_colors);
|
||||
|
||||
// Return the number of bits used to transmit each v palette color delta;
|
||||
// assign zero_count with the number of deltas being 0.
|
||||
|
|
@ -60,10 +59,17 @@ int av1_get_palette_delta_bits_v(const PALETTE_MODE_INFO *const pmi,
|
|||
#endif // CONFIG_PALETTE_DELTA_ENCODING
|
||||
|
||||
// Return the rate cost for transmitting luma palette color values.
|
||||
int av1_palette_color_cost_y(const PALETTE_MODE_INFO *const pmi, int bit_depth);
|
||||
int av1_palette_color_cost_y(const PALETTE_MODE_INFO *const pmi,
|
||||
#if CONFIG_PALETTE_DELTA_ENCODING
|
||||
uint16_t *color_cache, int n_cache,
|
||||
#endif // CONFIG_PALETTE_DELTA_ENCODING
|
||||
int bit_depth);
|
||||
|
||||
// Return the rate cost for transmitting chroma palette color values.
|
||||
int av1_palette_color_cost_uv(const PALETTE_MODE_INFO *const pmi,
|
||||
#if CONFIG_PALETTE_DELTA_ENCODING
|
||||
uint16_t *color_cache, int n_cache,
|
||||
#endif // CONFIG_PALETTE_DELTA_ENCODING
|
||||
int bit_depth);
|
||||
|
||||
#ifdef __cplusplus
|
||||
|
|
|
|||
694
third_party/aom/av1/encoder/pickrst.c
vendored
694
third_party/aom/av1/encoder/pickrst.c
vendored
|
|
@ -31,17 +31,18 @@
|
|||
#include "av1/encoder/encoder.h"
|
||||
#include "av1/encoder/picklpf.h"
|
||||
#include "av1/encoder/pickrst.h"
|
||||
#include "av1/encoder/mathutils.h"
|
||||
|
||||
// When set to RESTORE_WIENER or RESTORE_SGRPROJ only those are allowed.
|
||||
// When set to RESTORE_NONE (0) we allow switchable.
|
||||
const RestorationType force_restore_type = RESTORE_NONE;
|
||||
|
||||
// Number of Wiener iterations
|
||||
#define NUM_WIENER_ITERS 10
|
||||
#define NUM_WIENER_ITERS 5
|
||||
|
||||
typedef double (*search_restore_type)(const YV12_BUFFER_CONFIG *src,
|
||||
AV1_COMP *cpi, int partial_frame,
|
||||
RestorationInfo *info,
|
||||
int plane, RestorationInfo *info,
|
||||
RestorationType *rest_level,
|
||||
double *best_tile_cost,
|
||||
YV12_BUFFER_CONFIG *dst_frame);
|
||||
|
|
@ -216,6 +217,62 @@ static int64_t get_pixel_proj_error(uint8_t *src8, int width, int height,
|
|||
return err;
|
||||
}
|
||||
|
||||
#define USE_SGRPROJ_REFINEMENT_SEARCH 1
|
||||
static int64_t finer_search_pixel_proj_error(
|
||||
uint8_t *src8, int width, int height, int src_stride, uint8_t *dat8,
|
||||
int dat_stride, int bit_depth, int32_t *flt1, int flt1_stride,
|
||||
int32_t *flt2, int flt2_stride, int start_step, int *xqd) {
|
||||
int64_t err = get_pixel_proj_error(src8, width, height, src_stride, dat8,
|
||||
dat_stride, bit_depth, flt1, flt1_stride,
|
||||
flt2, flt2_stride, xqd);
|
||||
(void)start_step;
|
||||
#if USE_SGRPROJ_REFINEMENT_SEARCH
|
||||
int64_t err2;
|
||||
int tap_min[] = { SGRPROJ_PRJ_MIN0, SGRPROJ_PRJ_MIN1 };
|
||||
int tap_max[] = { SGRPROJ_PRJ_MAX0, SGRPROJ_PRJ_MAX1 };
|
||||
for (int s = start_step; s >= 1; s >>= 1) {
|
||||
for (int p = 0; p < 2; ++p) {
|
||||
int skip = 0;
|
||||
do {
|
||||
if (xqd[p] - s >= tap_min[p]) {
|
||||
xqd[p] -= s;
|
||||
err2 = get_pixel_proj_error(src8, width, height, src_stride, dat8,
|
||||
dat_stride, bit_depth, flt1, flt1_stride,
|
||||
flt2, flt2_stride, xqd);
|
||||
if (err2 > err) {
|
||||
xqd[p] += s;
|
||||
} else {
|
||||
err = err2;
|
||||
skip = 1;
|
||||
// At the highest step size continue moving in the same direction
|
||||
if (s == start_step) continue;
|
||||
}
|
||||
}
|
||||
break;
|
||||
} while (1);
|
||||
if (skip) break;
|
||||
do {
|
||||
if (xqd[p] + s <= tap_max[p]) {
|
||||
xqd[p] += s;
|
||||
err2 = get_pixel_proj_error(src8, width, height, src_stride, dat8,
|
||||
dat_stride, bit_depth, flt1, flt1_stride,
|
||||
flt2, flt2_stride, xqd);
|
||||
if (err2 > err) {
|
||||
xqd[p] -= s;
|
||||
} else {
|
||||
err = err2;
|
||||
// At the highest step size continue moving in the same direction
|
||||
if (s == start_step) continue;
|
||||
}
|
||||
}
|
||||
break;
|
||||
} while (1);
|
||||
}
|
||||
}
|
||||
#endif // USE_SGRPROJ_REFINEMENT_SEARCH
|
||||
return err;
|
||||
}
|
||||
|
||||
static void get_proj_subspace(uint8_t *src8, int width, int height,
|
||||
int src_stride, uint8_t *dat8, int dat_stride,
|
||||
int bit_depth, int32_t *flt1, int flt1_stride,
|
||||
|
|
@ -329,12 +386,14 @@ static void search_selfguided_restoration(uint8_t *dat8, int width, int height,
|
|||
#if CONFIG_HIGHBITDEPTH
|
||||
}
|
||||
#endif
|
||||
aom_clear_system_state();
|
||||
get_proj_subspace(src8, width, height, src_stride, dat8, dat_stride,
|
||||
bit_depth, flt1, width, flt2, width, exq);
|
||||
aom_clear_system_state();
|
||||
encode_xq(exq, exqd);
|
||||
err =
|
||||
get_pixel_proj_error(src8, width, height, src_stride, dat8, dat_stride,
|
||||
bit_depth, flt1, width, flt2, width, exqd);
|
||||
err = finer_search_pixel_proj_error(src8, width, height, src_stride, dat8,
|
||||
dat_stride, bit_depth, flt1, width,
|
||||
flt2, width, 2, exqd);
|
||||
if (besterr == -1 || err < besterr) {
|
||||
bestep = ep;
|
||||
besterr = err;
|
||||
|
|
@ -362,8 +421,9 @@ static int count_sgrproj_bits(SgrprojInfo *sgrproj_info,
|
|||
}
|
||||
|
||||
static double search_sgrproj(const YV12_BUFFER_CONFIG *src, AV1_COMP *cpi,
|
||||
int partial_frame, RestorationInfo *info,
|
||||
RestorationType *type, double *best_tile_cost,
|
||||
int partial_frame, int plane,
|
||||
RestorationInfo *info, RestorationType *type,
|
||||
double *best_tile_cost,
|
||||
YV12_BUFFER_CONFIG *dst_frame) {
|
||||
SgrprojInfo *sgrproj_info = info->sgrproj_info;
|
||||
double err, cost_norestore, cost_sgrproj;
|
||||
|
|
@ -374,44 +434,68 @@ static double search_sgrproj(const YV12_BUFFER_CONFIG *src, AV1_COMP *cpi,
|
|||
RestorationInfo *rsi = &cpi->rst_search[0];
|
||||
int tile_idx, tile_width, tile_height, nhtiles, nvtiles;
|
||||
int h_start, h_end, v_start, v_end;
|
||||
// Allocate for the src buffer at high precision
|
||||
const int ntiles = av1_get_rest_ntiles(
|
||||
cm->width, cm->height, cm->rst_info[0].restoration_tilesize, &tile_width,
|
||||
&tile_height, &nhtiles, &nvtiles);
|
||||
int width, height, src_stride, dgd_stride;
|
||||
uint8_t *dgd_buffer, *src_buffer;
|
||||
if (plane == AOM_PLANE_Y) {
|
||||
width = cm->width;
|
||||
height = cm->height;
|
||||
src_buffer = src->y_buffer;
|
||||
src_stride = src->y_stride;
|
||||
dgd_buffer = dgd->y_buffer;
|
||||
dgd_stride = dgd->y_stride;
|
||||
assert(width == dgd->y_crop_width);
|
||||
assert(height == dgd->y_crop_height);
|
||||
assert(width == src->y_crop_width);
|
||||
assert(height == src->y_crop_height);
|
||||
} else {
|
||||
width = src->uv_crop_width;
|
||||
height = src->uv_crop_height;
|
||||
src_stride = src->uv_stride;
|
||||
dgd_stride = dgd->uv_stride;
|
||||
src_buffer = plane == AOM_PLANE_U ? src->u_buffer : src->v_buffer;
|
||||
dgd_buffer = plane == AOM_PLANE_U ? dgd->u_buffer : dgd->v_buffer;
|
||||
assert(width == dgd->uv_crop_width);
|
||||
assert(height == dgd->uv_crop_height);
|
||||
}
|
||||
const int ntiles =
|
||||
av1_get_rest_ntiles(width, height, cm->rst_info[0].restoration_tilesize,
|
||||
&tile_width, &tile_height, &nhtiles, &nvtiles);
|
||||
SgrprojInfo ref_sgrproj_info;
|
||||
set_default_sgrproj(&ref_sgrproj_info);
|
||||
|
||||
rsi->frame_restoration_type = RESTORE_SGRPROJ;
|
||||
rsi[plane].frame_restoration_type = RESTORE_SGRPROJ;
|
||||
|
||||
for (tile_idx = 0; tile_idx < ntiles; ++tile_idx) {
|
||||
rsi->restoration_type[tile_idx] = RESTORE_NONE;
|
||||
rsi[plane].restoration_type[tile_idx] = RESTORE_NONE;
|
||||
}
|
||||
// Compute best Sgrproj filters for each tile
|
||||
for (tile_idx = 0; tile_idx < ntiles; ++tile_idx) {
|
||||
av1_get_rest_tile_limits(tile_idx, 0, 0, nhtiles, nvtiles, tile_width,
|
||||
tile_height, cm->width, cm->height, 0, 0, &h_start,
|
||||
&h_end, &v_start, &v_end);
|
||||
tile_height, width, height, 0, 0, &h_start, &h_end,
|
||||
&v_start, &v_end);
|
||||
err = sse_restoration_tile(src, cm->frame_to_show, cm, h_start,
|
||||
h_end - h_start, v_start, v_end - v_start, 1);
|
||||
h_end - h_start, v_start, v_end - v_start,
|
||||
(1 << plane));
|
||||
// #bits when a tile is not restored
|
||||
bits = av1_cost_bit(RESTORE_NONE_SGRPROJ_PROB, 0);
|
||||
cost_norestore = RDCOST_DBL(x->rdmult, x->rddiv, (bits >> 4), err);
|
||||
best_tile_cost[tile_idx] = DBL_MAX;
|
||||
search_selfguided_restoration(
|
||||
dgd->y_buffer + v_start * dgd->y_stride + h_start, h_end - h_start,
|
||||
v_end - v_start, dgd->y_stride,
|
||||
src->y_buffer + v_start * src->y_stride + h_start, src->y_stride,
|
||||
dgd_buffer + v_start * dgd_stride + h_start, h_end - h_start,
|
||||
v_end - v_start, dgd_stride,
|
||||
src_buffer + v_start * src_stride + h_start, src_stride,
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
cm->bit_depth,
|
||||
#else
|
||||
8,
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
&rsi->sgrproj_info[tile_idx].ep, rsi->sgrproj_info[tile_idx].xqd,
|
||||
cm->rst_internal.tmpbuf);
|
||||
rsi->restoration_type[tile_idx] = RESTORE_SGRPROJ;
|
||||
err = try_restoration_tile(src, cpi, rsi, 1, partial_frame, tile_idx, 0, 0,
|
||||
dst_frame);
|
||||
bits = count_sgrproj_bits(&rsi->sgrproj_info[tile_idx], &ref_sgrproj_info)
|
||||
&rsi[plane].sgrproj_info[tile_idx].ep,
|
||||
rsi[plane].sgrproj_info[tile_idx].xqd, cm->rst_internal.tmpbuf);
|
||||
rsi[plane].restoration_type[tile_idx] = RESTORE_SGRPROJ;
|
||||
err = try_restoration_tile(src, cpi, rsi, (1 << plane), partial_frame,
|
||||
tile_idx, 0, 0, dst_frame);
|
||||
bits = count_sgrproj_bits(&rsi[plane].sgrproj_info[tile_idx],
|
||||
&ref_sgrproj_info)
|
||||
<< AV1_PROB_COST_SHIFT;
|
||||
bits += av1_cost_bit(RESTORE_NONE_SGRPROJ_PROB, 1);
|
||||
cost_sgrproj = RDCOST_DBL(x->rdmult, x->rddiv, (bits >> 4), err);
|
||||
|
|
@ -419,35 +503,34 @@ static double search_sgrproj(const YV12_BUFFER_CONFIG *src, AV1_COMP *cpi,
|
|||
type[tile_idx] = RESTORE_NONE;
|
||||
} else {
|
||||
type[tile_idx] = RESTORE_SGRPROJ;
|
||||
memcpy(&sgrproj_info[tile_idx], &rsi->sgrproj_info[tile_idx],
|
||||
memcpy(&sgrproj_info[tile_idx], &rsi[plane].sgrproj_info[tile_idx],
|
||||
sizeof(sgrproj_info[tile_idx]));
|
||||
bits = count_sgrproj_bits(&rsi->sgrproj_info[tile_idx], &ref_sgrproj_info)
|
||||
<< AV1_PROB_COST_SHIFT;
|
||||
memcpy(&ref_sgrproj_info, &sgrproj_info[tile_idx],
|
||||
sizeof(ref_sgrproj_info));
|
||||
best_tile_cost[tile_idx] = err;
|
||||
}
|
||||
rsi->restoration_type[tile_idx] = RESTORE_NONE;
|
||||
rsi[plane].restoration_type[tile_idx] = RESTORE_NONE;
|
||||
}
|
||||
// Cost for Sgrproj filtering
|
||||
set_default_sgrproj(&ref_sgrproj_info);
|
||||
bits = frame_level_restore_bits[rsi->frame_restoration_type]
|
||||
bits = frame_level_restore_bits[rsi[plane].frame_restoration_type]
|
||||
<< AV1_PROB_COST_SHIFT;
|
||||
for (tile_idx = 0; tile_idx < ntiles; ++tile_idx) {
|
||||
bits +=
|
||||
av1_cost_bit(RESTORE_NONE_SGRPROJ_PROB, type[tile_idx] != RESTORE_NONE);
|
||||
memcpy(&rsi->sgrproj_info[tile_idx], &sgrproj_info[tile_idx],
|
||||
memcpy(&rsi[plane].sgrproj_info[tile_idx], &sgrproj_info[tile_idx],
|
||||
sizeof(sgrproj_info[tile_idx]));
|
||||
if (type[tile_idx] == RESTORE_SGRPROJ) {
|
||||
bits +=
|
||||
count_sgrproj_bits(&rsi->sgrproj_info[tile_idx], &ref_sgrproj_info)
|
||||
<< AV1_PROB_COST_SHIFT;
|
||||
memcpy(&ref_sgrproj_info, &rsi->sgrproj_info[tile_idx],
|
||||
bits += count_sgrproj_bits(&rsi[plane].sgrproj_info[tile_idx],
|
||||
&ref_sgrproj_info)
|
||||
<< AV1_PROB_COST_SHIFT;
|
||||
memcpy(&ref_sgrproj_info, &rsi[plane].sgrproj_info[tile_idx],
|
||||
sizeof(ref_sgrproj_info));
|
||||
}
|
||||
rsi->restoration_type[tile_idx] = type[tile_idx];
|
||||
rsi[plane].restoration_type[tile_idx] = type[tile_idx];
|
||||
}
|
||||
err = try_restoration_frame(src, cpi, rsi, 1, partial_frame, dst_frame);
|
||||
err = try_restoration_frame(src, cpi, rsi, (1 << plane), partial_frame,
|
||||
dst_frame);
|
||||
cost_sgrproj = RDCOST_DBL(x->rdmult, x->rddiv, (bits >> 4), err);
|
||||
|
||||
return cost_sgrproj;
|
||||
|
|
@ -560,46 +643,6 @@ static void compute_stats_highbd(uint8_t *dgd8, uint8_t *src8, int h_start,
|
|||
}
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
|
||||
// Solves Ax = b, where x and b are column vectors
|
||||
static int linsolve(int n, double *A, int stride, double *b, double *x) {
|
||||
int i, j, k;
|
||||
double c;
|
||||
|
||||
aom_clear_system_state();
|
||||
|
||||
// Forward elimination
|
||||
for (k = 0; k < n - 1; k++) {
|
||||
// Bring the largest magitude to the diagonal position
|
||||
for (i = n - 1; i > k; i--) {
|
||||
if (fabs(A[(i - 1) * stride + k]) < fabs(A[i * stride + k])) {
|
||||
for (j = 0; j < n; j++) {
|
||||
c = A[i * stride + j];
|
||||
A[i * stride + j] = A[(i - 1) * stride + j];
|
||||
A[(i - 1) * stride + j] = c;
|
||||
}
|
||||
c = b[i];
|
||||
b[i] = b[i - 1];
|
||||
b[i - 1] = c;
|
||||
}
|
||||
}
|
||||
for (i = k; i < n - 1; i++) {
|
||||
if (fabs(A[k * stride + k]) < 1e-10) return 0;
|
||||
c = A[(i + 1) * stride + k] / A[k * stride + k];
|
||||
for (j = 0; j < n; j++) A[(i + 1) * stride + j] -= c * A[k * stride + j];
|
||||
b[i + 1] -= c * b[k];
|
||||
}
|
||||
}
|
||||
// Backward substitution
|
||||
for (i = n - 1; i >= 0; i--) {
|
||||
if (fabs(A[i * stride + i]) < 1e-10) return 0;
|
||||
c = 0;
|
||||
for (j = i + 1; j <= n - 1; j++) c += A[i * stride + j] * x[j];
|
||||
x[i] = (b[i] - c) / A[i * stride + i];
|
||||
}
|
||||
|
||||
return 1;
|
||||
}
|
||||
|
||||
static INLINE int wrap_index(int i) {
|
||||
return (i >= WIENER_HALFWIN1 ? WIENER_WIN - 1 - i : i);
|
||||
}
|
||||
|
|
@ -696,8 +739,10 @@ static void update_b_sep_sym(double **Mc, double **Hc, double *a, double *b) {
|
|||
|
||||
static int wiener_decompose_sep_sym(double *M, double *H, double *a,
|
||||
double *b) {
|
||||
static const double init_filt[WIENER_WIN] = {
|
||||
0.035623, -0.127154, 0.211436, 0.760190, 0.211436, -0.127154, 0.035623,
|
||||
static const int init_filt[WIENER_WIN] = {
|
||||
WIENER_FILT_TAP0_MIDV, WIENER_FILT_TAP1_MIDV, WIENER_FILT_TAP2_MIDV,
|
||||
WIENER_FILT_TAP3_MIDV, WIENER_FILT_TAP2_MIDV, WIENER_FILT_TAP1_MIDV,
|
||||
WIENER_FILT_TAP0_MIDV,
|
||||
};
|
||||
int i, j, iter;
|
||||
double *Hc[WIENER_WIN2];
|
||||
|
|
@ -709,8 +754,9 @@ static int wiener_decompose_sep_sym(double *M, double *H, double *a,
|
|||
H + i * WIENER_WIN * WIENER_WIN2 + j * WIENER_WIN;
|
||||
}
|
||||
}
|
||||
memcpy(a, init_filt, sizeof(*a) * WIENER_WIN);
|
||||
memcpy(b, init_filt, sizeof(*b) * WIENER_WIN);
|
||||
for (i = 0; i < WIENER_WIN; i++) {
|
||||
a[i] = b[i] = (double)init_filt[i] / WIENER_FILT_STEP;
|
||||
}
|
||||
|
||||
iter = 1;
|
||||
while (iter < NUM_WIENER_ITERS) {
|
||||
|
|
@ -812,40 +858,161 @@ static int count_wiener_bits(WienerInfo *wiener_info,
|
|||
return bits;
|
||||
}
|
||||
|
||||
static double search_wiener_uv(const YV12_BUFFER_CONFIG *src, AV1_COMP *cpi,
|
||||
int partial_frame, int plane,
|
||||
RestorationInfo *info, RestorationType *type,
|
||||
YV12_BUFFER_CONFIG *dst_frame) {
|
||||
#define USE_WIENER_REFINEMENT_SEARCH 1
|
||||
static int64_t finer_tile_search_wiener(const YV12_BUFFER_CONFIG *src,
|
||||
AV1_COMP *cpi, RestorationInfo *rsi,
|
||||
int start_step, int plane, int tile_idx,
|
||||
int partial_frame,
|
||||
YV12_BUFFER_CONFIG *dst_frame) {
|
||||
int64_t err = try_restoration_tile(src, cpi, rsi, 1 << plane, partial_frame,
|
||||
tile_idx, 0, 0, dst_frame);
|
||||
(void)start_step;
|
||||
#if USE_WIENER_REFINEMENT_SEARCH
|
||||
int64_t err2;
|
||||
int tap_min[] = { WIENER_FILT_TAP0_MINV, WIENER_FILT_TAP1_MINV,
|
||||
WIENER_FILT_TAP2_MINV };
|
||||
int tap_max[] = { WIENER_FILT_TAP0_MAXV, WIENER_FILT_TAP1_MAXV,
|
||||
WIENER_FILT_TAP2_MAXV };
|
||||
// printf("err pre = %"PRId64"\n", err);
|
||||
for (int s = start_step; s >= 1; s >>= 1) {
|
||||
for (int p = 0; p < WIENER_HALFWIN; ++p) {
|
||||
int skip = 0;
|
||||
do {
|
||||
if (rsi[plane].wiener_info[tile_idx].hfilter[p] - s >= tap_min[p]) {
|
||||
rsi[plane].wiener_info[tile_idx].hfilter[p] -= s;
|
||||
rsi[plane].wiener_info[tile_idx].hfilter[WIENER_WIN - p - 1] -= s;
|
||||
rsi[plane].wiener_info[tile_idx].hfilter[WIENER_HALFWIN] += 2 * s;
|
||||
err2 = try_restoration_tile(src, cpi, rsi, 1 << plane, partial_frame,
|
||||
tile_idx, 0, 0, dst_frame);
|
||||
if (err2 > err) {
|
||||
rsi[plane].wiener_info[tile_idx].hfilter[p] += s;
|
||||
rsi[plane].wiener_info[tile_idx].hfilter[WIENER_WIN - p - 1] += s;
|
||||
rsi[plane].wiener_info[tile_idx].hfilter[WIENER_HALFWIN] -= 2 * s;
|
||||
} else {
|
||||
err = err2;
|
||||
skip = 1;
|
||||
// At the highest step size continue moving in the same direction
|
||||
if (s == start_step) continue;
|
||||
}
|
||||
}
|
||||
break;
|
||||
} while (1);
|
||||
if (skip) break;
|
||||
do {
|
||||
if (rsi[plane].wiener_info[tile_idx].hfilter[p] + s <= tap_max[p]) {
|
||||
rsi[plane].wiener_info[tile_idx].hfilter[p] += s;
|
||||
rsi[plane].wiener_info[tile_idx].hfilter[WIENER_WIN - p - 1] += s;
|
||||
rsi[plane].wiener_info[tile_idx].hfilter[WIENER_HALFWIN] -= 2 * s;
|
||||
err2 = try_restoration_tile(src, cpi, rsi, 1 << plane, partial_frame,
|
||||
tile_idx, 0, 0, dst_frame);
|
||||
if (err2 > err) {
|
||||
rsi[plane].wiener_info[tile_idx].hfilter[p] -= s;
|
||||
rsi[plane].wiener_info[tile_idx].hfilter[WIENER_WIN - p - 1] -= s;
|
||||
rsi[plane].wiener_info[tile_idx].hfilter[WIENER_HALFWIN] += 2 * s;
|
||||
} else {
|
||||
err = err2;
|
||||
// At the highest step size continue moving in the same direction
|
||||
if (s == start_step) continue;
|
||||
}
|
||||
}
|
||||
break;
|
||||
} while (1);
|
||||
}
|
||||
for (int p = 0; p < WIENER_HALFWIN; ++p) {
|
||||
int skip = 0;
|
||||
do {
|
||||
if (rsi[plane].wiener_info[tile_idx].vfilter[p] - s >= tap_min[p]) {
|
||||
rsi[plane].wiener_info[tile_idx].vfilter[p] -= s;
|
||||
rsi[plane].wiener_info[tile_idx].vfilter[WIENER_WIN - p - 1] -= s;
|
||||
rsi[plane].wiener_info[tile_idx].vfilter[WIENER_HALFWIN] += 2 * s;
|
||||
err2 = try_restoration_tile(src, cpi, rsi, 1 << plane, partial_frame,
|
||||
tile_idx, 0, 0, dst_frame);
|
||||
if (err2 > err) {
|
||||
rsi[plane].wiener_info[tile_idx].vfilter[p] += s;
|
||||
rsi[plane].wiener_info[tile_idx].vfilter[WIENER_WIN - p - 1] += s;
|
||||
rsi[plane].wiener_info[tile_idx].vfilter[WIENER_HALFWIN] -= 2 * s;
|
||||
} else {
|
||||
err = err2;
|
||||
skip = 1;
|
||||
// At the highest step size continue moving in the same direction
|
||||
if (s == start_step) continue;
|
||||
}
|
||||
}
|
||||
break;
|
||||
} while (1);
|
||||
if (skip) break;
|
||||
do {
|
||||
if (rsi[plane].wiener_info[tile_idx].vfilter[p] + s <= tap_max[p]) {
|
||||
rsi[plane].wiener_info[tile_idx].vfilter[p] += s;
|
||||
rsi[plane].wiener_info[tile_idx].vfilter[WIENER_WIN - p - 1] += s;
|
||||
rsi[plane].wiener_info[tile_idx].vfilter[WIENER_HALFWIN] -= 2 * s;
|
||||
err2 = try_restoration_tile(src, cpi, rsi, 1 << plane, partial_frame,
|
||||
tile_idx, 0, 0, dst_frame);
|
||||
if (err2 > err) {
|
||||
rsi[plane].wiener_info[tile_idx].vfilter[p] -= s;
|
||||
rsi[plane].wiener_info[tile_idx].vfilter[WIENER_WIN - p - 1] -= s;
|
||||
rsi[plane].wiener_info[tile_idx].vfilter[WIENER_HALFWIN] += 2 * s;
|
||||
} else {
|
||||
err = err2;
|
||||
// At the highest step size continue moving in the same direction
|
||||
if (s == start_step) continue;
|
||||
}
|
||||
}
|
||||
break;
|
||||
} while (1);
|
||||
}
|
||||
}
|
||||
// printf("err post = %"PRId64"\n", err);
|
||||
#endif // USE_WIENER_REFINEMENT_SEARCH
|
||||
return err;
|
||||
}
|
||||
|
||||
static double search_wiener(const YV12_BUFFER_CONFIG *src, AV1_COMP *cpi,
|
||||
int partial_frame, int plane, RestorationInfo *info,
|
||||
RestorationType *type, double *best_tile_cost,
|
||||
YV12_BUFFER_CONFIG *dst_frame) {
|
||||
WienerInfo *wiener_info = info->wiener_info;
|
||||
AV1_COMMON *const cm = &cpi->common;
|
||||
RestorationInfo *rsi = cpi->rst_search;
|
||||
int64_t err;
|
||||
int bits;
|
||||
double cost_wiener, cost_norestore, cost_wiener_frame, cost_norestore_frame;
|
||||
double cost_wiener, cost_norestore;
|
||||
MACROBLOCK *x = &cpi->td.mb;
|
||||
double M[WIENER_WIN2];
|
||||
double H[WIENER_WIN2 * WIENER_WIN2];
|
||||
double vfilterd[WIENER_WIN], hfilterd[WIENER_WIN];
|
||||
const YV12_BUFFER_CONFIG *dgd = cm->frame_to_show;
|
||||
const int width = src->uv_crop_width;
|
||||
const int height = src->uv_crop_height;
|
||||
const int src_stride = src->uv_stride;
|
||||
const int dgd_stride = dgd->uv_stride;
|
||||
int width, height, src_stride, dgd_stride;
|
||||
uint8_t *dgd_buffer, *src_buffer;
|
||||
if (plane == AOM_PLANE_Y) {
|
||||
width = cm->width;
|
||||
height = cm->height;
|
||||
src_buffer = src->y_buffer;
|
||||
src_stride = src->y_stride;
|
||||
dgd_buffer = dgd->y_buffer;
|
||||
dgd_stride = dgd->y_stride;
|
||||
assert(width == dgd->y_crop_width);
|
||||
assert(height == dgd->y_crop_height);
|
||||
assert(width == src->y_crop_width);
|
||||
assert(height == src->y_crop_height);
|
||||
} else {
|
||||
width = src->uv_crop_width;
|
||||
height = src->uv_crop_height;
|
||||
src_stride = src->uv_stride;
|
||||
dgd_stride = dgd->uv_stride;
|
||||
src_buffer = plane == AOM_PLANE_U ? src->u_buffer : src->v_buffer;
|
||||
dgd_buffer = plane == AOM_PLANE_U ? dgd->u_buffer : dgd->v_buffer;
|
||||
assert(width == dgd->uv_crop_width);
|
||||
assert(height == dgd->uv_crop_height);
|
||||
}
|
||||
double score;
|
||||
int tile_idx, tile_width, tile_height, nhtiles, nvtiles;
|
||||
int h_start, h_end, v_start, v_end;
|
||||
const int ntiles =
|
||||
av1_get_rest_ntiles(width, height, cm->rst_info[1].restoration_tilesize,
|
||||
&tile_width, &tile_height, &nhtiles, &nvtiles);
|
||||
const int ntiles = av1_get_rest_ntiles(
|
||||
width, height, cm->rst_info[plane].restoration_tilesize, &tile_width,
|
||||
&tile_height, &nhtiles, &nvtiles);
|
||||
WienerInfo ref_wiener_info;
|
||||
set_default_wiener(&ref_wiener_info);
|
||||
assert(width == dgd->uv_crop_width);
|
||||
assert(height == dgd->uv_crop_height);
|
||||
|
||||
rsi[plane].frame_restoration_type = RESTORE_NONE;
|
||||
err = sse_restoration_frame(cm, src, cm->frame_to_show, (1 << plane));
|
||||
bits = 0;
|
||||
cost_norestore_frame = RDCOST_DBL(x->rdmult, x->rddiv, (bits >> 4), err);
|
||||
|
||||
rsi[plane].frame_restoration_type = RESTORE_WIENER;
|
||||
|
||||
|
|
@ -853,6 +1020,15 @@ static double search_wiener_uv(const YV12_BUFFER_CONFIG *src, AV1_COMP *cpi,
|
|||
rsi[plane].restoration_type[tile_idx] = RESTORE_NONE;
|
||||
}
|
||||
|
||||
// Construct a (WIENER_HALFWIN)-pixel border around the frame
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
if (cm->use_highbitdepth)
|
||||
extend_frame_highbd(CONVERT_TO_SHORTPTR(dgd_buffer), width, height,
|
||||
dgd_stride);
|
||||
else
|
||||
#endif
|
||||
extend_frame(dgd_buffer, width, height, dgd_stride);
|
||||
|
||||
// Compute best Wiener filters for each tile
|
||||
for (tile_idx = 0; tile_idx < ntiles; ++tile_idx) {
|
||||
av1_get_rest_tile_limits(tile_idx, 0, 0, nhtiles, nvtiles, tile_width,
|
||||
|
|
@ -860,37 +1036,23 @@ static double search_wiener_uv(const YV12_BUFFER_CONFIG *src, AV1_COMP *cpi,
|
|||
&v_start, &v_end);
|
||||
err = sse_restoration_tile(src, cm->frame_to_show, cm, h_start,
|
||||
h_end - h_start, v_start, v_end - v_start,
|
||||
1 << plane);
|
||||
(1 << plane));
|
||||
// #bits when a tile is not restored
|
||||
bits = av1_cost_bit(RESTORE_NONE_WIENER_PROB, 0);
|
||||
cost_norestore = RDCOST_DBL(x->rdmult, x->rddiv, (bits >> 4), err);
|
||||
// best_tile_cost[tile_idx] = DBL_MAX;
|
||||
best_tile_cost[tile_idx] = DBL_MAX;
|
||||
|
||||
av1_get_rest_tile_limits(tile_idx, 0, 0, nhtiles, nvtiles, tile_width,
|
||||
tile_height, width, height, WIENER_HALFWIN,
|
||||
WIENER_HALFWIN, &h_start, &h_end, &v_start,
|
||||
&v_end);
|
||||
if (plane == AOM_PLANE_U) {
|
||||
tile_height, width, height, 0, 0, &h_start, &h_end,
|
||||
&v_start, &v_end);
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
if (cm->use_highbitdepth)
|
||||
compute_stats_highbd(dgd->u_buffer, src->u_buffer, h_start, h_end,
|
||||
v_start, v_end, dgd_stride, src_stride, M, H);
|
||||
else
|
||||
if (cm->use_highbitdepth)
|
||||
compute_stats_highbd(dgd_buffer, src_buffer, h_start, h_end, v_start,
|
||||
v_end, dgd_stride, src_stride, M, H);
|
||||
else
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
compute_stats(dgd->u_buffer, src->u_buffer, h_start, h_end, v_start,
|
||||
v_end, dgd_stride, src_stride, M, H);
|
||||
} else if (plane == AOM_PLANE_V) {
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
if (cm->use_highbitdepth)
|
||||
compute_stats_highbd(dgd->v_buffer, src->v_buffer, h_start, h_end,
|
||||
v_start, v_end, dgd_stride, src_stride, M, H);
|
||||
else
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
compute_stats(dgd->v_buffer, src->v_buffer, h_start, h_end, v_start,
|
||||
v_end, dgd_stride, src_stride, M, H);
|
||||
} else {
|
||||
assert(0);
|
||||
}
|
||||
compute_stats(dgd_buffer, src_buffer, h_start, h_end, v_start, v_end,
|
||||
dgd_stride, src_stride, M, H);
|
||||
|
||||
type[tile_idx] = RESTORE_WIENER;
|
||||
|
||||
|
|
@ -910,14 +1072,14 @@ static double search_wiener_uv(const YV12_BUFFER_CONFIG *src, AV1_COMP *cpi,
|
|||
type[tile_idx] = RESTORE_NONE;
|
||||
continue;
|
||||
}
|
||||
aom_clear_system_state();
|
||||
|
||||
rsi[plane].restoration_type[tile_idx] = RESTORE_WIENER;
|
||||
err = try_restoration_tile(src, cpi, rsi, 1 << plane, partial_frame,
|
||||
tile_idx, 0, 0, dst_frame);
|
||||
err = finer_tile_search_wiener(src, cpi, rsi, 4, plane, tile_idx,
|
||||
partial_frame, dst_frame);
|
||||
bits =
|
||||
count_wiener_bits(&rsi[plane].wiener_info[tile_idx], &ref_wiener_info)
|
||||
<< AV1_PROB_COST_SHIFT;
|
||||
// bits = WIENER_FILT_BITS << AV1_PROB_COST_SHIFT;
|
||||
bits += av1_cost_bit(RESTORE_NONE_WIENER_PROB, 1);
|
||||
cost_wiener = RDCOST_DBL(x->rdmult, x->rddiv, (bits >> 4), err);
|
||||
if (cost_wiener >= cost_norestore) {
|
||||
|
|
@ -928,12 +1090,14 @@ static double search_wiener_uv(const YV12_BUFFER_CONFIG *src, AV1_COMP *cpi,
|
|||
sizeof(wiener_info[tile_idx]));
|
||||
memcpy(&ref_wiener_info, &rsi[plane].wiener_info[tile_idx],
|
||||
sizeof(ref_wiener_info));
|
||||
best_tile_cost[tile_idx] = err;
|
||||
}
|
||||
rsi[plane].restoration_type[tile_idx] = RESTORE_NONE;
|
||||
}
|
||||
// Cost for Wiener filtering
|
||||
set_default_wiener(&ref_wiener_info);
|
||||
bits = 0;
|
||||
bits = frame_level_restore_bits[rsi[plane].frame_restoration_type]
|
||||
<< AV1_PROB_COST_SHIFT;
|
||||
for (tile_idx = 0; tile_idx < ntiles; ++tile_idx) {
|
||||
bits +=
|
||||
av1_cost_bit(RESTORE_NONE_WIENER_PROB, type[tile_idx] != RESTORE_NONE);
|
||||
|
|
@ -950,198 +1114,75 @@ static double search_wiener_uv(const YV12_BUFFER_CONFIG *src, AV1_COMP *cpi,
|
|||
}
|
||||
err = try_restoration_frame(src, cpi, rsi, 1 << plane, partial_frame,
|
||||
dst_frame);
|
||||
cost_wiener_frame = RDCOST_DBL(x->rdmult, x->rddiv, (bits >> 4), err);
|
||||
|
||||
if (cost_wiener_frame < cost_norestore_frame) {
|
||||
info->frame_restoration_type = RESTORE_WIENER;
|
||||
} else {
|
||||
info->frame_restoration_type = RESTORE_NONE;
|
||||
}
|
||||
|
||||
return info->frame_restoration_type == RESTORE_WIENER ? cost_wiener_frame
|
||||
: cost_norestore_frame;
|
||||
}
|
||||
|
||||
static double search_wiener(const YV12_BUFFER_CONFIG *src, AV1_COMP *cpi,
|
||||
int partial_frame, RestorationInfo *info,
|
||||
RestorationType *type, double *best_tile_cost,
|
||||
YV12_BUFFER_CONFIG *dst_frame) {
|
||||
WienerInfo *wiener_info = info->wiener_info;
|
||||
AV1_COMMON *const cm = &cpi->common;
|
||||
RestorationInfo *rsi = cpi->rst_search;
|
||||
int64_t err;
|
||||
int bits;
|
||||
double cost_wiener, cost_norestore;
|
||||
MACROBLOCK *x = &cpi->td.mb;
|
||||
double M[WIENER_WIN2];
|
||||
double H[WIENER_WIN2 * WIENER_WIN2];
|
||||
double vfilterd[WIENER_WIN], hfilterd[WIENER_WIN];
|
||||
const YV12_BUFFER_CONFIG *dgd = cm->frame_to_show;
|
||||
const int width = cm->width;
|
||||
const int height = cm->height;
|
||||
const int src_stride = src->y_stride;
|
||||
const int dgd_stride = dgd->y_stride;
|
||||
double score;
|
||||
int tile_idx, tile_width, tile_height, nhtiles, nvtiles;
|
||||
int h_start, h_end, v_start, v_end;
|
||||
const int ntiles =
|
||||
av1_get_rest_ntiles(width, height, cm->rst_info[0].restoration_tilesize,
|
||||
&tile_width, &tile_height, &nhtiles, &nvtiles);
|
||||
WienerInfo ref_wiener_info;
|
||||
set_default_wiener(&ref_wiener_info);
|
||||
|
||||
assert(width == dgd->y_crop_width);
|
||||
assert(height == dgd->y_crop_height);
|
||||
assert(width == src->y_crop_width);
|
||||
assert(height == src->y_crop_height);
|
||||
|
||||
rsi->frame_restoration_type = RESTORE_WIENER;
|
||||
|
||||
for (tile_idx = 0; tile_idx < ntiles; ++tile_idx) {
|
||||
rsi->restoration_type[tile_idx] = RESTORE_NONE;
|
||||
}
|
||||
|
||||
// Construct a (WIENER_HALFWIN)-pixel border around the frame
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
if (cm->use_highbitdepth)
|
||||
extend_frame_highbd(CONVERT_TO_SHORTPTR(dgd->y_buffer), width, height,
|
||||
dgd_stride);
|
||||
else
|
||||
#endif
|
||||
extend_frame(dgd->y_buffer, width, height, dgd_stride);
|
||||
|
||||
// Compute best Wiener filters for each tile
|
||||
for (tile_idx = 0; tile_idx < ntiles; ++tile_idx) {
|
||||
av1_get_rest_tile_limits(tile_idx, 0, 0, nhtiles, nvtiles, tile_width,
|
||||
tile_height, width, height, 0, 0, &h_start, &h_end,
|
||||
&v_start, &v_end);
|
||||
err = sse_restoration_tile(src, cm->frame_to_show, cm, h_start,
|
||||
h_end - h_start, v_start, v_end - v_start, 1);
|
||||
// #bits when a tile is not restored
|
||||
bits = av1_cost_bit(RESTORE_NONE_WIENER_PROB, 0);
|
||||
cost_norestore = RDCOST_DBL(x->rdmult, x->rddiv, (bits >> 4), err);
|
||||
best_tile_cost[tile_idx] = DBL_MAX;
|
||||
|
||||
av1_get_rest_tile_limits(tile_idx, 0, 0, nhtiles, nvtiles, tile_width,
|
||||
tile_height, width, height, 0, 0, &h_start, &h_end,
|
||||
&v_start, &v_end);
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
if (cm->use_highbitdepth)
|
||||
compute_stats_highbd(dgd->y_buffer, src->y_buffer, h_start, h_end,
|
||||
v_start, v_end, dgd_stride, src_stride, M, H);
|
||||
else
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
compute_stats(dgd->y_buffer, src->y_buffer, h_start, h_end, v_start,
|
||||
v_end, dgd_stride, src_stride, M, H);
|
||||
|
||||
type[tile_idx] = RESTORE_WIENER;
|
||||
|
||||
if (!wiener_decompose_sep_sym(M, H, vfilterd, hfilterd)) {
|
||||
type[tile_idx] = RESTORE_NONE;
|
||||
continue;
|
||||
}
|
||||
quantize_sym_filter(vfilterd, rsi->wiener_info[tile_idx].vfilter);
|
||||
quantize_sym_filter(hfilterd, rsi->wiener_info[tile_idx].hfilter);
|
||||
|
||||
// Filter score computes the value of the function x'*A*x - x'*b for the
|
||||
// learned filter and compares it against identity filer. If there is no
|
||||
// reduction in the function, the filter is reverted back to identity
|
||||
score = compute_score(M, H, rsi->wiener_info[tile_idx].vfilter,
|
||||
rsi->wiener_info[tile_idx].hfilter);
|
||||
if (score > 0.0) {
|
||||
type[tile_idx] = RESTORE_NONE;
|
||||
continue;
|
||||
}
|
||||
|
||||
rsi->restoration_type[tile_idx] = RESTORE_WIENER;
|
||||
err = try_restoration_tile(src, cpi, rsi, 1, partial_frame, tile_idx, 0, 0,
|
||||
dst_frame);
|
||||
bits = count_wiener_bits(&rsi->wiener_info[tile_idx], &ref_wiener_info)
|
||||
<< AV1_PROB_COST_SHIFT;
|
||||
bits += av1_cost_bit(RESTORE_NONE_WIENER_PROB, 1);
|
||||
cost_wiener = RDCOST_DBL(x->rdmult, x->rddiv, (bits >> 4), err);
|
||||
if (cost_wiener >= cost_norestore) {
|
||||
type[tile_idx] = RESTORE_NONE;
|
||||
} else {
|
||||
type[tile_idx] = RESTORE_WIENER;
|
||||
memcpy(&wiener_info[tile_idx], &rsi->wiener_info[tile_idx],
|
||||
sizeof(wiener_info[tile_idx]));
|
||||
memcpy(&ref_wiener_info, &rsi->wiener_info[tile_idx],
|
||||
sizeof(ref_wiener_info));
|
||||
bits = count_wiener_bits(&wiener_info[tile_idx], &ref_wiener_info)
|
||||
<< AV1_PROB_COST_SHIFT;
|
||||
best_tile_cost[tile_idx] = err;
|
||||
}
|
||||
rsi->restoration_type[tile_idx] = RESTORE_NONE;
|
||||
}
|
||||
// Cost for Wiener filtering
|
||||
set_default_wiener(&ref_wiener_info);
|
||||
bits = frame_level_restore_bits[rsi->frame_restoration_type]
|
||||
<< AV1_PROB_COST_SHIFT;
|
||||
for (tile_idx = 0; tile_idx < ntiles; ++tile_idx) {
|
||||
bits +=
|
||||
av1_cost_bit(RESTORE_NONE_WIENER_PROB, type[tile_idx] != RESTORE_NONE);
|
||||
memcpy(&rsi->wiener_info[tile_idx], &wiener_info[tile_idx],
|
||||
sizeof(wiener_info[tile_idx]));
|
||||
if (type[tile_idx] == RESTORE_WIENER) {
|
||||
bits += count_wiener_bits(&rsi->wiener_info[tile_idx], &ref_wiener_info)
|
||||
<< AV1_PROB_COST_SHIFT;
|
||||
memcpy(&ref_wiener_info, &rsi->wiener_info[tile_idx],
|
||||
sizeof(ref_wiener_info));
|
||||
}
|
||||
rsi->restoration_type[tile_idx] = type[tile_idx];
|
||||
}
|
||||
err = try_restoration_frame(src, cpi, rsi, 1, partial_frame, dst_frame);
|
||||
cost_wiener = RDCOST_DBL(x->rdmult, x->rddiv, (bits >> 4), err);
|
||||
|
||||
return cost_wiener;
|
||||
}
|
||||
|
||||
static double search_norestore(const YV12_BUFFER_CONFIG *src, AV1_COMP *cpi,
|
||||
int partial_frame, RestorationInfo *info,
|
||||
RestorationType *type, double *best_tile_cost,
|
||||
int partial_frame, int plane,
|
||||
RestorationInfo *info, RestorationType *type,
|
||||
double *best_tile_cost,
|
||||
YV12_BUFFER_CONFIG *dst_frame) {
|
||||
double err, cost_norestore;
|
||||
int64_t err;
|
||||
double cost_norestore;
|
||||
int bits;
|
||||
MACROBLOCK *x = &cpi->td.mb;
|
||||
AV1_COMMON *const cm = &cpi->common;
|
||||
int tile_idx, tile_width, tile_height, nhtiles, nvtiles;
|
||||
int h_start, h_end, v_start, v_end;
|
||||
int width, height;
|
||||
if (plane == AOM_PLANE_Y) {
|
||||
width = cm->width;
|
||||
height = cm->height;
|
||||
} else {
|
||||
width = src->uv_crop_width;
|
||||
height = src->uv_crop_height;
|
||||
}
|
||||
const int ntiles = av1_get_rest_ntiles(
|
||||
cm->width, cm->height, cm->rst_info[0].restoration_tilesize, &tile_width,
|
||||
width, height, cm->rst_info[plane].restoration_tilesize, &tile_width,
|
||||
&tile_height, &nhtiles, &nvtiles);
|
||||
(void)info;
|
||||
(void)dst_frame;
|
||||
(void)partial_frame;
|
||||
|
||||
info->frame_restoration_type = RESTORE_NONE;
|
||||
for (tile_idx = 0; tile_idx < ntiles; ++tile_idx) {
|
||||
av1_get_rest_tile_limits(tile_idx, 0, 0, nhtiles, nvtiles, tile_width,
|
||||
tile_height, cm->width, cm->height, 0, 0, &h_start,
|
||||
&h_end, &v_start, &v_end);
|
||||
tile_height, width, height, 0, 0, &h_start, &h_end,
|
||||
&v_start, &v_end);
|
||||
err = sse_restoration_tile(src, cm->frame_to_show, cm, h_start,
|
||||
h_end - h_start, v_start, v_end - v_start, 1);
|
||||
h_end - h_start, v_start, v_end - v_start,
|
||||
1 << plane);
|
||||
type[tile_idx] = RESTORE_NONE;
|
||||
best_tile_cost[tile_idx] = err;
|
||||
}
|
||||
// RD cost associated with no restoration
|
||||
err = sse_restoration_tile(src, cm->frame_to_show, cm, 0, cm->width, 0,
|
||||
cm->height, 1);
|
||||
err = sse_restoration_frame(cm, src, cm->frame_to_show, (1 << plane));
|
||||
bits = frame_level_restore_bits[RESTORE_NONE] << AV1_PROB_COST_SHIFT;
|
||||
cost_norestore = RDCOST_DBL(x->rdmult, x->rddiv, (bits >> 4), err);
|
||||
return cost_norestore;
|
||||
}
|
||||
|
||||
static double search_switchable_restoration(
|
||||
AV1_COMP *cpi, int partial_frame, RestorationInfo *rsi,
|
||||
AV1_COMP *cpi, int partial_frame, int plane, RestorationInfo *rsi,
|
||||
double *tile_cost[RESTORE_SWITCHABLE_TYPES]) {
|
||||
AV1_COMMON *const cm = &cpi->common;
|
||||
MACROBLOCK *x = &cpi->td.mb;
|
||||
double cost_switchable = 0;
|
||||
int bits, tile_idx;
|
||||
RestorationType r;
|
||||
const int ntiles = av1_get_rest_ntiles(cm->width, cm->height,
|
||||
cm->rst_info[0].restoration_tilesize,
|
||||
NULL, NULL, NULL, NULL);
|
||||
int width, height;
|
||||
if (plane == AOM_PLANE_Y) {
|
||||
width = cm->width;
|
||||
height = cm->height;
|
||||
} else {
|
||||
width = ROUND_POWER_OF_TWO(cm->width, cm->subsampling_x);
|
||||
height = ROUND_POWER_OF_TWO(cm->height, cm->subsampling_y);
|
||||
}
|
||||
const int ntiles = av1_get_rest_ntiles(
|
||||
width, height, cm->rst_info[plane].restoration_tilesize, NULL, NULL, NULL,
|
||||
NULL);
|
||||
SgrprojInfo ref_sgrproj_info;
|
||||
set_default_sgrproj(&ref_sgrproj_info);
|
||||
WienerInfo ref_wiener_info;
|
||||
|
|
@ -1203,57 +1244,60 @@ void av1_pick_filter_restoration(const YV12_BUFFER_CONFIG *src, AV1_COMP *cpi,
|
|||
double best_cost_restore;
|
||||
RestorationType r, best_restore;
|
||||
|
||||
const int ntiles = av1_get_rest_ntiles(cm->width, cm->height,
|
||||
cm->rst_info[0].restoration_tilesize,
|
||||
NULL, NULL, NULL, NULL);
|
||||
const int ntiles_y = av1_get_rest_ntiles(cm->width, cm->height,
|
||||
cm->rst_info[0].restoration_tilesize,
|
||||
NULL, NULL, NULL, NULL);
|
||||
const int ntiles_uv = av1_get_rest_ntiles(
|
||||
ROUND_POWER_OF_TWO(cm->width, cm->subsampling_x),
|
||||
ROUND_POWER_OF_TWO(cm->height, cm->subsampling_y),
|
||||
cm->rst_info[1].restoration_tilesize, NULL, NULL, NULL, NULL);
|
||||
|
||||
// Assume ntiles_uv is never larger that ntiles_y and so the same arrays work.
|
||||
for (r = 0; r < RESTORE_SWITCHABLE_TYPES; r++) {
|
||||
tile_cost[r] = (double *)aom_malloc(sizeof(*tile_cost[0]) * ntiles);
|
||||
tile_cost[r] = (double *)aom_malloc(sizeof(*tile_cost[0]) * ntiles_y);
|
||||
restore_types[r] =
|
||||
(RestorationType *)aom_malloc(sizeof(*restore_types[0]) * ntiles);
|
||||
(RestorationType *)aom_malloc(sizeof(*restore_types[0]) * ntiles_y);
|
||||
}
|
||||
|
||||
for (r = 0; r < RESTORE_SWITCHABLE_TYPES; ++r) {
|
||||
for (int plane = AOM_PLANE_Y; plane <= AOM_PLANE_V; ++plane) {
|
||||
for (r = 0; r < RESTORE_SWITCHABLE_TYPES; ++r) {
|
||||
cost_restore[r] = DBL_MAX;
|
||||
if (force_restore_type != 0)
|
||||
if (r != RESTORE_NONE && r != force_restore_type) continue;
|
||||
cost_restore[r] =
|
||||
search_restore_fun[r](src, cpi, method == LPF_PICK_FROM_SUBIMAGE,
|
||||
plane, &cm->rst_info[plane], restore_types[r],
|
||||
tile_cost[r], &cpi->trial_frame_rst);
|
||||
}
|
||||
if (plane == AOM_PLANE_Y)
|
||||
cost_restore[RESTORE_SWITCHABLE] =
|
||||
search_switchable_restoration(cpi, method == LPF_PICK_FROM_SUBIMAGE,
|
||||
plane, &cm->rst_info[plane], tile_cost);
|
||||
else
|
||||
cost_restore[RESTORE_SWITCHABLE] = DBL_MAX;
|
||||
best_cost_restore = DBL_MAX;
|
||||
best_restore = 0;
|
||||
for (r = 0; r < RESTORE_TYPES; ++r) {
|
||||
if (force_restore_type != 0)
|
||||
if (r != RESTORE_NONE && r != force_restore_type) continue;
|
||||
if (cost_restore[r] < best_cost_restore) {
|
||||
best_restore = r;
|
||||
best_cost_restore = cost_restore[r];
|
||||
}
|
||||
}
|
||||
cm->rst_info[plane].frame_restoration_type = best_restore;
|
||||
if (force_restore_type != 0)
|
||||
if (r != RESTORE_NONE && r != force_restore_type) continue;
|
||||
cost_restore[r] = search_restore_fun[r](
|
||||
src, cpi, method == LPF_PICK_FROM_SUBIMAGE, &cm->rst_info[0],
|
||||
restore_types[r], tile_cost[r], &cpi->trial_frame_rst);
|
||||
}
|
||||
cost_restore[RESTORE_SWITCHABLE] = search_switchable_restoration(
|
||||
cpi, method == LPF_PICK_FROM_SUBIMAGE, &cm->rst_info[0], tile_cost);
|
||||
|
||||
best_cost_restore = DBL_MAX;
|
||||
best_restore = 0;
|
||||
for (r = 0; r < RESTORE_TYPES; ++r) {
|
||||
if (force_restore_type != 0)
|
||||
if (r != RESTORE_NONE && r != force_restore_type) continue;
|
||||
if (cost_restore[r] < best_cost_restore) {
|
||||
best_restore = r;
|
||||
best_cost_restore = cost_restore[r];
|
||||
assert(best_restore == force_restore_type ||
|
||||
best_restore == RESTORE_NONE);
|
||||
if (best_restore != RESTORE_SWITCHABLE) {
|
||||
const int nt = (plane == AOM_PLANE_Y ? ntiles_y : ntiles_uv);
|
||||
memcpy(cm->rst_info[plane].restoration_type, restore_types[best_restore],
|
||||
nt * sizeof(restore_types[best_restore][0]));
|
||||
}
|
||||
}
|
||||
cm->rst_info[0].frame_restoration_type = best_restore;
|
||||
if (force_restore_type != 0)
|
||||
assert(best_restore == force_restore_type || best_restore == RESTORE_NONE);
|
||||
if (best_restore != RESTORE_SWITCHABLE) {
|
||||
memcpy(cm->rst_info[0].restoration_type, restore_types[best_restore],
|
||||
ntiles * sizeof(restore_types[best_restore][0]));
|
||||
}
|
||||
|
||||
// Color components
|
||||
search_wiener_uv(src, cpi, method == LPF_PICK_FROM_SUBIMAGE, AOM_PLANE_U,
|
||||
&cm->rst_info[AOM_PLANE_U],
|
||||
cm->rst_info[AOM_PLANE_U].restoration_type,
|
||||
&cpi->trial_frame_rst);
|
||||
search_wiener_uv(src, cpi, method == LPF_PICK_FROM_SUBIMAGE, AOM_PLANE_V,
|
||||
&cm->rst_info[AOM_PLANE_V],
|
||||
cm->rst_info[AOM_PLANE_V].restoration_type,
|
||||
&cpi->trial_frame_rst);
|
||||
/*
|
||||
printf("Frame %d/%d restore types: %d %d %d\n",
|
||||
cm->current_video_frame, cm->show_frame,
|
||||
cm->rst_info[0].frame_restoration_type,
|
||||
printf("Frame %d/%d restore types: %d %d %d\n", cm->current_video_frame,
|
||||
cm->show_frame, cm->rst_info[0].frame_restoration_type,
|
||||
cm->rst_info[1].frame_restoration_type,
|
||||
cm->rst_info[2].frame_restoration_type);
|
||||
printf("Frame %d/%d frame_restore_type %d : %f %f %f %f\n",
|
||||
|
|
|
|||
52
third_party/aom/av1/encoder/pvq_encoder.c
vendored
52
third_party/aom/av1/encoder/pvq_encoder.c
vendored
|
|
@ -247,23 +247,23 @@ static double od_pvq_rate(int qg, int icgr, int theta, int ts,
|
|||
aom_writer w;
|
||||
od_pvq_codeword_ctx cd;
|
||||
int tell;
|
||||
#if CONFIG_DAALA_EC
|
||||
#if !CONFIG_ANS
|
||||
od_ec_enc_init(&w.ec, 1000);
|
||||
#else
|
||||
# error "CONFIG_PVQ currently requires CONFIG_DAALA_EC."
|
||||
# error "CONFIG_PVQ currently requires !CONFIG_ANS."
|
||||
#endif
|
||||
OD_COPY(&cd, &adapt->pvq.pvq_codeword_ctx, 1);
|
||||
#if CONFIG_DAALA_EC
|
||||
#if !CONFIG_ANS
|
||||
tell = od_ec_enc_tell_frac(&w.ec);
|
||||
#else
|
||||
# error "CONFIG_PVQ currently requires CONFIG_DAALA_EC."
|
||||
# error "CONFIG_PVQ currently requires !CONFIG_ANS."
|
||||
#endif
|
||||
aom_encode_pvq_codeword(&w, &cd, y0, n - (theta != -1), k);
|
||||
#if CONFIG_DAALA_EC
|
||||
#if !CONFIG_ANS
|
||||
rate = (od_ec_enc_tell_frac(&w.ec)-tell)/8.;
|
||||
od_ec_enc_clear(&w.ec);
|
||||
#else
|
||||
# error "CONFIG_PVQ currently requires CONFIG_DAALA_EC."
|
||||
# error "CONFIG_PVQ currently requires !CONFIG_ANS."
|
||||
#endif
|
||||
}
|
||||
if (qg > 0 && theta >= 0) {
|
||||
|
|
@ -847,22 +847,22 @@ PVQ_SKIP_TYPE od_pvq_encode(daala_enc_ctx *enc,
|
|||
int tell2;
|
||||
od_rollback_buffer dc_buf;
|
||||
|
||||
dc_rate = -OD_LOG2((double)(skip_cdf[3] - skip_cdf[2])/
|
||||
(double)(skip_cdf[2] - skip_cdf[1]));
|
||||
dc_rate = -OD_LOG2((double)(OD_ICDF(skip_cdf[3]) - OD_ICDF(skip_cdf[2]))/
|
||||
(double)(OD_ICDF(skip_cdf[2]) - OD_ICDF(skip_cdf[1])));
|
||||
dc_rate += 1;
|
||||
|
||||
#if CONFIG_DAALA_EC
|
||||
#if !CONFIG_ANS
|
||||
tell2 = od_ec_enc_tell_frac(&enc->w.ec);
|
||||
#else
|
||||
#error "CONFIG_PVQ currently requires CONFIG_DAALA_EC."
|
||||
#error "CONFIG_PVQ currently requires !CONFIG_ANS."
|
||||
#endif
|
||||
od_encode_checkpoint(enc, &dc_buf);
|
||||
generic_encode(&enc->w, &enc->state.adapt->model_dc[pli],
|
||||
n - 1, &enc->state.adapt->ex_dc[pli][bs][0], 2);
|
||||
#if CONFIG_DAALA_EC
|
||||
#if !CONFIG_ANS
|
||||
tell2 = od_ec_enc_tell_frac(&enc->w.ec) - tell2;
|
||||
#else
|
||||
#error "CONFIG_PVQ currently requires CONFIG_DAALA_EC."
|
||||
#error "CONFIG_PVQ currently requires !CONFIG_ANS."
|
||||
#endif
|
||||
dc_rate += tell2/8.0;
|
||||
od_encode_rollback(enc, &dc_buf);
|
||||
|
|
@ -871,10 +871,10 @@ PVQ_SKIP_TYPE od_pvq_encode(daala_enc_ctx *enc,
|
|||
enc->pvq_norm_lambda);
|
||||
}
|
||||
}
|
||||
#if CONFIG_DAALA_EC
|
||||
#if !CONFIG_ANS
|
||||
tell = od_ec_enc_tell_frac(&enc->w.ec);
|
||||
#else
|
||||
#error "CONFIG_PVQ currently requires CONFIG_DAALA_EC."
|
||||
#error "CONFIG_PVQ currently requires !CONFIG_ANS."
|
||||
#endif
|
||||
/* Code as if we're not skipping. */
|
||||
aom_write_symbol(&enc->w, 2 + (out[0] != 0), skip_cdf, 4);
|
||||
|
|
@ -921,22 +921,22 @@ PVQ_SKIP_TYPE od_pvq_encode(daala_enc_ctx *enc,
|
|||
}
|
||||
if (encode_flip) cfl_encoded = 1;
|
||||
}
|
||||
#if CONFIG_DAALA_EC
|
||||
#if !CONFIG_ANS
|
||||
tell = od_ec_enc_tell_frac(&enc->w.ec) - tell;
|
||||
#else
|
||||
#error "CONFIG_PVQ currently requires CONFIG_DAALA_EC."
|
||||
#error "CONFIG_PVQ currently requires !CONFIG_ANS."
|
||||
#endif
|
||||
/* Account for the rate of skipping the AC, based on the same DC decision
|
||||
we made when trying to not skip AC. */
|
||||
{
|
||||
double skip_rate;
|
||||
if (out[0] != 0) {
|
||||
skip_rate = -OD_LOG2((skip_cdf[1] - skip_cdf[0])/
|
||||
(double)skip_cdf[3]);
|
||||
skip_rate = -OD_LOG2((OD_ICDF(skip_cdf[1]) - OD_ICDF(skip_cdf[0]))/
|
||||
(double)OD_ICDF(skip_cdf[3]));
|
||||
}
|
||||
else {
|
||||
skip_rate = -OD_LOG2(skip_cdf[0]/
|
||||
(double)skip_cdf[3]);
|
||||
skip_rate = -OD_LOG2(OD_ICDF(skip_cdf[0])/
|
||||
(double)OD_ICDF(skip_cdf[3]));
|
||||
}
|
||||
tell -= (int)floor(.5+8*skip_rate);
|
||||
}
|
||||
|
|
@ -951,22 +951,22 @@ PVQ_SKIP_TYPE od_pvq_encode(daala_enc_ctx *enc,
|
|||
int tell2;
|
||||
od_rollback_buffer dc_buf;
|
||||
|
||||
dc_rate = -OD_LOG2((double)(skip_cdf[1] - skip_cdf[0])/
|
||||
(double)skip_cdf[0]);
|
||||
dc_rate = -OD_LOG2((double)(OD_ICDF(skip_cdf[1]) - OD_ICDF(skip_cdf[0]))/
|
||||
(double)OD_ICDF(skip_cdf[0]));
|
||||
dc_rate += 1;
|
||||
|
||||
#if CONFIG_DAALA_EC
|
||||
#if !CONFIG_ANS
|
||||
tell2 = od_ec_enc_tell_frac(&enc->w.ec);
|
||||
#else
|
||||
#error "CONFIG_PVQ currently requires CONFIG_DAALA_EC."
|
||||
#error "CONFIG_PVQ currently requires !CONFIG_ANS."
|
||||
#endif
|
||||
od_encode_checkpoint(enc, &dc_buf);
|
||||
generic_encode(&enc->w, &enc->state.adapt->model_dc[pli],
|
||||
n - 1, &enc->state.adapt->ex_dc[pli][bs][0], 2);
|
||||
#if CONFIG_DAALA_EC
|
||||
#if !CONFIG_ANS
|
||||
tell2 = od_ec_enc_tell_frac(&enc->w.ec) - tell2;
|
||||
#else
|
||||
#error "CONFIG_PVQ currently requires CONFIG_DAALA_EC."
|
||||
#error "CONFIG_PVQ currently requires !CONFIG_ANS."
|
||||
#endif
|
||||
dc_rate += tell2/8.0;
|
||||
od_encode_rollback(enc, &dc_buf);
|
||||
|
|
|
|||
335
third_party/aom/av1/encoder/ransac.c
vendored
335
third_party/aom/av1/encoder/ransac.c
vendored
|
|
@ -8,7 +8,6 @@
|
|||
* Media Patent License 1.0 was not distributed with this source code in the
|
||||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
#define _POSIX_C_SOURCE 200112L // rand_r()
|
||||
#include <memory.h>
|
||||
#include <math.h>
|
||||
#include <time.h>
|
||||
|
|
@ -17,6 +16,7 @@
|
|||
#include <assert.h>
|
||||
|
||||
#include "av1/encoder/ransac.h"
|
||||
#include "av1/encoder/mathutils.h"
|
||||
|
||||
#define MAX_MINPTS 4
|
||||
#define MAX_DEGENERATE_ITER 10
|
||||
|
|
@ -133,309 +133,6 @@ static void project_points_double_homography(double *mat, double *points,
|
|||
}
|
||||
}
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
// svdcmp
|
||||
// Adopted from Numerical Recipes in C
|
||||
|
||||
static const double TINY_NEAR_ZERO = 1.0E-12;
|
||||
|
||||
static INLINE double sign(double a, double b) {
|
||||
return ((b) >= 0 ? fabs(a) : -fabs(a));
|
||||
}
|
||||
|
||||
static INLINE double pythag(double a, double b) {
|
||||
double ct;
|
||||
const double absa = fabs(a);
|
||||
const double absb = fabs(b);
|
||||
|
||||
if (absa > absb) {
|
||||
ct = absb / absa;
|
||||
return absa * sqrt(1.0 + ct * ct);
|
||||
} else {
|
||||
ct = absa / absb;
|
||||
return (absb == 0) ? 0 : absb * sqrt(1.0 + ct * ct);
|
||||
}
|
||||
}
|
||||
|
||||
static void multiply_mat(const double *m1, const double *m2, double *res,
|
||||
const int m1_rows, const int inner_dim,
|
||||
const int m2_cols) {
|
||||
double sum;
|
||||
|
||||
int row, col, inner;
|
||||
for (row = 0; row < m1_rows; ++row) {
|
||||
for (col = 0; col < m2_cols; ++col) {
|
||||
sum = 0;
|
||||
for (inner = 0; inner < inner_dim; ++inner)
|
||||
sum += m1[row * inner_dim + inner] * m2[inner * m2_cols + col];
|
||||
*(res++) = sum;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static int svdcmp(double **u, int m, int n, double w[], double **v) {
|
||||
const int max_its = 30;
|
||||
int flag, i, its, j, jj, k, l, nm;
|
||||
double anorm, c, f, g, h, s, scale, x, y, z;
|
||||
double *rv1 = (double *)aom_malloc(sizeof(*rv1) * (n + 1));
|
||||
g = scale = anorm = 0.0;
|
||||
for (i = 0; i < n; i++) {
|
||||
l = i + 1;
|
||||
rv1[i] = scale * g;
|
||||
g = s = scale = 0.0;
|
||||
if (i < m) {
|
||||
for (k = i; k < m; k++) scale += fabs(u[k][i]);
|
||||
if (scale != 0.) {
|
||||
for (k = i; k < m; k++) {
|
||||
u[k][i] /= scale;
|
||||
s += u[k][i] * u[k][i];
|
||||
}
|
||||
f = u[i][i];
|
||||
g = -sign(sqrt(s), f);
|
||||
h = f * g - s;
|
||||
u[i][i] = f - g;
|
||||
for (j = l; j < n; j++) {
|
||||
for (s = 0.0, k = i; k < m; k++) s += u[k][i] * u[k][j];
|
||||
f = s / h;
|
||||
for (k = i; k < m; k++) u[k][j] += f * u[k][i];
|
||||
}
|
||||
for (k = i; k < m; k++) u[k][i] *= scale;
|
||||
}
|
||||
}
|
||||
w[i] = scale * g;
|
||||
g = s = scale = 0.0;
|
||||
if (i < m && i != n - 1) {
|
||||
for (k = l; k < n; k++) scale += fabs(u[i][k]);
|
||||
if (scale != 0.) {
|
||||
for (k = l; k < n; k++) {
|
||||
u[i][k] /= scale;
|
||||
s += u[i][k] * u[i][k];
|
||||
}
|
||||
f = u[i][l];
|
||||
g = -sign(sqrt(s), f);
|
||||
h = f * g - s;
|
||||
u[i][l] = f - g;
|
||||
for (k = l; k < n; k++) rv1[k] = u[i][k] / h;
|
||||
for (j = l; j < m; j++) {
|
||||
for (s = 0.0, k = l; k < n; k++) s += u[j][k] * u[i][k];
|
||||
for (k = l; k < n; k++) u[j][k] += s * rv1[k];
|
||||
}
|
||||
for (k = l; k < n; k++) u[i][k] *= scale;
|
||||
}
|
||||
}
|
||||
anorm = fmax(anorm, (fabs(w[i]) + fabs(rv1[i])));
|
||||
}
|
||||
|
||||
for (i = n - 1; i >= 0; i--) {
|
||||
if (i < n - 1) {
|
||||
if (g != 0.) {
|
||||
for (j = l; j < n; j++) v[j][i] = (u[i][j] / u[i][l]) / g;
|
||||
for (j = l; j < n; j++) {
|
||||
for (s = 0.0, k = l; k < n; k++) s += u[i][k] * v[k][j];
|
||||
for (k = l; k < n; k++) v[k][j] += s * v[k][i];
|
||||
}
|
||||
}
|
||||
for (j = l; j < n; j++) v[i][j] = v[j][i] = 0.0;
|
||||
}
|
||||
v[i][i] = 1.0;
|
||||
g = rv1[i];
|
||||
l = i;
|
||||
}
|
||||
for (i = AOMMIN(m, n) - 1; i >= 0; i--) {
|
||||
l = i + 1;
|
||||
g = w[i];
|
||||
for (j = l; j < n; j++) u[i][j] = 0.0;
|
||||
if (g != 0.) {
|
||||
g = 1.0 / g;
|
||||
for (j = l; j < n; j++) {
|
||||
for (s = 0.0, k = l; k < m; k++) s += u[k][i] * u[k][j];
|
||||
f = (s / u[i][i]) * g;
|
||||
for (k = i; k < m; k++) u[k][j] += f * u[k][i];
|
||||
}
|
||||
for (j = i; j < m; j++) u[j][i] *= g;
|
||||
} else {
|
||||
for (j = i; j < m; j++) u[j][i] = 0.0;
|
||||
}
|
||||
++u[i][i];
|
||||
}
|
||||
for (k = n - 1; k >= 0; k--) {
|
||||
for (its = 0; its < max_its; its++) {
|
||||
flag = 1;
|
||||
for (l = k; l >= 0; l--) {
|
||||
nm = l - 1;
|
||||
if ((double)(fabs(rv1[l]) + anorm) == anorm || nm < 0) {
|
||||
flag = 0;
|
||||
break;
|
||||
}
|
||||
if ((double)(fabs(w[nm]) + anorm) == anorm) break;
|
||||
}
|
||||
if (flag) {
|
||||
c = 0.0;
|
||||
s = 1.0;
|
||||
for (i = l; i <= k; i++) {
|
||||
f = s * rv1[i];
|
||||
rv1[i] = c * rv1[i];
|
||||
if ((double)(fabs(f) + anorm) == anorm) break;
|
||||
g = w[i];
|
||||
h = pythag(f, g);
|
||||
w[i] = h;
|
||||
h = 1.0 / h;
|
||||
c = g * h;
|
||||
s = -f * h;
|
||||
for (j = 0; j < m; j++) {
|
||||
y = u[j][nm];
|
||||
z = u[j][i];
|
||||
u[j][nm] = y * c + z * s;
|
||||
u[j][i] = z * c - y * s;
|
||||
}
|
||||
}
|
||||
}
|
||||
z = w[k];
|
||||
if (l == k) {
|
||||
if (z < 0.0) {
|
||||
w[k] = -z;
|
||||
for (j = 0; j < n; j++) v[j][k] = -v[j][k];
|
||||
}
|
||||
break;
|
||||
}
|
||||
if (its == max_its - 1) {
|
||||
aom_free(rv1);
|
||||
return 1;
|
||||
}
|
||||
assert(k > 0);
|
||||
x = w[l];
|
||||
nm = k - 1;
|
||||
y = w[nm];
|
||||
g = rv1[nm];
|
||||
h = rv1[k];
|
||||
f = ((y - z) * (y + z) + (g - h) * (g + h)) / (2.0 * h * y);
|
||||
g = pythag(f, 1.0);
|
||||
f = ((x - z) * (x + z) + h * ((y / (f + sign(g, f))) - h)) / x;
|
||||
c = s = 1.0;
|
||||
for (j = l; j <= nm; j++) {
|
||||
i = j + 1;
|
||||
g = rv1[i];
|
||||
y = w[i];
|
||||
h = s * g;
|
||||
g = c * g;
|
||||
z = pythag(f, h);
|
||||
rv1[j] = z;
|
||||
c = f / z;
|
||||
s = h / z;
|
||||
f = x * c + g * s;
|
||||
g = g * c - x * s;
|
||||
h = y * s;
|
||||
y *= c;
|
||||
for (jj = 0; jj < n; jj++) {
|
||||
x = v[jj][j];
|
||||
z = v[jj][i];
|
||||
v[jj][j] = x * c + z * s;
|
||||
v[jj][i] = z * c - x * s;
|
||||
}
|
||||
z = pythag(f, h);
|
||||
w[j] = z;
|
||||
if (z != 0.) {
|
||||
z = 1.0 / z;
|
||||
c = f * z;
|
||||
s = h * z;
|
||||
}
|
||||
f = c * g + s * y;
|
||||
x = c * y - s * g;
|
||||
for (jj = 0; jj < m; jj++) {
|
||||
y = u[jj][j];
|
||||
z = u[jj][i];
|
||||
u[jj][j] = y * c + z * s;
|
||||
u[jj][i] = z * c - y * s;
|
||||
}
|
||||
}
|
||||
rv1[l] = 0.0;
|
||||
rv1[k] = f;
|
||||
w[k] = x;
|
||||
}
|
||||
}
|
||||
aom_free(rv1);
|
||||
return 0;
|
||||
}
|
||||
|
||||
static int SVD(double *U, double *W, double *V, double *matx, int M, int N) {
|
||||
// Assumes allocation for U is MxN
|
||||
double **nrU = (double **)aom_malloc((M) * sizeof(*nrU));
|
||||
double **nrV = (double **)aom_malloc((N) * sizeof(*nrV));
|
||||
int problem, i;
|
||||
|
||||
problem = !(nrU && nrV);
|
||||
if (!problem) {
|
||||
for (i = 0; i < M; i++) {
|
||||
nrU[i] = &U[i * N];
|
||||
}
|
||||
for (i = 0; i < N; i++) {
|
||||
nrV[i] = &V[i * N];
|
||||
}
|
||||
} else {
|
||||
if (nrU) aom_free(nrU);
|
||||
if (nrV) aom_free(nrV);
|
||||
return 1;
|
||||
}
|
||||
|
||||
/* copy from given matx into nrU */
|
||||
for (i = 0; i < M; i++) {
|
||||
memcpy(&(nrU[i][0]), matx + N * i, N * sizeof(*matx));
|
||||
}
|
||||
|
||||
/* HERE IT IS: do SVD */
|
||||
if (svdcmp(nrU, M, N, W, nrV)) {
|
||||
aom_free(nrU);
|
||||
aom_free(nrV);
|
||||
return 1;
|
||||
}
|
||||
|
||||
/* aom_free Numerical Recipes arrays */
|
||||
aom_free(nrU);
|
||||
aom_free(nrV);
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
int pseudo_inverse(double *inv, double *matx, const int M, const int N) {
|
||||
double ans;
|
||||
int i, j, k;
|
||||
double *const U = (double *)aom_malloc(M * N * sizeof(*matx));
|
||||
double *const W = (double *)aom_malloc(N * sizeof(*matx));
|
||||
double *const V = (double *)aom_malloc(N * N * sizeof(*matx));
|
||||
|
||||
if (!(U && W && V)) {
|
||||
return 1;
|
||||
}
|
||||
if (SVD(U, W, V, matx, M, N)) {
|
||||
aom_free(U);
|
||||
aom_free(W);
|
||||
aom_free(V);
|
||||
return 1;
|
||||
}
|
||||
for (i = 0; i < N; i++) {
|
||||
if (fabs(W[i]) < TINY_NEAR_ZERO) {
|
||||
aom_free(U);
|
||||
aom_free(W);
|
||||
aom_free(V);
|
||||
return 1;
|
||||
}
|
||||
}
|
||||
|
||||
for (i = 0; i < N; i++) {
|
||||
for (j = 0; j < M; j++) {
|
||||
ans = 0;
|
||||
for (k = 0; k < N; k++) {
|
||||
ans += V[k + N * i] * U[k + N * j] / W[k];
|
||||
}
|
||||
inv[j + M * i] = ans;
|
||||
}
|
||||
}
|
||||
aom_free(U);
|
||||
aom_free(W);
|
||||
aom_free(V);
|
||||
return 0;
|
||||
}
|
||||
|
||||
static void normalize_homography(double *pts, int n, double *T) {
|
||||
double *p = pts;
|
||||
double mean[2] = { 0, 0 };
|
||||
|
|
@ -597,7 +294,7 @@ static int find_translation(int np, double *pts1, double *pts2, double *mat) {
|
|||
|
||||
static int find_rotzoom(int np, double *pts1, double *pts2, double *mat) {
|
||||
const int np2 = np * 2;
|
||||
double *a = (double *)aom_malloc(sizeof(*a) * np2 * 9);
|
||||
double *a = (double *)aom_malloc(sizeof(*a) * (np2 * 5 + 20));
|
||||
double *b = a + np2 * 4;
|
||||
double *temp = b + np2;
|
||||
int i;
|
||||
|
|
@ -625,11 +322,10 @@ static int find_rotzoom(int np, double *pts1, double *pts2, double *mat) {
|
|||
b[2 * i] = dx;
|
||||
b[2 * i + 1] = dy;
|
||||
}
|
||||
if (pseudo_inverse(temp, a, np2, 4)) {
|
||||
if (!least_squares(4, a, np2, 4, b, temp, mat)) {
|
||||
aom_free(a);
|
||||
return 1;
|
||||
}
|
||||
multiply_mat(temp, b, mat, 4, np2, 1);
|
||||
denormalize_rotzoom_reorder(mat, T1, T2);
|
||||
aom_free(a);
|
||||
return 0;
|
||||
|
|
@ -637,7 +333,7 @@ static int find_rotzoom(int np, double *pts1, double *pts2, double *mat) {
|
|||
|
||||
static int find_affine(int np, double *pts1, double *pts2, double *mat) {
|
||||
const int np2 = np * 2;
|
||||
double *a = (double *)aom_malloc(sizeof(*a) * np2 * 13);
|
||||
double *a = (double *)aom_malloc(sizeof(*a) * (np2 * 7 + 42));
|
||||
double *b = a + np2 * 6;
|
||||
double *temp = b + np2;
|
||||
int i;
|
||||
|
|
@ -669,11 +365,10 @@ static int find_affine(int np, double *pts1, double *pts2, double *mat) {
|
|||
b[2 * i] = dx;
|
||||
b[2 * i + 1] = dy;
|
||||
}
|
||||
if (pseudo_inverse(temp, a, np2, 6)) {
|
||||
if (!least_squares(6, a, np2, 6, b, temp, mat)) {
|
||||
aom_free(a);
|
||||
return 1;
|
||||
}
|
||||
multiply_mat(temp, b, mat, 6, np2, 1);
|
||||
denormalize_affine_reorder(mat, T1, T2);
|
||||
aom_free(a);
|
||||
return 0;
|
||||
|
|
@ -890,16 +585,22 @@ static int find_homography(int np, double *pts1, double *pts2, double *mat) {
|
|||
return 0;
|
||||
}
|
||||
|
||||
// Generate a random number in the range [0, 32768).
|
||||
static unsigned int lcg_rand16(unsigned int *state) {
|
||||
*state = (unsigned int)(*state * 1103515245ULL + 12345);
|
||||
return *state / 65536 % 32768;
|
||||
}
|
||||
|
||||
static int get_rand_indices(int npoints, int minpts, int *indices,
|
||||
unsigned int *seed) {
|
||||
int i, j;
|
||||
int ptr = rand_r(seed) % npoints;
|
||||
int ptr = lcg_rand16(seed) % npoints;
|
||||
if (minpts > npoints) return 0;
|
||||
indices[0] = ptr;
|
||||
ptr = (ptr == npoints - 1 ? 0 : ptr + 1);
|
||||
i = 1;
|
||||
while (i < minpts) {
|
||||
int index = rand_r(seed) % npoints;
|
||||
int index = lcg_rand16(seed) % npoints;
|
||||
while (index) {
|
||||
ptr = (ptr == npoints - 1 ? 0 : ptr + 1);
|
||||
for (j = 0; j < i; ++j) {
|
||||
|
|
@ -986,6 +687,9 @@ static int ransac(const int *matched_points, int npoints,
|
|||
|
||||
double *cnp1, *cnp2;
|
||||
|
||||
for (i = 0; i < num_desired_motions; ++i) {
|
||||
num_inliers_by_motion[i] = 0;
|
||||
}
|
||||
if (npoints < minpts * MINPTS_MULTIPLIER || npoints == 0) {
|
||||
return 1;
|
||||
}
|
||||
|
|
@ -1072,7 +776,7 @@ static int ransac(const int *matched_points, int npoints,
|
|||
if (current_motion.num_inliers >= worst_kept_motion->num_inliers &&
|
||||
current_motion.num_inliers > 1) {
|
||||
int temp;
|
||||
double fracinliers, pNoOutliers, mean_distance;
|
||||
double fracinliers, pNoOutliers, mean_distance, dtemp;
|
||||
mean_distance = sum_distance / ((double)current_motion.num_inliers);
|
||||
current_motion.variance =
|
||||
sum_distance_squared / ((double)current_motion.num_inliers - 1.0) -
|
||||
|
|
@ -1092,7 +796,10 @@ static int ransac(const int *matched_points, int npoints,
|
|||
pNoOutliers = 1 - pow(fracinliers, minpts);
|
||||
pNoOutliers = fmax(EPS, pNoOutliers);
|
||||
pNoOutliers = fmin(1 - EPS, pNoOutliers);
|
||||
temp = (int)(log(1.0 - PROBABILITY_REQUIRED) / log(pNoOutliers));
|
||||
dtemp = log(1.0 - PROBABILITY_REQUIRED) / log(pNoOutliers);
|
||||
temp = (dtemp > (double)INT32_MAX)
|
||||
? INT32_MAX
|
||||
: dtemp < (double)INT32_MIN ? INT32_MIN : (int)dtemp;
|
||||
|
||||
if (temp > 0 && temp < N) {
|
||||
N = AOMMAX(temp, MIN_TRIALS);
|
||||
|
|
|
|||
120
third_party/aom/av1/encoder/ratectrl.c
vendored
120
third_party/aom/av1/encoder/ratectrl.c
vendored
|
|
@ -93,6 +93,11 @@ static int gf_low = 400;
|
|||
static int kf_high = 5000;
|
||||
static int kf_low = 400;
|
||||
|
||||
double av1_resize_rate_factor(const AV1_COMP *cpi) {
|
||||
return (double)(cpi->resize_scale_den * cpi->resize_scale_den) /
|
||||
(cpi->resize_scale_num * cpi->resize_scale_num);
|
||||
}
|
||||
|
||||
// Functions to compute the active minq lookup table entries based on a
|
||||
// formulaic approach to facilitate easier adjustment of the Q tables.
|
||||
// The formulae were derived from computing a 3rd order polynomial best
|
||||
|
|
@ -384,7 +389,7 @@ static double get_rate_correction_factor(const AV1_COMP *cpi) {
|
|||
else
|
||||
rcf = rc->rate_correction_factors[INTER_NORMAL];
|
||||
}
|
||||
rcf *= rcf_mult[rc->frame_size_selector];
|
||||
rcf *= av1_resize_rate_factor(cpi);
|
||||
return fclamp(rcf, MIN_BPB_FACTOR, MAX_BPB_FACTOR);
|
||||
}
|
||||
|
||||
|
|
@ -392,7 +397,7 @@ static void set_rate_correction_factor(AV1_COMP *cpi, double factor) {
|
|||
RATE_CONTROL *const rc = &cpi->rc;
|
||||
|
||||
// Normalize RCF to account for the size-dependent scaling factor.
|
||||
factor /= rcf_mult[cpi->rc.frame_size_selector];
|
||||
factor /= av1_resize_rate_factor(cpi);
|
||||
|
||||
factor = fclamp(factor, MIN_BPB_FACTOR, MAX_BPB_FACTOR);
|
||||
|
||||
|
|
@ -1076,7 +1081,7 @@ static int rc_pick_q_and_bounds_two_pass(const AV1_COMP *cpi, int *bottom_index,
|
|||
}
|
||||
|
||||
// Modify active_best_quality for downscaled normal frames.
|
||||
if (rc->frame_size_selector != UNSCALED && !frame_is_kf_gf_arf(cpi)) {
|
||||
if (!av1_resize_unscaled(cpi) && !frame_is_kf_gf_arf(cpi)) {
|
||||
int qdelta = av1_compute_qdelta_by_rate(
|
||||
rc, cm->frame_type, active_best_quality, 2.0, cm->bit_depth);
|
||||
active_best_quality =
|
||||
|
|
@ -1158,11 +1163,10 @@ void av1_rc_set_frame_target(AV1_COMP *cpi, int target) {
|
|||
|
||||
rc->this_frame_target = target;
|
||||
|
||||
// Modify frame size target when down-scaling.
|
||||
if (cpi->oxcf.resize_mode == RESIZE_DYNAMIC &&
|
||||
rc->frame_size_selector != UNSCALED)
|
||||
rc->this_frame_target = (int)(rc->this_frame_target *
|
||||
rate_thresh_mult[rc->frame_size_selector]);
|
||||
// Modify frame size target when down-scaled.
|
||||
if (cpi->oxcf.resize_mode == RESIZE_DYNAMIC && !av1_resize_unscaled(cpi))
|
||||
rc->this_frame_target =
|
||||
(int)(rc->this_frame_target * av1_resize_rate_factor(cpi));
|
||||
|
||||
// Target rate per SB64 (including partial SB64s.
|
||||
rc->sb64_target_rate = (int)((int64_t)rc->this_frame_target * 64 * 64) /
|
||||
|
|
@ -1225,7 +1229,6 @@ static void update_golden_frame_stats(AV1_COMP *cpi) {
|
|||
|
||||
void av1_rc_postencode_update(AV1_COMP *cpi, uint64_t bytes_used) {
|
||||
const AV1_COMMON *const cm = &cpi->common;
|
||||
const AV1EncoderConfig *const oxcf = &cpi->oxcf;
|
||||
RATE_CONTROL *const rc = &cpi->rc;
|
||||
const int qindex = cm->base_qindex;
|
||||
|
||||
|
|
@ -1317,13 +1320,6 @@ void av1_rc_postencode_update(AV1_COMP *cpi, uint64_t bytes_used) {
|
|||
rc->frames_since_key++;
|
||||
rc->frames_to_key--;
|
||||
}
|
||||
|
||||
// Trigger the resizing of the next frame if it is scaled.
|
||||
if (oxcf->pass != 0) {
|
||||
cpi->resize_pending =
|
||||
rc->next_frame_size_selector != rc->frame_size_selector;
|
||||
rc->frame_size_selector = rc->next_frame_size_selector;
|
||||
}
|
||||
}
|
||||
|
||||
void av1_rc_postencode_update_drop_frame(AV1_COMP *cpi) {
|
||||
|
|
@ -1501,10 +1497,7 @@ void av1_rc_get_one_pass_cbr_params(AV1_COMP *cpi) {
|
|||
target = calc_pframe_target_size_one_pass_cbr(cpi);
|
||||
|
||||
av1_rc_set_frame_target(cpi, target);
|
||||
if (cpi->oxcf.resize_mode == RESIZE_DYNAMIC)
|
||||
cpi->resize_pending = av1_resize_one_pass_cbr(cpi);
|
||||
else
|
||||
cpi->resize_pending = 0;
|
||||
// TODO(afergs): Decide whether to scale up, down, or not at all
|
||||
}
|
||||
|
||||
int av1_compute_qdelta(const RATE_CONTROL *rc, double qstart, double qtarget,
|
||||
|
|
@ -1670,90 +1663,3 @@ void av1_set_target_rate(AV1_COMP *cpi) {
|
|||
vbr_rate_correction(cpi, &target_rate);
|
||||
av1_rc_set_frame_target(cpi, target_rate);
|
||||
}
|
||||
|
||||
// Check if we should resize, based on average QP from past x frames.
|
||||
// Only allow for resize at most one scale down for now, scaling factor is 2.
|
||||
int av1_resize_one_pass_cbr(AV1_COMP *cpi) {
|
||||
const AV1_COMMON *const cm = &cpi->common;
|
||||
RATE_CONTROL *const rc = &cpi->rc;
|
||||
int resize_now = 0;
|
||||
cpi->resize_scale_num = 1;
|
||||
cpi->resize_scale_den = 1;
|
||||
// Don't resize on key frame; reset the counters on key frame.
|
||||
if (cm->frame_type == KEY_FRAME) {
|
||||
cpi->resize_avg_qp = 0;
|
||||
cpi->resize_count = 0;
|
||||
return 0;
|
||||
}
|
||||
// Resize based on average buffer underflow and QP over some window.
|
||||
// Ignore samples close to key frame, since QP is usually high after key.
|
||||
if (cpi->rc.frames_since_key > 2 * cpi->framerate) {
|
||||
const int window = (int)(5 * cpi->framerate);
|
||||
cpi->resize_avg_qp += cm->base_qindex;
|
||||
if (cpi->rc.buffer_level < (int)(30 * rc->optimal_buffer_level / 100))
|
||||
++cpi->resize_buffer_underflow;
|
||||
++cpi->resize_count;
|
||||
// Check for resize action every "window" frames.
|
||||
if (cpi->resize_count >= window) {
|
||||
int avg_qp = cpi->resize_avg_qp / cpi->resize_count;
|
||||
// Resize down if buffer level has underflowed sufficent amount in past
|
||||
// window, and we are at original resolution.
|
||||
// Resize back up if average QP is low, and we are currently in a resized
|
||||
// down state.
|
||||
if (cpi->resize_state == 0 &&
|
||||
cpi->resize_buffer_underflow > (cpi->resize_count >> 2)) {
|
||||
resize_now = 1;
|
||||
cpi->resize_state = 1;
|
||||
} else if (cpi->resize_state == 1 &&
|
||||
avg_qp < 40 * cpi->rc.worst_quality / 100) {
|
||||
resize_now = -1;
|
||||
cpi->resize_state = 0;
|
||||
}
|
||||
// Reset for next window measurement.
|
||||
cpi->resize_avg_qp = 0;
|
||||
cpi->resize_count = 0;
|
||||
cpi->resize_buffer_underflow = 0;
|
||||
}
|
||||
}
|
||||
// If decision is to resize, reset some quantities, and check is we should
|
||||
// reduce rate correction factor,
|
||||
if (resize_now != 0) {
|
||||
int target_bits_per_frame;
|
||||
int active_worst_quality;
|
||||
int qindex;
|
||||
int tot_scale_change;
|
||||
// For now, resize is by 1/2 x 1/2.
|
||||
cpi->resize_scale_num = 1;
|
||||
cpi->resize_scale_den = 2;
|
||||
tot_scale_change = (cpi->resize_scale_den * cpi->resize_scale_den) /
|
||||
(cpi->resize_scale_num * cpi->resize_scale_num);
|
||||
// Reset buffer level to optimal, update target size.
|
||||
rc->buffer_level = rc->optimal_buffer_level;
|
||||
rc->bits_off_target = rc->optimal_buffer_level;
|
||||
rc->this_frame_target = calc_pframe_target_size_one_pass_cbr(cpi);
|
||||
// Reset cyclic refresh parameters.
|
||||
if (cpi->oxcf.aq_mode == CYCLIC_REFRESH_AQ && cm->seg.enabled)
|
||||
av1_cyclic_refresh_reset_resize(cpi);
|
||||
// Get the projected qindex, based on the scaled target frame size (scaled
|
||||
// so target_bits_per_mb in av1_rc_regulate_q will be correct target).
|
||||
target_bits_per_frame = (resize_now == 1)
|
||||
? rc->this_frame_target * tot_scale_change
|
||||
: rc->this_frame_target / tot_scale_change;
|
||||
active_worst_quality = calc_active_worst_quality_one_pass_cbr(cpi);
|
||||
qindex = av1_rc_regulate_q(cpi, target_bits_per_frame, rc->best_quality,
|
||||
active_worst_quality);
|
||||
// If resize is down, check if projected q index is close to worst_quality,
|
||||
// and if so, reduce the rate correction factor (since likely can afford
|
||||
// lower q for resized frame).
|
||||
if (resize_now == 1 && qindex > 90 * cpi->rc.worst_quality / 100) {
|
||||
rc->rate_correction_factors[INTER_NORMAL] *= 0.85;
|
||||
}
|
||||
// If resize is back up, check if projected q index is too much above the
|
||||
// current base_qindex, and if so, reduce the rate correction factor
|
||||
// (since prefer to keep q for resized frame at least close to previous q).
|
||||
if (resize_now == -1 && qindex > 130 * cm->base_qindex / 100) {
|
||||
rc->rate_correction_factors[INTER_NORMAL] *= 0.9;
|
||||
}
|
||||
}
|
||||
return resize_now;
|
||||
}
|
||||
|
|
|
|||
29
third_party/aom/av1/encoder/ratectrl.h
vendored
29
third_party/aom/av1/encoder/ratectrl.h
vendored
|
|
@ -49,27 +49,6 @@ typedef enum {
|
|||
} RATE_FACTOR_LEVEL;
|
||||
#endif // CONFIG_EXT_REFS
|
||||
|
||||
// Internal frame scaling level.
|
||||
typedef enum {
|
||||
UNSCALED = 0, // Frame is unscaled.
|
||||
SCALE_STEP1 = 1, // First-level down-scaling.
|
||||
FRAME_SCALE_STEPS
|
||||
} FRAME_SCALE_LEVEL;
|
||||
|
||||
// Frame dimensions multiplier wrt the native frame size, in 1/16ths,
|
||||
// specified for the scale-up case.
|
||||
// e.g. 24 => 16/24 = 2/3 of native size. The restriction to 1/16th is
|
||||
// intended to match the capabilities of the normative scaling filters,
|
||||
// giving precedence to the up-scaling accuracy.
|
||||
static const int frame_scale_factor[FRAME_SCALE_STEPS] = { 16, 24 };
|
||||
|
||||
// Multiplier of the target rate to be used as threshold for triggering scaling.
|
||||
static const double rate_thresh_mult[FRAME_SCALE_STEPS] = { 1.0, 2.0 };
|
||||
|
||||
// Scale dependent Rate Correction Factor multipliers. Compensates for the
|
||||
// greater number of bits per pixel generated in down-scaled frames.
|
||||
static const double rcf_mult[FRAME_SCALE_STEPS] = { 1.0, 2.0 };
|
||||
|
||||
typedef struct {
|
||||
// Rate targetting variables
|
||||
int base_frame_target; // A baseline frame target before adjustment
|
||||
|
|
@ -162,10 +141,6 @@ typedef struct {
|
|||
int q_2_frame;
|
||||
|
||||
// Auto frame-scaling variables.
|
||||
FRAME_SCALE_LEVEL frame_size_selector;
|
||||
FRAME_SCALE_LEVEL next_frame_size_selector;
|
||||
int frame_width[FRAME_SCALE_STEPS];
|
||||
int frame_height[FRAME_SCALE_STEPS];
|
||||
int rf_level_maxq[RATE_FACTOR_LEVELS];
|
||||
} RATE_CONTROL;
|
||||
|
||||
|
|
@ -214,6 +189,10 @@ int av1_rc_get_default_max_gf_interval(double framerate, int min_frame_rate);
|
|||
void av1_rc_get_one_pass_vbr_params(struct AV1_COMP *cpi);
|
||||
void av1_rc_get_one_pass_cbr_params(struct AV1_COMP *cpi);
|
||||
|
||||
// How many times less pixels there are to encode given the current scaling.
|
||||
// Temporary replacement for rcf_mult and rate_thresh_mult.
|
||||
double av1_resize_rate_factor(const struct AV1_COMP *cpi);
|
||||
|
||||
// Post encode update of the rate control parameters based
|
||||
// on bytes used
|
||||
void av1_rc_postencode_update(struct AV1_COMP *cpi, uint64_t bytes_used);
|
||||
|
|
|
|||
144
third_party/aom/av1/encoder/rd.c
vendored
144
third_party/aom/av1/encoder/rd.c
vendored
|
|
@ -330,7 +330,6 @@ static void set_block_thresholds(const AV1_COMMON *cm, RD_OPT *rd) {
|
|||
}
|
||||
}
|
||||
|
||||
#if CONFIG_REF_MV
|
||||
void av1_set_mvcost(MACROBLOCK *x, MV_REFERENCE_FRAME ref_frame, int ref,
|
||||
int ref_mv_idx) {
|
||||
MB_MODE_INFO_EXT *mbmi_ext = x->mbmi_ext;
|
||||
|
|
@ -340,19 +339,14 @@ void av1_set_mvcost(MACROBLOCK *x, MV_REFERENCE_FRAME ref_frame, int ref,
|
|||
(void)ref_frame;
|
||||
x->mvcost = x->mv_cost_stack[nmv_ctx];
|
||||
x->nmvjointcost = x->nmv_vec_cost[nmv_ctx];
|
||||
x->mvsadcost = x->mvcost;
|
||||
x->nmvjointsadcost = x->nmvjointcost;
|
||||
}
|
||||
#endif
|
||||
|
||||
void av1_initialize_rd_consts(AV1_COMP *cpi) {
|
||||
AV1_COMMON *const cm = &cpi->common;
|
||||
MACROBLOCK *const x = &cpi->td.mb;
|
||||
RD_OPT *const rd = &cpi->rd;
|
||||
int i;
|
||||
#if CONFIG_REF_MV
|
||||
int nmv_ctx;
|
||||
#endif
|
||||
|
||||
aom_clear_system_state();
|
||||
|
||||
|
|
@ -363,7 +357,6 @@ void av1_initialize_rd_consts(AV1_COMP *cpi) {
|
|||
|
||||
set_block_thresholds(cm, rd);
|
||||
|
||||
#if CONFIG_REF_MV
|
||||
for (nmv_ctx = 0; nmv_ctx < NMV_CONTEXTS; ++nmv_ctx) {
|
||||
av1_build_nmv_cost_table(
|
||||
x->nmv_vec_cost[nmv_ctx],
|
||||
|
|
@ -373,19 +366,11 @@ void av1_initialize_rd_consts(AV1_COMP *cpi) {
|
|||
}
|
||||
x->mvcost = x->mv_cost_stack[0];
|
||||
x->nmvjointcost = x->nmv_vec_cost[0];
|
||||
x->mvsadcost = x->mvcost;
|
||||
x->nmvjointsadcost = x->nmvjointcost;
|
||||
#else
|
||||
av1_build_nmv_cost_table(
|
||||
x->nmvjointcost, cm->allow_high_precision_mv ? x->nmvcost_hp : x->nmvcost,
|
||||
&cm->fc->nmvc, cm->allow_high_precision_mv);
|
||||
#endif
|
||||
|
||||
if (cpi->oxcf.pass != 1) {
|
||||
av1_fill_token_costs(x->token_costs, cm->fc->coef_probs);
|
||||
|
||||
if (cpi->sf.partition_search_type != VAR_BASED_PARTITION ||
|
||||
cm->frame_type == KEY_FRAME) {
|
||||
if (cm->frame_type == KEY_FRAME) {
|
||||
#if CONFIG_EXT_PARTITION_TYPES
|
||||
for (i = 0; i < PARTITION_PLOFFSET; ++i)
|
||||
av1_cost_tokens(cpi->partition_cost[i], cm->fc->partition_prob[i],
|
||||
|
|
@ -425,7 +410,6 @@ void av1_initialize_rd_consts(AV1_COMP *cpi) {
|
|||
fill_mode_costs(cpi);
|
||||
|
||||
if (!frame_is_intra_only(cm)) {
|
||||
#if CONFIG_REF_MV
|
||||
for (i = 0; i < NEWMV_MODE_CONTEXTS; ++i) {
|
||||
cpi->newmv_mode_cost[i][0] = av1_cost_bit(cm->fc->newmv_prob[i], 0);
|
||||
cpi->newmv_mode_cost[i][1] = av1_cost_bit(cm->fc->newmv_prob[i], 1);
|
||||
|
|
@ -445,20 +429,17 @@ void av1_initialize_rd_consts(AV1_COMP *cpi) {
|
|||
cpi->drl_mode_cost0[i][0] = av1_cost_bit(cm->fc->drl_prob[i], 0);
|
||||
cpi->drl_mode_cost0[i][1] = av1_cost_bit(cm->fc->drl_prob[i], 1);
|
||||
}
|
||||
#else
|
||||
for (i = 0; i < INTER_MODE_CONTEXTS; ++i)
|
||||
av1_cost_tokens((int *)cpi->inter_mode_cost[i],
|
||||
cm->fc->inter_mode_probs[i], av1_inter_mode_tree);
|
||||
#endif // CONFIG_REF_MV
|
||||
#if CONFIG_EXT_INTER
|
||||
for (i = 0; i < INTER_MODE_CONTEXTS; ++i)
|
||||
av1_cost_tokens((int *)cpi->inter_compound_mode_cost[i],
|
||||
cm->fc->inter_compound_mode_probs[i],
|
||||
av1_inter_compound_mode_tree);
|
||||
#if CONFIG_INTERINTRA
|
||||
for (i = 0; i < BLOCK_SIZE_GROUPS; ++i)
|
||||
av1_cost_tokens((int *)cpi->interintra_mode_cost[i],
|
||||
cm->fc->interintra_mode_prob[i],
|
||||
av1_interintra_mode_tree);
|
||||
#endif // CONFIG_INTERINTRA
|
||||
#endif // CONFIG_EXT_INTER
|
||||
#if CONFIG_MOTION_VAR || CONFIG_WARPED_MOTION
|
||||
for (i = BLOCK_8X8; i < BLOCK_SIZES; i++) {
|
||||
|
|
@ -575,9 +556,15 @@ static void get_entropy_contexts_plane(
|
|||
const ENTROPY_CONTEXT *const above = pd->above_context;
|
||||
const ENTROPY_CONTEXT *const left = pd->left_context;
|
||||
|
||||
#if CONFIG_LV_MAP
|
||||
memcpy(t_above, above, sizeof(ENTROPY_CONTEXT) * num_4x4_w);
|
||||
memcpy(t_left, left, sizeof(ENTROPY_CONTEXT) * num_4x4_h);
|
||||
return;
|
||||
#endif // CONFIG_LV_MAP
|
||||
|
||||
int i;
|
||||
|
||||
#if CONFIG_CB4X4
|
||||
#if CONFIG_CHROMA_2X2
|
||||
switch (tx_size) {
|
||||
case TX_2X2:
|
||||
memcpy(t_above, above, sizeof(ENTROPY_CONTEXT) * num_4x4_w);
|
||||
|
|
@ -609,6 +596,20 @@ static void get_entropy_contexts_plane(
|
|||
t_left[i] =
|
||||
!!(*(const uint64_t *)&left[i] | *(const uint64_t *)&left[i + 8]);
|
||||
break;
|
||||
#if CONFIG_TX64X64
|
||||
case TX_64X64:
|
||||
for (i = 0; i < num_4x4_w; i += 32)
|
||||
t_above[i] =
|
||||
!!(*(const uint64_t *)&above[i] | *(const uint64_t *)&above[i + 8] |
|
||||
*(const uint64_t *)&above[i + 16] |
|
||||
*(const uint64_t *)&above[i + 24]);
|
||||
for (i = 0; i < num_4x4_h; i += 32)
|
||||
t_left[i] =
|
||||
!!(*(const uint64_t *)&left[i] | *(const uint64_t *)&left[i + 8] |
|
||||
*(const uint64_t *)&left[i + 16] |
|
||||
*(const uint64_t *)&left[i + 24]);
|
||||
break;
|
||||
#endif // CONFIG_TX64X64
|
||||
case TX_4X8:
|
||||
for (i = 0; i < num_4x4_w; i += 2)
|
||||
t_above[i] = !!*(const uint16_t *)&above[i];
|
||||
|
|
@ -647,11 +648,39 @@ static void get_entropy_contexts_plane(
|
|||
for (i = 0; i < num_4x4_h; i += 8)
|
||||
t_left[i] = !!*(const uint64_t *)&left[i];
|
||||
break;
|
||||
#if CONFIG_EXT_TX && CONFIG_RECT_TX && CONFIG_RECT_TX_EXT
|
||||
case TX_4X16:
|
||||
for (i = 0; i < num_4x4_w; i += 2)
|
||||
t_above[i] = !!*(const uint16_t *)&above[i];
|
||||
for (i = 0; i < num_4x4_h; i += 8)
|
||||
t_left[i] = !!*(const uint64_t *)&left[i];
|
||||
break;
|
||||
case TX_16X4:
|
||||
for (i = 0; i < num_4x4_w; i += 8)
|
||||
t_above[i] = !!*(const uint64_t *)&above[i];
|
||||
for (i = 0; i < num_4x4_h; i += 2)
|
||||
t_left[i] = !!*(const uint16_t *)&left[i];
|
||||
break;
|
||||
case TX_8X32:
|
||||
for (i = 0; i < num_4x4_w; i += 4)
|
||||
t_above[i] = !!*(const uint32_t *)&above[i];
|
||||
for (i = 0; i < num_4x4_h; i += 16)
|
||||
t_left[i] =
|
||||
!!(*(const uint64_t *)&left[i] | *(const uint64_t *)&left[i + 8]);
|
||||
break;
|
||||
case TX_32X8:
|
||||
for (i = 0; i < num_4x4_w; i += 16)
|
||||
t_above[i] =
|
||||
!!(*(const uint64_t *)&above[i] | *(const uint64_t *)&above[i + 8]);
|
||||
for (i = 0; i < num_4x4_h; i += 4)
|
||||
t_left[i] = !!*(const uint32_t *)&left[i];
|
||||
break;
|
||||
#endif // CONFIG_EXT_TX && CONFIG_RECT_TX && CONFIG_RECT_TX_EXT
|
||||
|
||||
default: assert(0 && "Invalid transform size."); break;
|
||||
}
|
||||
return;
|
||||
#endif
|
||||
#endif // CONFIG_CHROMA_2X2
|
||||
|
||||
switch (tx_size) {
|
||||
case TX_4X4:
|
||||
|
|
@ -720,6 +749,30 @@ static void get_entropy_contexts_plane(
|
|||
for (i = 0; i < num_4x4_h; i += 4)
|
||||
t_left[i] = !!*(const uint32_t *)&left[i];
|
||||
break;
|
||||
#if CONFIG_EXT_TX && CONFIG_RECT_TX && CONFIG_RECT_TX_EXT
|
||||
case TX_4X16:
|
||||
memcpy(t_above, above, sizeof(ENTROPY_CONTEXT) * num_4x4_w);
|
||||
for (i = 0; i < num_4x4_h; i += 4)
|
||||
t_left[i] = !!*(const uint32_t *)&left[i];
|
||||
break;
|
||||
case TX_16X4:
|
||||
for (i = 0; i < num_4x4_w; i += 4)
|
||||
t_above[i] = !!*(const uint32_t *)&above[i];
|
||||
memcpy(t_left, left, sizeof(ENTROPY_CONTEXT) * num_4x4_h);
|
||||
break;
|
||||
case TX_8X32:
|
||||
for (i = 0; i < num_4x4_w; i += 2)
|
||||
t_above[i] = !!*(const uint16_t *)&above[i];
|
||||
for (i = 0; i < num_4x4_h; i += 8)
|
||||
t_left[i] = !!*(const uint64_t *)&left[i];
|
||||
break;
|
||||
case TX_32X8:
|
||||
for (i = 0; i < num_4x4_w; i += 8)
|
||||
t_above[i] = !!*(const uint64_t *)&above[i];
|
||||
for (i = 0; i < num_4x4_h; i += 2)
|
||||
t_left[i] = !!*(const uint16_t *)&left[i];
|
||||
break;
|
||||
#endif // CONFIG_EXT_TX && CONFIG_RECT_TX && CONFIG_RECT_TX_EXT
|
||||
default: assert(0 && "Invalid transform size."); break;
|
||||
}
|
||||
}
|
||||
|
|
@ -728,7 +781,12 @@ void av1_get_entropy_contexts(BLOCK_SIZE bsize, TX_SIZE tx_size,
|
|||
const struct macroblockd_plane *pd,
|
||||
ENTROPY_CONTEXT t_above[2 * MAX_MIB_SIZE],
|
||||
ENTROPY_CONTEXT t_left[2 * MAX_MIB_SIZE]) {
|
||||
#if CONFIG_CB4X4 && !CONFIG_CHROMA_2X2
|
||||
const BLOCK_SIZE plane_bsize =
|
||||
AOMMAX(BLOCK_4X4, get_plane_block_size(bsize, pd));
|
||||
#else
|
||||
const BLOCK_SIZE plane_bsize = get_plane_block_size(bsize, pd);
|
||||
#endif
|
||||
get_entropy_contexts_plane(plane_bsize, tx_size, pd, t_above, t_left);
|
||||
}
|
||||
|
||||
|
|
@ -740,27 +798,25 @@ void av1_mv_pred(const AV1_COMP *cpi, MACROBLOCK *x, uint8_t *ref_y_buffer,
|
|||
int best_sad = INT_MAX;
|
||||
int this_sad = INT_MAX;
|
||||
int max_mv = 0;
|
||||
int near_same_nearest;
|
||||
uint8_t *src_y_ptr = x->plane[0].src.buf;
|
||||
uint8_t *ref_y_ptr;
|
||||
const int num_mv_refs =
|
||||
MAX_MV_REF_CANDIDATES +
|
||||
(cpi->sf.adaptive_motion_search && block_size < x->max_partition_size);
|
||||
MV pred_mv[MAX_MV_REF_CANDIDATES + 1];
|
||||
int num_mv_refs = 0;
|
||||
|
||||
pred_mv[num_mv_refs++] = x->mbmi_ext->ref_mvs[ref_frame][0].as_mv;
|
||||
if (x->mbmi_ext->ref_mvs[ref_frame][0].as_int !=
|
||||
x->mbmi_ext->ref_mvs[ref_frame][1].as_int) {
|
||||
pred_mv[num_mv_refs++] = x->mbmi_ext->ref_mvs[ref_frame][1].as_mv;
|
||||
}
|
||||
if (cpi->sf.adaptive_motion_search && block_size < x->max_partition_size)
|
||||
pred_mv[num_mv_refs++] = x->pred_mv[ref_frame];
|
||||
|
||||
MV pred_mv[3];
|
||||
pred_mv[0] = x->mbmi_ext->ref_mvs[ref_frame][0].as_mv;
|
||||
pred_mv[1] = x->mbmi_ext->ref_mvs[ref_frame][1].as_mv;
|
||||
pred_mv[2] = x->pred_mv[ref_frame];
|
||||
assert(num_mv_refs <= (int)(sizeof(pred_mv) / sizeof(pred_mv[0])));
|
||||
|
||||
near_same_nearest = x->mbmi_ext->ref_mvs[ref_frame][0].as_int ==
|
||||
x->mbmi_ext->ref_mvs[ref_frame][1].as_int;
|
||||
// Get the sad for each candidate reference mv.
|
||||
for (i = 0; i < num_mv_refs; ++i) {
|
||||
const MV *this_mv = &pred_mv[i];
|
||||
int fp_row, fp_col;
|
||||
|
||||
if (i == 1 && near_same_nearest) continue;
|
||||
fp_row = (this_mv->row + 3 + (this_mv->row >= 0)) >> 3;
|
||||
fp_col = (this_mv->col + 3 + (this_mv->col >= 0)) >> 3;
|
||||
max_mv = AOMMAX(max_mv, AOMMAX(abs(this_mv->row), abs(this_mv->col)) >> 3);
|
||||
|
|
@ -959,8 +1015,6 @@ void av1_set_rd_speed_thresholds(AV1_COMP *cpi) {
|
|||
|
||||
#if CONFIG_EXT_INTER
|
||||
|
||||
rd->thresh_mult[THR_COMP_NEAREST_NEARLA] += 1200;
|
||||
rd->thresh_mult[THR_COMP_NEAR_NEARESTLA] += 1200;
|
||||
rd->thresh_mult[THR_COMP_NEAR_NEARLA] += 1200;
|
||||
rd->thresh_mult[THR_COMP_NEAREST_NEWLA] += 1500;
|
||||
rd->thresh_mult[THR_COMP_NEW_NEARESTLA] += 1500;
|
||||
|
|
@ -970,8 +1024,6 @@ void av1_set_rd_speed_thresholds(AV1_COMP *cpi) {
|
|||
rd->thresh_mult[THR_COMP_ZERO_ZEROLA] += 2500;
|
||||
|
||||
#if CONFIG_EXT_REFS
|
||||
rd->thresh_mult[THR_COMP_NEAREST_NEARL2A] += 1200;
|
||||
rd->thresh_mult[THR_COMP_NEAR_NEARESTL2A] += 1200;
|
||||
rd->thresh_mult[THR_COMP_NEAR_NEARL2A] += 1200;
|
||||
rd->thresh_mult[THR_COMP_NEAREST_NEWL2A] += 1500;
|
||||
rd->thresh_mult[THR_COMP_NEW_NEARESTL2A] += 1500;
|
||||
|
|
@ -980,8 +1032,6 @@ void av1_set_rd_speed_thresholds(AV1_COMP *cpi) {
|
|||
rd->thresh_mult[THR_COMP_NEW_NEWL2A] += 2000;
|
||||
rd->thresh_mult[THR_COMP_ZERO_ZEROL2A] += 2500;
|
||||
|
||||
rd->thresh_mult[THR_COMP_NEAREST_NEARL3A] += 1200;
|
||||
rd->thresh_mult[THR_COMP_NEAR_NEARESTL3A] += 1200;
|
||||
rd->thresh_mult[THR_COMP_NEAR_NEARL3A] += 1200;
|
||||
rd->thresh_mult[THR_COMP_NEAREST_NEWL3A] += 1500;
|
||||
rd->thresh_mult[THR_COMP_NEW_NEARESTL3A] += 1500;
|
||||
|
|
@ -991,8 +1041,6 @@ void av1_set_rd_speed_thresholds(AV1_COMP *cpi) {
|
|||
rd->thresh_mult[THR_COMP_ZERO_ZEROL3A] += 2500;
|
||||
#endif // CONFIG_EXT_REFS
|
||||
|
||||
rd->thresh_mult[THR_COMP_NEAREST_NEARGA] += 1200;
|
||||
rd->thresh_mult[THR_COMP_NEAR_NEARESTGA] += 1200;
|
||||
rd->thresh_mult[THR_COMP_NEAR_NEARGA] += 1200;
|
||||
rd->thresh_mult[THR_COMP_NEAREST_NEWGA] += 1500;
|
||||
rd->thresh_mult[THR_COMP_NEW_NEARESTGA] += 1500;
|
||||
|
|
@ -1002,8 +1050,6 @@ void av1_set_rd_speed_thresholds(AV1_COMP *cpi) {
|
|||
rd->thresh_mult[THR_COMP_ZERO_ZEROGA] += 2500;
|
||||
|
||||
#if CONFIG_EXT_REFS
|
||||
rd->thresh_mult[THR_COMP_NEAREST_NEARLB] += 1200;
|
||||
rd->thresh_mult[THR_COMP_NEAR_NEARESTLB] += 1200;
|
||||
rd->thresh_mult[THR_COMP_NEAR_NEARLB] += 1200;
|
||||
rd->thresh_mult[THR_COMP_NEAREST_NEWLB] += 1500;
|
||||
rd->thresh_mult[THR_COMP_NEW_NEARESTLB] += 1500;
|
||||
|
|
@ -1012,8 +1058,6 @@ void av1_set_rd_speed_thresholds(AV1_COMP *cpi) {
|
|||
rd->thresh_mult[THR_COMP_NEW_NEWLB] += 2000;
|
||||
rd->thresh_mult[THR_COMP_ZERO_ZEROLB] += 2500;
|
||||
|
||||
rd->thresh_mult[THR_COMP_NEAREST_NEARL2B] += 1200;
|
||||
rd->thresh_mult[THR_COMP_NEAR_NEARESTL2B] += 1200;
|
||||
rd->thresh_mult[THR_COMP_NEAR_NEARL2B] += 1200;
|
||||
rd->thresh_mult[THR_COMP_NEAREST_NEWL2B] += 1500;
|
||||
rd->thresh_mult[THR_COMP_NEW_NEARESTL2B] += 1500;
|
||||
|
|
@ -1022,8 +1066,6 @@ void av1_set_rd_speed_thresholds(AV1_COMP *cpi) {
|
|||
rd->thresh_mult[THR_COMP_NEW_NEWL2B] += 2000;
|
||||
rd->thresh_mult[THR_COMP_ZERO_ZEROL2B] += 2500;
|
||||
|
||||
rd->thresh_mult[THR_COMP_NEAREST_NEARL3B] += 1200;
|
||||
rd->thresh_mult[THR_COMP_NEAR_NEARESTL3B] += 1200;
|
||||
rd->thresh_mult[THR_COMP_NEAR_NEARL3B] += 1200;
|
||||
rd->thresh_mult[THR_COMP_NEAREST_NEWL3B] += 1500;
|
||||
rd->thresh_mult[THR_COMP_NEW_NEARESTL3B] += 1500;
|
||||
|
|
@ -1032,8 +1074,6 @@ void av1_set_rd_speed_thresholds(AV1_COMP *cpi) {
|
|||
rd->thresh_mult[THR_COMP_NEW_NEWL3B] += 2000;
|
||||
rd->thresh_mult[THR_COMP_ZERO_ZEROL3B] += 2500;
|
||||
|
||||
rd->thresh_mult[THR_COMP_NEAREST_NEARGB] += 1200;
|
||||
rd->thresh_mult[THR_COMP_NEAR_NEARESTGB] += 1200;
|
||||
rd->thresh_mult[THR_COMP_NEAR_NEARGB] += 1200;
|
||||
rd->thresh_mult[THR_COMP_NEAREST_NEWGB] += 1500;
|
||||
rd->thresh_mult[THR_COMP_NEW_NEARESTGB] += 1500;
|
||||
|
|
|
|||
15
third_party/aom/av1/encoder/rd.h
vendored
15
third_party/aom/av1/encoder/rd.h
vendored
|
|
@ -130,6 +130,10 @@ typedef enum {
|
|||
|
||||
#if CONFIG_ALT_INTRA
|
||||
THR_SMOOTH,
|
||||
#if CONFIG_SMOOTH_HV
|
||||
THR_SMOOTH_V,
|
||||
THR_SMOOTH_H,
|
||||
#endif // CONFIG_SMOOTH_HV
|
||||
#endif // CONFIG_ALT_INTRA
|
||||
|
||||
#if CONFIG_EXT_INTER
|
||||
|
|
@ -357,6 +361,9 @@ static INLINE void av1_init_rd_stats(RD_STATS *rd_stats) {
|
|||
rd_stats->rdcost = 0;
|
||||
rd_stats->sse = 0;
|
||||
rd_stats->skip = 1;
|
||||
#if CONFIG_DAALA_DIST && CONFIG_CB4X4
|
||||
rd_stats->dist_y = 0;
|
||||
#endif
|
||||
#if CONFIG_RD_DEBUG
|
||||
for (plane = 0; plane < MAX_MB_PLANE; ++plane) {
|
||||
rd_stats->txb_coeff_cost[plane] = 0;
|
||||
|
|
@ -381,6 +388,9 @@ static INLINE void av1_invalid_rd_stats(RD_STATS *rd_stats) {
|
|||
rd_stats->rdcost = INT64_MAX;
|
||||
rd_stats->sse = INT64_MAX;
|
||||
rd_stats->skip = 0;
|
||||
#if CONFIG_DAALA_DIST && CONFIG_CB4X4
|
||||
rd_stats->dist_y = INT64_MAX;
|
||||
#endif
|
||||
#if CONFIG_RD_DEBUG
|
||||
for (plane = 0; plane < MAX_MB_PLANE; ++plane) {
|
||||
rd_stats->txb_coeff_cost[plane] = INT_MAX;
|
||||
|
|
@ -405,6 +415,9 @@ static INLINE void av1_merge_rd_stats(RD_STATS *rd_stats_dst,
|
|||
rd_stats_dst->dist += rd_stats_src->dist;
|
||||
rd_stats_dst->sse += rd_stats_src->sse;
|
||||
rd_stats_dst->skip &= rd_stats_src->skip;
|
||||
#if CONFIG_DAALA_DIST && CONFIG_CB4X4
|
||||
rd_stats_dst->dist_y += rd_stats_src->dist_y;
|
||||
#endif
|
||||
#if CONFIG_RD_DEBUG
|
||||
for (plane = 0; plane < MAX_MB_PLANE; ++plane) {
|
||||
rd_stats_dst->txb_coeff_cost[plane] += rd_stats_src->txb_coeff_cost[plane];
|
||||
|
|
@ -454,10 +467,8 @@ YV12_BUFFER_CONFIG *av1_get_scaled_ref_frame(const struct AV1_COMP *cpi,
|
|||
|
||||
void av1_init_me_luts(void);
|
||||
|
||||
#if CONFIG_REF_MV
|
||||
void av1_set_mvcost(MACROBLOCK *x, MV_REFERENCE_FRAME ref_frame, int ref,
|
||||
int ref_mv_idx);
|
||||
#endif
|
||||
|
||||
void av1_get_entropy_contexts(BLOCK_SIZE bsize, TX_SIZE tx_size,
|
||||
const struct macroblockd_plane *pd,
|
||||
|
|
|
|||
3789
third_party/aom/av1/encoder/rdopt.c
vendored
3789
third_party/aom/av1/encoder/rdopt.c
vendored
File diff suppressed because it is too large
Load diff
16
third_party/aom/av1/encoder/rdopt.h
vendored
16
third_party/aom/av1/encoder/rdopt.h
vendored
|
|
@ -62,6 +62,12 @@ void av1_dist_block(const AV1_COMP *cpi, MACROBLOCK *x, int plane,
|
|||
TX_SIZE tx_size, int64_t *out_dist, int64_t *out_sse,
|
||||
OUTPUT_STATUS output_status);
|
||||
|
||||
#if CONFIG_DAALA_DIST
|
||||
int64_t av1_daala_dist(const uint8_t *src, int src_stride, const uint8_t *dst,
|
||||
int dst_stride, int bsw, int bsh, int qm,
|
||||
int use_activity_masking, int qindex);
|
||||
#endif
|
||||
|
||||
#if !CONFIG_PVQ || CONFIG_VAR_TX
|
||||
int av1_cost_coeffs(const AV1_COMP *const cpi, MACROBLOCK *x, int plane,
|
||||
int block, TX_SIZE tx_size, const SCAN_ORDER *scan_order,
|
||||
|
|
@ -101,16 +107,6 @@ int av1_active_h_edge(const struct AV1_COMP *cpi, int mi_row, int mi_step);
|
|||
int av1_active_v_edge(const struct AV1_COMP *cpi, int mi_col, int mi_step);
|
||||
int av1_active_edge_sb(const struct AV1_COMP *cpi, int mi_row, int mi_col);
|
||||
|
||||
void av1_rd_pick_inter_mode_sub8x8(const struct AV1_COMP *cpi,
|
||||
struct TileDataEnc *tile_data,
|
||||
struct macroblock *x, int mi_row, int mi_col,
|
||||
struct RD_STATS *rd_cost,
|
||||
#if CONFIG_SUPERTX
|
||||
int *returnrate_nocoef,
|
||||
#endif // CONFIG_SUPERTX
|
||||
BLOCK_SIZE bsize, PICK_MODE_CONTEXT *ctx,
|
||||
int64_t best_rd_so_far);
|
||||
|
||||
#if CONFIG_MOTION_VAR && CONFIG_NCOBMC
|
||||
void av1_check_ncobmc_rd(const struct AV1_COMP *cpi, struct macroblock *x,
|
||||
int mi_row, int mi_col);
|
||||
|
|
|
|||
23
third_party/aom/av1/encoder/speed_features.c
vendored
23
third_party/aom/av1/encoder/speed_features.c
vendored
|
|
@ -139,8 +139,10 @@ static void set_good_speed_feature_framesize_dependent(AV1_COMP *cpi,
|
|||
}
|
||||
}
|
||||
|
||||
static void set_good_speed_feature(AV1_COMP *cpi, AV1_COMMON *cm,
|
||||
SPEED_FEATURES *sf, int speed) {
|
||||
static void set_good_speed_features_framesize_independent(AV1_COMP *cpi,
|
||||
SPEED_FEATURES *sf,
|
||||
int speed) {
|
||||
AV1_COMMON *const cm = &cpi->common;
|
||||
const int boosted = frame_is_boosted(cpi);
|
||||
|
||||
if (speed >= 1) {
|
||||
|
|
@ -205,6 +207,9 @@ static void set_good_speed_feature(AV1_COMP *cpi, AV1_COMMON *cm,
|
|||
#if CONFIG_EXT_TX
|
||||
sf->tx_type_search.prune_mode = PRUNE_TWO;
|
||||
#endif
|
||||
#if CONFIG_GLOBAL_MOTION
|
||||
sf->gm_search_type = GM_DISABLE_SEARCH;
|
||||
#endif // CONFIG_GLOBAL_MOTION
|
||||
}
|
||||
|
||||
if (speed >= 4) {
|
||||
|
|
@ -286,6 +291,12 @@ static void set_good_speed_feature(AV1_COMP *cpi, AV1_COMMON *cm,
|
|||
sf->coeff_prob_appx_step = 4;
|
||||
sf->mode_search_skip_flags |= FLAG_SKIP_INTRA_DIRMISMATCH;
|
||||
}
|
||||
if (speed >= 8) {
|
||||
sf->mv.search_method = FAST_DIAMOND;
|
||||
sf->mv.fullpel_search_step_param = 10;
|
||||
sf->mv.subpel_force_stop = 2;
|
||||
sf->lpf_pick = LPF_PICK_MINIMAL_LPF;
|
||||
}
|
||||
}
|
||||
|
||||
void av1_set_speed_features_framesize_dependent(AV1_COMP *cpi) {
|
||||
|
|
@ -339,12 +350,13 @@ void av1_set_speed_features_framesize_dependent(AV1_COMP *cpi) {
|
|||
}
|
||||
|
||||
void av1_set_speed_features_framesize_independent(AV1_COMP *cpi) {
|
||||
SPEED_FEATURES *const sf = &cpi->sf;
|
||||
AV1_COMMON *const cm = &cpi->common;
|
||||
SPEED_FEATURES *const sf = &cpi->sf;
|
||||
MACROBLOCK *const x = &cpi->td.mb;
|
||||
const AV1EncoderConfig *const oxcf = &cpi->oxcf;
|
||||
int i;
|
||||
|
||||
(void)cm;
|
||||
// best quality defaults
|
||||
sf->frame_parameter_update = 1;
|
||||
sf->mv.search_method = NSTEP;
|
||||
|
|
@ -418,13 +430,16 @@ void av1_set_speed_features_framesize_independent(AV1_COMP *cpi) {
|
|||
|
||||
// Set this at the appropriate speed levels
|
||||
sf->use_transform_domain_distortion = 0;
|
||||
#if CONFIG_GLOBAL_MOTION
|
||||
sf->gm_search_type = GM_FULL_SEARCH;
|
||||
#endif // CONFIG_GLOBAL_MOTION
|
||||
|
||||
if (oxcf->mode == GOOD
|
||||
#if CONFIG_XIPHRC
|
||||
|| oxcf->pass == 1
|
||||
#endif
|
||||
)
|
||||
set_good_speed_feature(cpi, cm, sf, oxcf->speed);
|
||||
set_good_speed_features_framesize_independent(cpi, sf, oxcf->speed);
|
||||
|
||||
// sf->partition_search_breakout_dist_thr is set assuming max 64x64
|
||||
// blocks. Normalise this if the blocks are bigger.
|
||||
|
|
|
|||
62
third_party/aom/av1/encoder/speed_features.h
vendored
62
third_party/aom/av1/encoder/speed_features.h
vendored
|
|
@ -24,6 +24,9 @@ enum {
|
|||
(1 << D207_PRED) | (1 << D63_PRED) |
|
||||
#if CONFIG_ALT_INTRA
|
||||
(1 << SMOOTH_PRED) |
|
||||
#if CONFIG_SMOOTH_HV
|
||||
(1 << SMOOTH_V_PRED) | (1 << SMOOTH_H_PRED) |
|
||||
#endif // CONFIG_SMOOTH_HV
|
||||
#endif // CONFIG_ALT_INTRA
|
||||
(1 << TM_PRED),
|
||||
INTRA_DC = (1 << DC_PRED),
|
||||
|
|
@ -36,37 +39,33 @@ enum {
|
|||
#if CONFIG_EXT_INTER
|
||||
enum {
|
||||
INTER_ALL = (1 << NEARESTMV) | (1 << NEARMV) | (1 << ZEROMV) | (1 << NEWMV) |
|
||||
(1 << NEAREST_NEARESTMV) | (1 << NEAR_NEARMV) |
|
||||
(1 << NEAREST_NEARMV) | (1 << NEAR_NEARESTMV) | (1 << NEW_NEWMV) |
|
||||
(1 << NEAREST_NEARESTMV) | (1 << NEAR_NEARMV) | (1 << NEW_NEWMV) |
|
||||
(1 << NEAREST_NEWMV) | (1 << NEAR_NEWMV) | (1 << NEW_NEARMV) |
|
||||
(1 << NEW_NEARESTMV) | (1 << ZERO_ZEROMV),
|
||||
INTER_NEAREST = (1 << NEARESTMV) | (1 << NEAREST_NEARESTMV) |
|
||||
(1 << NEAREST_NEARMV) | (1 << NEAR_NEARESTMV) |
|
||||
(1 << NEW_NEARESTMV) | (1 << NEAREST_NEWMV),
|
||||
INTER_NEAREST_NEW = (1 << NEARESTMV) | (1 << NEWMV) |
|
||||
(1 << NEAREST_NEARESTMV) | (1 << NEW_NEWMV) |
|
||||
(1 << NEAR_NEARESTMV) | (1 << NEAREST_NEARMV) |
|
||||
(1 << NEW_NEARESTMV) | (1 << NEAREST_NEWMV) |
|
||||
(1 << NEW_NEARMV) | (1 << NEAR_NEWMV),
|
||||
INTER_NEAREST_ZERO = (1 << NEARESTMV) | (1 << ZEROMV) |
|
||||
(1 << NEAREST_NEARESTMV) | (1 << ZERO_ZEROMV) |
|
||||
(1 << NEAREST_NEARMV) | (1 << NEAR_NEARESTMV) |
|
||||
(1 << NEAREST_NEWMV) | (1 << NEW_NEARESTMV),
|
||||
INTER_NEAREST_NEW_ZERO =
|
||||
(1 << NEARESTMV) | (1 << ZEROMV) | (1 << NEWMV) |
|
||||
(1 << NEAREST_NEARESTMV) | (1 << ZERO_ZEROMV) | (1 << NEW_NEWMV) |
|
||||
(1 << NEAREST_NEARMV) | (1 << NEAR_NEARESTMV) | (1 << NEW_NEARESTMV) |
|
||||
(1 << NEAREST_NEWMV) | (1 << NEW_NEARMV) | (1 << NEAR_NEWMV),
|
||||
INTER_NEAREST_NEAR_NEW =
|
||||
(1 << NEARESTMV) | (1 << NEARMV) | (1 << NEWMV) |
|
||||
(1 << NEAREST_NEARESTMV) | (1 << NEW_NEWMV) | (1 << NEAREST_NEARMV) |
|
||||
(1 << NEAR_NEARESTMV) | (1 << NEW_NEARESTMV) | (1 << NEAREST_NEWMV) |
|
||||
(1 << NEW_NEARMV) | (1 << NEAR_NEWMV) | (1 << NEAR_NEARMV),
|
||||
INTER_NEAREST_NEAR_ZERO =
|
||||
(1 << NEARESTMV) | (1 << NEARMV) | (1 << ZEROMV) |
|
||||
(1 << NEAREST_NEARESTMV) | (1 << ZERO_ZEROMV) | (1 << NEAREST_NEARMV) |
|
||||
(1 << NEAR_NEARESTMV) | (1 << NEAREST_NEWMV) | (1 << NEW_NEARESTMV) |
|
||||
(1 << NEW_NEARMV) | (1 << NEAR_NEWMV) | (1 << NEAR_NEARMV),
|
||||
INTER_NEAREST_NEW_ZERO = (1 << NEARESTMV) | (1 << ZEROMV) | (1 << NEWMV) |
|
||||
(1 << NEAREST_NEARESTMV) | (1 << ZERO_ZEROMV) |
|
||||
(1 << NEW_NEWMV) | (1 << NEW_NEARESTMV) |
|
||||
(1 << NEAREST_NEWMV) | (1 << NEW_NEARMV) |
|
||||
(1 << NEAR_NEWMV),
|
||||
INTER_NEAREST_NEAR_NEW = (1 << NEARESTMV) | (1 << NEARMV) | (1 << NEWMV) |
|
||||
(1 << NEAREST_NEARESTMV) | (1 << NEW_NEWMV) |
|
||||
(1 << NEW_NEARESTMV) | (1 << NEAREST_NEWMV) |
|
||||
(1 << NEW_NEARMV) | (1 << NEAR_NEWMV) |
|
||||
(1 << NEAR_NEARMV),
|
||||
INTER_NEAREST_NEAR_ZERO = (1 << NEARESTMV) | (1 << NEARMV) | (1 << ZEROMV) |
|
||||
(1 << NEAREST_NEARESTMV) | (1 << ZERO_ZEROMV) |
|
||||
(1 << NEAREST_NEWMV) | (1 << NEW_NEARESTMV) |
|
||||
(1 << NEW_NEARMV) | (1 << NEAR_NEWMV) |
|
||||
(1 << NEAR_NEARMV),
|
||||
};
|
||||
#else
|
||||
enum {
|
||||
|
|
@ -196,14 +195,7 @@ typedef enum {
|
|||
// Always use a fixed size partition
|
||||
FIXED_PARTITION,
|
||||
|
||||
REFERENCE_PARTITION,
|
||||
|
||||
// Use an arbitrary partitioning scheme based on source variance within
|
||||
// a 64X64 SB
|
||||
VAR_BASED_PARTITION,
|
||||
|
||||
// Use non-fixed partitions based on source variance
|
||||
SOURCE_VAR_BASED_PARTITION
|
||||
REFERENCE_PARTITION
|
||||
} PARTITION_SEARCH_TYPE;
|
||||
|
||||
typedef enum {
|
||||
|
|
@ -251,6 +243,14 @@ typedef struct MESH_PATTERN {
|
|||
int interval;
|
||||
} MESH_PATTERN;
|
||||
|
||||
#if CONFIG_GLOBAL_MOTION
|
||||
typedef enum {
|
||||
GM_FULL_SEARCH,
|
||||
GM_REDUCED_REF_SEARCH,
|
||||
GM_DISABLE_SEARCH
|
||||
} GM_SEARCH_TYPE;
|
||||
#endif // CONFIG_GLOBAL_MOTION
|
||||
|
||||
typedef struct SPEED_FEATURES {
|
||||
MV_SPEED_FEATURES mv;
|
||||
|
||||
|
|
@ -432,7 +432,7 @@ typedef struct SPEED_FEATURES {
|
|||
// TODO(aconverse): Fold this into one of the other many mode skips
|
||||
BLOCK_SIZE max_intra_bsize;
|
||||
|
||||
// The frequency that we check if SOURCE_VAR_BASED_PARTITION or
|
||||
// The frequency that we check if
|
||||
// FIXED_PARTITION search type should be used.
|
||||
int search_type_check_frequency;
|
||||
|
||||
|
|
@ -470,6 +470,10 @@ typedef struct SPEED_FEATURES {
|
|||
// Whether to compute distortion in the image domain (slower but
|
||||
// more accurate), or in the transform domain (faster but less acurate).
|
||||
int use_transform_domain_distortion;
|
||||
|
||||
#if CONFIG_GLOBAL_MOTION
|
||||
GM_SEARCH_TYPE gm_search_type;
|
||||
#endif // CONFIG_GLOBAL_MOTION
|
||||
} SPEED_FEATURES;
|
||||
|
||||
struct AV1_COMP;
|
||||
|
|
|
|||
77
third_party/aom/av1/encoder/subexp.c
vendored
77
third_party/aom/av1/encoder/subexp.c
vendored
|
|
@ -179,83 +179,6 @@ int av1_prob_diff_update_savings_search_model(const unsigned int *ct,
|
|||
return bestsavings;
|
||||
}
|
||||
|
||||
#if CONFIG_SUBFRAME_PROB_UPDATE
|
||||
static int get_cost(unsigned int ct[][2], aom_prob p, int n) {
|
||||
int i, p0 = p;
|
||||
unsigned int total_ct[2] = { 0, 0 };
|
||||
int cost = 0;
|
||||
|
||||
for (i = 0; i <= n; ++i) {
|
||||
cost += cost_branch256(ct[i], p);
|
||||
total_ct[0] += ct[i][0];
|
||||
total_ct[1] += ct[i][1];
|
||||
if (i < n)
|
||||
p = av1_merge_probs(p0, total_ct, COEF_COUNT_SAT, COEF_MAX_UPDATE_FACTOR);
|
||||
}
|
||||
return cost;
|
||||
}
|
||||
|
||||
int av1_prob_update_search_subframe(unsigned int ct[][2], aom_prob oldp,
|
||||
aom_prob *bestp, aom_prob upd, int n) {
|
||||
const int old_b = get_cost(ct, oldp, n);
|
||||
int bestsavings = 0;
|
||||
const int upd_cost = av1_cost_one(upd) - av1_cost_zero(upd);
|
||||
aom_prob newp, bestnewp = oldp;
|
||||
const int step = *bestp > oldp ? -1 : 1;
|
||||
|
||||
for (newp = *bestp; newp != oldp; newp += step) {
|
||||
const int new_b = get_cost(ct, newp, n);
|
||||
const int update_b = prob_diff_update_cost(newp, oldp) + upd_cost;
|
||||
const int savings = old_b - new_b - update_b;
|
||||
if (savings > bestsavings) {
|
||||
bestsavings = savings;
|
||||
bestnewp = newp;
|
||||
}
|
||||
}
|
||||
*bestp = bestnewp;
|
||||
return bestsavings;
|
||||
}
|
||||
|
||||
int av1_prob_update_search_model_subframe(
|
||||
unsigned int ct[ENTROPY_NODES][COEF_PROBS_BUFS][2], const aom_prob *oldp,
|
||||
aom_prob *bestp, aom_prob upd, int stepsize, int n) {
|
||||
int i, old_b, new_b, update_b, savings, bestsavings;
|
||||
int newp;
|
||||
const int step_sign = *bestp > oldp[PIVOT_NODE] ? -1 : 1;
|
||||
const int step = stepsize * step_sign;
|
||||
const int upd_cost = av1_cost_one(upd) - av1_cost_zero(upd);
|
||||
aom_prob bestnewp, newplist[ENTROPY_NODES], oldplist[ENTROPY_NODES];
|
||||
av1_model_to_full_probs(oldp, oldplist);
|
||||
memcpy(newplist, oldp, sizeof(aom_prob) * UNCONSTRAINED_NODES);
|
||||
for (i = UNCONSTRAINED_NODES, old_b = 0; i < ENTROPY_NODES; ++i)
|
||||
old_b += get_cost(ct[i], oldplist[i], n);
|
||||
old_b += get_cost(ct[PIVOT_NODE], oldplist[PIVOT_NODE], n);
|
||||
|
||||
bestsavings = 0;
|
||||
bestnewp = oldp[PIVOT_NODE];
|
||||
|
||||
assert(stepsize > 0);
|
||||
|
||||
for (newp = *bestp; (newp - oldp[PIVOT_NODE]) * step_sign < 0; newp += step) {
|
||||
if (newp < 1 || newp > 255) continue;
|
||||
newplist[PIVOT_NODE] = newp;
|
||||
av1_model_to_full_probs(newplist, newplist);
|
||||
for (i = UNCONSTRAINED_NODES, new_b = 0; i < ENTROPY_NODES; ++i)
|
||||
new_b += get_cost(ct[i], newplist[i], n);
|
||||
new_b += get_cost(ct[PIVOT_NODE], newplist[PIVOT_NODE], n);
|
||||
update_b = prob_diff_update_cost(newp, oldp[PIVOT_NODE]) + upd_cost;
|
||||
savings = old_b - new_b - update_b;
|
||||
if (savings > bestsavings) {
|
||||
bestsavings = savings;
|
||||
bestnewp = newp;
|
||||
}
|
||||
}
|
||||
|
||||
*bestp = bestnewp;
|
||||
return bestsavings;
|
||||
}
|
||||
#endif // CONFIG_SUBFRAME_PROB_UPDATE
|
||||
|
||||
void av1_cond_prob_diff_update(aom_writer *w, aom_prob *oldp,
|
||||
const unsigned int ct[2], int probwt) {
|
||||
const aom_prob upd = DIFF_UPDATE_PROB;
|
||||
|
|
|
|||
7
third_party/aom/av1/encoder/subexp.h
vendored
7
third_party/aom/av1/encoder/subexp.h
vendored
|
|
@ -35,13 +35,6 @@ int av1_prob_diff_update_savings_search_model(const unsigned int *ct,
|
|||
|
||||
int av1_cond_prob_diff_update_savings(aom_prob *oldp, const unsigned int ct[2],
|
||||
int probwt);
|
||||
#if CONFIG_SUBFRAME_PROB_UPDATE
|
||||
int av1_prob_update_search_subframe(unsigned int ct[][2], aom_prob oldp,
|
||||
aom_prob *bestp, aom_prob upd, int n);
|
||||
int av1_prob_update_search_model_subframe(
|
||||
unsigned int ct[ENTROPY_NODES][COEF_PROBS_BUFS][2], const aom_prob *oldp,
|
||||
aom_prob *bestp, aom_prob upd, int stepsize, int n);
|
||||
#endif // CONFIG_SUBFRAME_PROB_UPDATE
|
||||
#ifdef __cplusplus
|
||||
} // extern "C"
|
||||
#endif
|
||||
|
|
|
|||
13
third_party/aom/av1/encoder/temporal_filter.c
vendored
13
third_party/aom/av1/encoder/temporal_filter.c
vendored
|
|
@ -281,14 +281,10 @@ static int temporal_filter_find_matching_mb_c(AV1_COMP *cpi,
|
|||
|
||||
av1_set_mv_search_range(&x->mv_limits, &best_ref_mv1);
|
||||
|
||||
#if CONFIG_REF_MV
|
||||
x->mvcost = x->mv_cost_stack[0];
|
||||
x->nmvjointcost = x->nmv_vec_cost[0];
|
||||
x->mvsadcost = x->mvcost;
|
||||
x->nmvjointsadcost = x->nmvjointcost;
|
||||
#endif
|
||||
|
||||
// Ignore mv costing by sending NULL pointer instead of cost arrays
|
||||
// Use mv costing from x->mvcost directly
|
||||
av1_hex_search(x, &best_ref_mv1_full, step_param, sadpb, 1,
|
||||
cond_cost_list(cpi, cost_list), &cpi->fn_ptr[BLOCK_16X16], 0,
|
||||
&best_ref_mv1);
|
||||
|
|
@ -299,8 +295,11 @@ static int temporal_filter_find_matching_mb_c(AV1_COMP *cpi,
|
|||
bestsme = cpi->find_fractional_mv_step(
|
||||
x, &best_ref_mv1, cpi->common.allow_high_precision_mv, x->errorperbit,
|
||||
&cpi->fn_ptr[BLOCK_16X16], 0, mv_sf->subpel_iters_per_step,
|
||||
cond_cost_list(cpi, cost_list), NULL, NULL, &distortion, &sse, NULL, 0, 0,
|
||||
0);
|
||||
cond_cost_list(cpi, cost_list), NULL, NULL, &distortion, &sse, NULL,
|
||||
#if CONFIG_EXT_INTER
|
||||
NULL, 0, 0,
|
||||
#endif
|
||||
0, 0, 0);
|
||||
|
||||
x->e_mbd.mi[0]->bmi[0].as_mv[0] = x->best_mv;
|
||||
|
||||
|
|
|
|||
123
third_party/aom/av1/encoder/tokenize.c
vendored
123
third_party/aom/av1/encoder/tokenize.c
vendored
|
|
@ -23,6 +23,9 @@
|
|||
|
||||
#include "av1/encoder/cost.h"
|
||||
#include "av1/encoder/encoder.h"
|
||||
#if CONFIG_LV_MAP
|
||||
#include "av1/encoder/encodetxb.c"
|
||||
#endif
|
||||
#include "av1/encoder/rdopt.h"
|
||||
#include "av1/encoder/tokenize.h"
|
||||
|
||||
|
|
@ -261,20 +264,6 @@ const av1_extra_bit av1_extra_bits[ENTROPY_TOKENS] = {
|
|||
};
|
||||
#endif
|
||||
|
||||
#if !CONFIG_EC_MULTISYMBOL
|
||||
const struct av1_token av1_coef_encodings[ENTROPY_TOKENS] = {
|
||||
{ 2, 2 }, { 6, 3 }, { 28, 5 }, { 58, 6 }, { 59, 6 }, { 60, 6 },
|
||||
{ 61, 6 }, { 124, 7 }, { 125, 7 }, { 126, 7 }, { 127, 7 }, { 0, 1 }
|
||||
};
|
||||
#endif // !CONFIG_EC_MULTISYMBOL
|
||||
|
||||
struct tokenize_b_args {
|
||||
const AV1_COMP *cpi;
|
||||
ThreadData *td;
|
||||
TOKENEXTRA **tp;
|
||||
int this_rate;
|
||||
};
|
||||
|
||||
#if !CONFIG_PVQ || CONFIG_VAR_TX
|
||||
static void cost_coeffs_b(int plane, int block, int blk_row, int blk_col,
|
||||
BLOCK_SIZE plane_bsize, TX_SIZE tx_size, void *arg) {
|
||||
|
|
@ -314,7 +303,6 @@ static void set_entropy_context_b(int plane, int block, int blk_row,
|
|||
blk_row);
|
||||
}
|
||||
|
||||
#if CONFIG_NEW_TOKENSET
|
||||
static INLINE void add_token(TOKENEXTRA **t,
|
||||
aom_cdf_prob (*tail_cdf)[CDF_SIZE(ENTROPY_TOKENS)],
|
||||
aom_cdf_prob (*head_cdf)[CDF_SIZE(ENTROPY_TOKENS)],
|
||||
|
|
@ -328,25 +316,6 @@ static INLINE void add_token(TOKENEXTRA **t,
|
|||
(*t)->first_val = first_val;
|
||||
(*t)++;
|
||||
}
|
||||
|
||||
#else // CONFIG_NEW_TOKENSET
|
||||
static INLINE void add_token(
|
||||
TOKENEXTRA **t, const aom_prob *context_tree,
|
||||
#if CONFIG_EC_MULTISYMBOL
|
||||
aom_cdf_prob (*token_cdf)[CDF_SIZE(ENTROPY_TOKENS)],
|
||||
#endif // CONFIG_EC_MULTISYMBOL
|
||||
int32_t extra, uint8_t token, uint8_t skip_eob_node, unsigned int *counts) {
|
||||
(*t)->token = token;
|
||||
(*t)->extra = extra;
|
||||
(*t)->context_tree = context_tree;
|
||||
#if CONFIG_EC_MULTISYMBOL
|
||||
(*t)->token_cdf = token_cdf;
|
||||
#endif // CONFIG_EC_MULTISYMBOL
|
||||
(*t)->skip_eob_node = skip_eob_node;
|
||||
(*t)++;
|
||||
++counts[token];
|
||||
}
|
||||
#endif // CONFIG_NEW_TOKENSET
|
||||
#endif // !CONFIG_PVQ || CONFIG_VAR_TX
|
||||
|
||||
#if CONFIG_PALETTE
|
||||
|
|
@ -471,22 +440,11 @@ static void tokenize_b(int plane, int block, int blk_row, int blk_col,
|
|||
const int ref = is_inter_block(mbmi);
|
||||
unsigned int(*const counts)[COEFF_CONTEXTS][ENTROPY_TOKENS] =
|
||||
td->rd_counts.coef_counts[txsize_sqr_map[tx_size]][type][ref];
|
||||
#if !CONFIG_NEW_TOKENSET
|
||||
#if CONFIG_SUBFRAME_PROB_UPDATE
|
||||
const aom_prob(*coef_probs)[COEFF_CONTEXTS][UNCONSTRAINED_NODES] =
|
||||
cpi->subframe_stats.coef_probs_buf[cpi->common.coef_probs_update_idx]
|
||||
[txsize_sqr_map[tx_size]][type][ref];
|
||||
#else
|
||||
aom_prob(*const coef_probs)[COEFF_CONTEXTS][UNCONSTRAINED_NODES] =
|
||||
cpi->common.fc->coef_probs[txsize_sqr_map[tx_size]][type][ref];
|
||||
#endif // CONFIG_SUBFRAME_PROB_UPDATE
|
||||
#endif // !CONFIG_NEW_TOKENSET
|
||||
#if CONFIG_EC_ADAPT
|
||||
FRAME_CONTEXT *ec_ctx = xd->tile_ctx;
|
||||
#elif CONFIG_EC_MULTISYMBOL
|
||||
#else
|
||||
FRAME_CONTEXT *ec_ctx = cpi->common.fc;
|
||||
#endif
|
||||
#if CONFIG_NEW_TOKENSET
|
||||
aom_cdf_prob(
|
||||
*const coef_head_cdfs)[COEFF_CONTEXTS][CDF_SIZE(ENTROPY_TOKENS)] =
|
||||
ec_ctx->coef_head_cdfs[txsize_sqr_map[tx_size]][type][ref];
|
||||
|
|
@ -497,13 +455,6 @@ static void tokenize_b(int plane, int block, int blk_row, int blk_col,
|
|||
td->counts->blockz_count[txsize_sqr_map[tx_size]][type][ref];
|
||||
int eob_val;
|
||||
int first_val = 1;
|
||||
#else
|
||||
#if CONFIG_EC_MULTISYMBOL
|
||||
aom_cdf_prob(*const coef_cdfs)[COEFF_CONTEXTS][CDF_SIZE(ENTROPY_TOKENS)] =
|
||||
ec_ctx->coef_cdfs[txsize_sqr_map[tx_size]][type][ref];
|
||||
#endif
|
||||
int skip_eob = 0;
|
||||
#endif
|
||||
const int seg_eob = get_tx_eob(&cpi->common.seg, segment_id, tx_size);
|
||||
unsigned int(*const eob_branch)[COEFF_CONTEXTS] =
|
||||
td->counts->eob_branch[txsize_sqr_map[tx_size]][type][ref];
|
||||
|
|
@ -517,7 +468,6 @@ static void tokenize_b(int plane, int block, int blk_row, int blk_col,
|
|||
nb = scan_order->neighbors;
|
||||
c = 0;
|
||||
|
||||
#if CONFIG_NEW_TOKENSET
|
||||
if (eob == 0)
|
||||
add_token(&t, &coef_tail_cdfs[band[c]][pt], &coef_head_cdfs[band[c]][pt], 1,
|
||||
1, 0, BLOCK_Z_TOKEN);
|
||||
|
|
@ -553,33 +503,6 @@ static void tokenize_b(int plane, int block, int blk_row, int blk_col,
|
|||
++c;
|
||||
pt = get_coef_context(nb, token_cache, AOMMIN(c, eob - 1));
|
||||
}
|
||||
#else
|
||||
while (c < eob) {
|
||||
const int v = qcoeff[scan[c]];
|
||||
eob_branch[band[c]][pt] += !skip_eob;
|
||||
|
||||
av1_get_token_extra(v, &token, &extra);
|
||||
|
||||
add_token(&t, coef_probs[band[c]][pt],
|
||||
#if CONFIG_EC_MULTISYMBOL
|
||||
&coef_cdfs[band[c]][pt],
|
||||
#endif
|
||||
extra, (uint8_t)token, (uint8_t)skip_eob, counts[band[c]][pt]);
|
||||
|
||||
token_cache[scan[c]] = av1_pt_energy_class[token];
|
||||
++c;
|
||||
pt = get_coef_context(nb, token_cache, c);
|
||||
skip_eob = (token == ZERO_TOKEN);
|
||||
}
|
||||
if (c < seg_eob) {
|
||||
add_token(&t, coef_probs[band[c]][pt],
|
||||
#if CONFIG_EC_MULTISYMBOL
|
||||
NULL,
|
||||
#endif
|
||||
0, EOB_TOKEN, 0, counts[band[c]][pt]);
|
||||
++eob_branch[band[c]][pt];
|
||||
}
|
||||
#endif // CONFIG_NEW_TOKENSET
|
||||
|
||||
#if CONFIG_COEF_INTERLEAVE
|
||||
t->token = EOSB_TOKEN;
|
||||
|
|
@ -651,6 +574,18 @@ void tokenize_vartx(ThreadData *td, TOKENEXTRA **t, RUN_TYPE dry_run,
|
|||
|
||||
if (tx_size == plane_tx_size) {
|
||||
plane_bsize = get_plane_block_size(mbmi->sb_type, pd);
|
||||
#if CONFIG_LV_MAP
|
||||
if (!dry_run) {
|
||||
av1_update_and_record_txb_context(plane, block, blk_row, blk_col,
|
||||
plane_bsize, tx_size, arg);
|
||||
} else if (dry_run == DRY_RUN_NORMAL) {
|
||||
av1_update_txb_context_b(plane, block, blk_row, blk_col, plane_bsize,
|
||||
tx_size, arg);
|
||||
} else {
|
||||
printf("DRY_RUN_COSTCOEFFS is not supported yet\n");
|
||||
assert(0);
|
||||
}
|
||||
#else
|
||||
if (!dry_run)
|
||||
tokenize_b(plane, block, blk_row, blk_col, plane_bsize, tx_size, arg);
|
||||
else if (dry_run == DRY_RUN_NORMAL)
|
||||
|
|
@ -658,6 +593,7 @@ void tokenize_vartx(ThreadData *td, TOKENEXTRA **t, RUN_TYPE dry_run,
|
|||
tx_size, arg);
|
||||
else if (dry_run == DRY_RUN_COSTCOEFFS)
|
||||
cost_coeffs_b(plane, block, blk_row, blk_col, plane_bsize, tx_size, arg);
|
||||
#endif
|
||||
} else {
|
||||
// Half the block size in transform block unit.
|
||||
const TX_SIZE sub_txs = sub_tx_size_map[tx_size];
|
||||
|
|
@ -688,7 +624,11 @@ void av1_tokenize_sb_vartx(const AV1_COMP *cpi, ThreadData *td, TOKENEXTRA **t,
|
|||
MACROBLOCK *const x = &td->mb;
|
||||
MACROBLOCKD *const xd = &x->e_mbd;
|
||||
MB_MODE_INFO *const mbmi = &xd->mi[0]->mbmi;
|
||||
#if CONFIG_LV_MAP
|
||||
(void)t;
|
||||
#else
|
||||
TOKENEXTRA *t_backup = *t;
|
||||
#endif
|
||||
const int ctx = av1_get_skip_context(xd);
|
||||
const int skip_inc =
|
||||
!segfeature_active(&cm->seg, mbmi->segment_id, SEG_LVL_SKIP);
|
||||
|
|
@ -698,22 +638,25 @@ void av1_tokenize_sb_vartx(const AV1_COMP *cpi, ThreadData *td, TOKENEXTRA **t,
|
|||
|
||||
if (mbmi->skip) {
|
||||
if (!dry_run) td->counts->skip[ctx][1] += skip_inc;
|
||||
reset_skip_context(xd, bsize);
|
||||
av1_reset_skip_context(xd, mi_row, mi_col, bsize);
|
||||
#if !CONFIG_LV_MAP
|
||||
if (dry_run) *t = t_backup;
|
||||
#endif
|
||||
return;
|
||||
}
|
||||
|
||||
if (!dry_run)
|
||||
td->counts->skip[ctx][0] += skip_inc;
|
||||
if (!dry_run) td->counts->skip[ctx][0] += skip_inc;
|
||||
#if !CONFIG_LV_MAP
|
||||
else
|
||||
*t = t_backup;
|
||||
#endif
|
||||
|
||||
for (plane = 0; plane < MAX_MB_PLANE; ++plane) {
|
||||
#if CONFIG_CB4X4
|
||||
if (!is_chroma_reference(mi_row, mi_col, bsize,
|
||||
xd->plane[plane].subsampling_x,
|
||||
xd->plane[plane].subsampling_y)) {
|
||||
#if !CONFIG_PVQ
|
||||
#if !CONFIG_PVQ || !CONFIG_LV_MAP
|
||||
if (!dry_run) {
|
||||
(*t)->token = EOSB_TOKEN;
|
||||
(*t)++;
|
||||
|
|
@ -746,10 +689,12 @@ void av1_tokenize_sb_vartx(const AV1_COMP *cpi, ThreadData *td, TOKENEXTRA **t,
|
|||
}
|
||||
}
|
||||
|
||||
#if !CONFIG_LV_MAP
|
||||
if (!dry_run) {
|
||||
(*t)->token = EOSB_TOKEN;
|
||||
(*t)++;
|
||||
}
|
||||
#endif
|
||||
}
|
||||
if (rate) *rate += arg.this_rate;
|
||||
}
|
||||
|
|
@ -768,7 +713,7 @@ void av1_tokenize_sb(const AV1_COMP *cpi, ThreadData *td, TOKENEXTRA **t,
|
|||
struct tokenize_b_args arg = { cpi, td, t, 0 };
|
||||
if (mbmi->skip) {
|
||||
if (!dry_run) td->counts->skip[ctx][1] += skip_inc;
|
||||
reset_skip_context(xd, bsize);
|
||||
av1_reset_skip_context(xd, mi_row, mi_col, bsize);
|
||||
return;
|
||||
}
|
||||
|
||||
|
|
@ -843,8 +788,8 @@ void av1_tokenize_sb(const AV1_COMP *cpi, ThreadData *td, TOKENEXTRA **t,
|
|||
|
||||
#if CONFIG_SUPERTX
|
||||
void av1_tokenize_sb_supertx(const AV1_COMP *cpi, ThreadData *td,
|
||||
TOKENEXTRA **t, RUN_TYPE dry_run, BLOCK_SIZE bsize,
|
||||
int *rate) {
|
||||
TOKENEXTRA **t, RUN_TYPE dry_run, int mi_row,
|
||||
int mi_col, BLOCK_SIZE bsize, int *rate) {
|
||||
const AV1_COMMON *const cm = &cpi->common;
|
||||
MACROBLOCKD *const xd = &td->mb.e_mbd;
|
||||
MB_MODE_INFO *const mbmi = &xd->mi[0]->mbmi;
|
||||
|
|
@ -855,7 +800,7 @@ void av1_tokenize_sb_supertx(const AV1_COMP *cpi, ThreadData *td,
|
|||
struct tokenize_b_args arg = { cpi, td, t, 0 };
|
||||
if (mbmi->skip) {
|
||||
if (!dry_run) td->counts->skip[ctx][1] += skip_inc;
|
||||
reset_skip_context(xd, bsize);
|
||||
av1_reset_skip_context(xd, mi_row, mi_col, bsize);
|
||||
if (dry_run) *t = t_backup;
|
||||
return;
|
||||
}
|
||||
|
|
|
|||
18
third_party/aom/av1/encoder/tokenize.h
vendored
18
third_party/aom/av1/encoder/tokenize.h
vendored
|
|
@ -35,14 +35,10 @@ typedef struct {
|
|||
} TOKENVALUE;
|
||||
|
||||
typedef struct {
|
||||
#if CONFIG_NEW_TOKENSET
|
||||
aom_cdf_prob (*tail_cdf)[CDF_SIZE(ENTROPY_TOKENS)];
|
||||
aom_cdf_prob (*head_cdf)[CDF_SIZE(ENTROPY_TOKENS)];
|
||||
int eob_val;
|
||||
int first_val;
|
||||
#elif CONFIG_EC_MULTISYMBOL
|
||||
aom_cdf_prob (*token_cdf)[CDF_SIZE(ENTROPY_TOKENS)];
|
||||
#endif
|
||||
const aom_prob *context_tree;
|
||||
EXTRABIT extra;
|
||||
uint8_t token;
|
||||
|
|
@ -51,15 +47,19 @@ typedef struct {
|
|||
|
||||
extern const aom_tree_index av1_coef_tree[];
|
||||
extern const aom_tree_index av1_coef_con_tree[];
|
||||
#if !CONFIG_EC_MULTISYMBOL
|
||||
extern const struct av1_token av1_coef_encodings[];
|
||||
#endif // !CONFIG_EC_MULTISYMBOL
|
||||
|
||||
int av1_is_skippable_in_plane(MACROBLOCK *x, BLOCK_SIZE bsize, int plane);
|
||||
|
||||
struct AV1_COMP;
|
||||
struct ThreadData;
|
||||
|
||||
struct tokenize_b_args {
|
||||
const struct AV1_COMP *cpi;
|
||||
struct ThreadData *td;
|
||||
TOKENEXTRA **tp;
|
||||
int this_rate;
|
||||
};
|
||||
|
||||
typedef enum {
|
||||
OUTPUT_ENABLED = 0,
|
||||
DRY_RUN_NORMAL,
|
||||
|
|
@ -85,8 +85,8 @@ void av1_tokenize_sb(const struct AV1_COMP *cpi, struct ThreadData *td,
|
|||
int *rate, const int mi_row, const int mi_col);
|
||||
#if CONFIG_SUPERTX
|
||||
void av1_tokenize_sb_supertx(const struct AV1_COMP *cpi, struct ThreadData *td,
|
||||
TOKENEXTRA **t, RUN_TYPE dry_run, BLOCK_SIZE bsize,
|
||||
int *rate);
|
||||
TOKENEXTRA **t, RUN_TYPE dry_run, int mi_row,
|
||||
int mi_col, BLOCK_SIZE bsize, int *rate);
|
||||
#endif
|
||||
|
||||
extern const int16_t *av1_dct_value_cost_ptr;
|
||||
|
|
|
|||
61
third_party/aom/av1/encoder/variance_tree.c
vendored
61
third_party/aom/av1/encoder/variance_tree.c
vendored
|
|
@ -1,61 +0,0 @@
|
|||
/*
|
||||
* Copyright (c) 2016, Alliance for Open Media. All rights reserved
|
||||
*
|
||||
* This source code is subject to the terms of the BSD 2 Clause License and
|
||||
* the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
|
||||
* was not distributed with this source code in the LICENSE file, you can
|
||||
* obtain it at www.aomedia.org/license/software. If the Alliance for Open
|
||||
* Media Patent License 1.0 was not distributed with this source code in the
|
||||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#include "av1/encoder/variance_tree.h"
|
||||
#include "av1/encoder/encoder.h"
|
||||
|
||||
void av1_setup_var_tree(struct AV1Common *cm, ThreadData *td) {
|
||||
int i, j;
|
||||
#if CONFIG_EXT_PARTITION
|
||||
const int leaf_nodes = 1024;
|
||||
const int tree_nodes = 1024 + 256 + 64 + 16 + 4 + 1;
|
||||
#else
|
||||
const int leaf_nodes = 256;
|
||||
const int tree_nodes = 256 + 64 + 16 + 4 + 1;
|
||||
#endif // CONFIG_EXT_PARTITION
|
||||
int index = 0;
|
||||
VAR_TREE *this_var;
|
||||
int nodes;
|
||||
|
||||
aom_free(td->var_tree);
|
||||
CHECK_MEM_ERROR(cm, td->var_tree,
|
||||
aom_calloc(tree_nodes, sizeof(*td->var_tree)));
|
||||
|
||||
this_var = &td->var_tree[0];
|
||||
|
||||
// Sets up all the leaf nodes in the tree.
|
||||
for (index = 0; index < leaf_nodes; ++index) {
|
||||
VAR_TREE *const leaf = &td->var_tree[index];
|
||||
leaf->split[0] = NULL;
|
||||
}
|
||||
|
||||
// Each node has 4 leaf nodes, fill in the child pointers
|
||||
// from leafs to the root.
|
||||
for (nodes = leaf_nodes >> 2; nodes > 0; nodes >>= 2) {
|
||||
for (i = 0; i < nodes; ++i, ++index) {
|
||||
VAR_TREE *const node = &td->var_tree[index];
|
||||
for (j = 0; j < 4; j++) node->split[j] = this_var++;
|
||||
}
|
||||
}
|
||||
|
||||
// Set up the root node for the largest superblock size
|
||||
i = MAX_MIB_SIZE_LOG2 - MIN_MIB_SIZE_LOG2;
|
||||
td->var_root[i] = &td->var_tree[tree_nodes - 1];
|
||||
// Set up the root nodes for the rest of the possible superblock sizes
|
||||
while (--i >= 0) {
|
||||
td->var_root[i] = td->var_root[i + 1]->split[0];
|
||||
}
|
||||
}
|
||||
|
||||
void av1_free_var_tree(ThreadData *td) {
|
||||
aom_free(td->var_tree);
|
||||
td->var_tree = NULL;
|
||||
}
|
||||
96
third_party/aom/av1/encoder/variance_tree.h
vendored
96
third_party/aom/av1/encoder/variance_tree.h
vendored
|
|
@ -1,96 +0,0 @@
|
|||
/*
|
||||
* Copyright (c) 2016, Alliance for Open Media. All rights reserved
|
||||
*
|
||||
* This source code is subject to the terms of the BSD 2 Clause License and
|
||||
* the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
|
||||
* was not distributed with this source code in the LICENSE file, you can
|
||||
* obtain it at www.aomedia.org/license/software. If the Alliance for Open
|
||||
* Media Patent License 1.0 was not distributed with this source code in the
|
||||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#ifndef AV1_ENCODER_VARIANCE_TREE_H_
|
||||
#define AV1_ENCODER_VARIANCE_TREE_H_
|
||||
|
||||
#include <assert.h>
|
||||
|
||||
#include "./aom_config.h"
|
||||
|
||||
#include "aom/aom_integer.h"
|
||||
|
||||
#include "av1/common/enums.h"
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
|
||||
struct AV1Common;
|
||||
struct ThreadData;
|
||||
|
||||
typedef struct {
|
||||
int64_t sum_square_error;
|
||||
int64_t sum_error;
|
||||
int log2_count;
|
||||
int variance;
|
||||
} VAR;
|
||||
|
||||
typedef struct {
|
||||
VAR none;
|
||||
VAR horz[2];
|
||||
VAR vert[2];
|
||||
} partition_variance;
|
||||
|
||||
typedef struct VAR_TREE {
|
||||
int force_split;
|
||||
partition_variance variances;
|
||||
struct VAR_TREE *split[4];
|
||||
BLOCK_SIZE bsize;
|
||||
const uint8_t *src;
|
||||
const uint8_t *ref;
|
||||
int src_stride;
|
||||
int ref_stride;
|
||||
int width;
|
||||
int height;
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
int highbd;
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
} VAR_TREE;
|
||||
|
||||
void av1_setup_var_tree(struct AV1Common *cm, struct ThreadData *td);
|
||||
void av1_free_var_tree(struct ThreadData *td);
|
||||
|
||||
// Set variance values given sum square error, sum error, count.
|
||||
static INLINE void fill_variance(int64_t s2, int64_t s, int c, VAR *v) {
|
||||
v->sum_square_error = s2;
|
||||
v->sum_error = s;
|
||||
v->log2_count = c;
|
||||
v->variance =
|
||||
(int)(256 * (v->sum_square_error -
|
||||
((v->sum_error * v->sum_error) >> v->log2_count)) >>
|
||||
v->log2_count);
|
||||
}
|
||||
|
||||
static INLINE void sum_2_variances(const VAR *a, const VAR *b, VAR *r) {
|
||||
assert(a->log2_count == b->log2_count);
|
||||
fill_variance(a->sum_square_error + b->sum_square_error,
|
||||
a->sum_error + b->sum_error, a->log2_count + 1, r);
|
||||
}
|
||||
|
||||
static INLINE void fill_variance_node(VAR_TREE *vt) {
|
||||
sum_2_variances(&vt->split[0]->variances.none, &vt->split[1]->variances.none,
|
||||
&vt->variances.horz[0]);
|
||||
sum_2_variances(&vt->split[2]->variances.none, &vt->split[3]->variances.none,
|
||||
&vt->variances.horz[1]);
|
||||
sum_2_variances(&vt->split[0]->variances.none, &vt->split[2]->variances.none,
|
||||
&vt->variances.vert[0]);
|
||||
sum_2_variances(&vt->split[1]->variances.none, &vt->split[3]->variances.none,
|
||||
&vt->variances.vert[1]);
|
||||
sum_2_variances(&vt->variances.vert[0], &vt->variances.vert[1],
|
||||
&vt->variances.none);
|
||||
}
|
||||
|
||||
#ifdef __cplusplus
|
||||
} // extern "C"
|
||||
#endif
|
||||
|
||||
#endif /* AV1_ENCODER_VARIANCE_TREE_H_ */
|
||||
|
|
@ -15,13 +15,65 @@
|
|||
#include "./av1_rtcd.h"
|
||||
#include "aom/aom_integer.h"
|
||||
|
||||
void av1_quantize_fp_sse2(const int16_t *coeff_ptr, intptr_t n_coeffs,
|
||||
static INLINE void read_coeff(const tran_low_t *coeff, intptr_t offset,
|
||||
__m128i *c0, __m128i *c1) {
|
||||
const tran_low_t *addr = coeff + offset;
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
const __m128i x0 = _mm_load_si128((const __m128i *)addr);
|
||||
const __m128i x1 = _mm_load_si128((const __m128i *)addr + 1);
|
||||
const __m128i x2 = _mm_load_si128((const __m128i *)addr + 2);
|
||||
const __m128i x3 = _mm_load_si128((const __m128i *)addr + 3);
|
||||
*c0 = _mm_packs_epi32(x0, x1);
|
||||
*c1 = _mm_packs_epi32(x2, x3);
|
||||
#else
|
||||
*c0 = _mm_load_si128((const __m128i *)addr);
|
||||
*c1 = _mm_load_si128((const __m128i *)addr + 1);
|
||||
#endif
|
||||
}
|
||||
|
||||
static INLINE void write_qcoeff(const __m128i *qc0, const __m128i *qc1,
|
||||
tran_low_t *qcoeff, intptr_t offset) {
|
||||
tran_low_t *addr = qcoeff + offset;
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
const __m128i zero = _mm_setzero_si128();
|
||||
__m128i sign_bits = _mm_cmplt_epi16(*qc0, zero);
|
||||
__m128i y0 = _mm_unpacklo_epi16(*qc0, sign_bits);
|
||||
__m128i y1 = _mm_unpackhi_epi16(*qc0, sign_bits);
|
||||
_mm_store_si128((__m128i *)addr, y0);
|
||||
_mm_store_si128((__m128i *)addr + 1, y1);
|
||||
|
||||
sign_bits = _mm_cmplt_epi16(*qc1, zero);
|
||||
y0 = _mm_unpacklo_epi16(*qc1, sign_bits);
|
||||
y1 = _mm_unpackhi_epi16(*qc1, sign_bits);
|
||||
_mm_store_si128((__m128i *)addr + 2, y0);
|
||||
_mm_store_si128((__m128i *)addr + 3, y1);
|
||||
#else
|
||||
_mm_store_si128((__m128i *)addr, *qc0);
|
||||
_mm_store_si128((__m128i *)addr + 1, *qc1);
|
||||
#endif
|
||||
}
|
||||
|
||||
static INLINE void write_zero(tran_low_t *qcoeff, intptr_t offset) {
|
||||
const __m128i zero = _mm_setzero_si128();
|
||||
tran_low_t *addr = qcoeff + offset;
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
_mm_store_si128((__m128i *)addr, zero);
|
||||
_mm_store_si128((__m128i *)addr + 1, zero);
|
||||
_mm_store_si128((__m128i *)addr + 2, zero);
|
||||
_mm_store_si128((__m128i *)addr + 3, zero);
|
||||
#else
|
||||
_mm_store_si128((__m128i *)addr, zero);
|
||||
_mm_store_si128((__m128i *)addr + 1, zero);
|
||||
#endif
|
||||
}
|
||||
|
||||
void av1_quantize_fp_sse2(const tran_low_t *coeff_ptr, intptr_t n_coeffs,
|
||||
int skip_block, const int16_t *zbin_ptr,
|
||||
const int16_t *round_ptr, const int16_t *quant_ptr,
|
||||
const int16_t *quant_shift_ptr, int16_t *qcoeff_ptr,
|
||||
int16_t *dqcoeff_ptr, const int16_t *dequant_ptr,
|
||||
uint16_t *eob_ptr, const int16_t *scan_ptr,
|
||||
const int16_t *iscan_ptr) {
|
||||
const int16_t *quant_shift_ptr,
|
||||
tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr,
|
||||
const int16_t *dequant_ptr, uint16_t *eob_ptr,
|
||||
const int16_t *scan_ptr, const int16_t *iscan_ptr) {
|
||||
__m128i zero;
|
||||
__m128i thr;
|
||||
int16_t nzflag;
|
||||
|
|
@ -54,8 +106,7 @@ void av1_quantize_fp_sse2(const int16_t *coeff_ptr, intptr_t n_coeffs,
|
|||
__m128i qcoeff0, qcoeff1;
|
||||
__m128i qtmp0, qtmp1;
|
||||
// Do DC and first 15 AC
|
||||
coeff0 = _mm_load_si128((const __m128i *)(coeff_ptr + n_coeffs));
|
||||
coeff1 = _mm_load_si128((const __m128i *)(coeff_ptr + n_coeffs) + 1);
|
||||
read_coeff(coeff_ptr, n_coeffs, &coeff0, &coeff1);
|
||||
|
||||
// Poor man's sign extract
|
||||
coeff0_sign = _mm_srai_epi16(coeff0, 15);
|
||||
|
|
@ -78,15 +129,13 @@ void av1_quantize_fp_sse2(const int16_t *coeff_ptr, intptr_t n_coeffs,
|
|||
qcoeff0 = _mm_sub_epi16(qcoeff0, coeff0_sign);
|
||||
qcoeff1 = _mm_sub_epi16(qcoeff1, coeff1_sign);
|
||||
|
||||
_mm_store_si128((__m128i *)(qcoeff_ptr + n_coeffs), qcoeff0);
|
||||
_mm_store_si128((__m128i *)(qcoeff_ptr + n_coeffs) + 1, qcoeff1);
|
||||
write_qcoeff(&qcoeff0, &qcoeff1, qcoeff_ptr, n_coeffs);
|
||||
|
||||
coeff0 = _mm_mullo_epi16(qcoeff0, dequant);
|
||||
dequant = _mm_unpackhi_epi64(dequant, dequant);
|
||||
coeff1 = _mm_mullo_epi16(qcoeff1, dequant);
|
||||
|
||||
_mm_store_si128((__m128i *)(dqcoeff_ptr + n_coeffs), coeff0);
|
||||
_mm_store_si128((__m128i *)(dqcoeff_ptr + n_coeffs) + 1, coeff1);
|
||||
write_qcoeff(&coeff0, &coeff1, dqcoeff_ptr, n_coeffs);
|
||||
}
|
||||
|
||||
{
|
||||
|
|
@ -121,8 +170,7 @@ void av1_quantize_fp_sse2(const int16_t *coeff_ptr, intptr_t n_coeffs,
|
|||
__m128i qcoeff0, qcoeff1;
|
||||
__m128i qtmp0, qtmp1;
|
||||
|
||||
coeff0 = _mm_load_si128((const __m128i *)(coeff_ptr + n_coeffs));
|
||||
coeff1 = _mm_load_si128((const __m128i *)(coeff_ptr + n_coeffs) + 1);
|
||||
read_coeff(coeff_ptr, n_coeffs, &coeff0, &coeff1);
|
||||
|
||||
// Poor man's sign extract
|
||||
coeff0_sign = _mm_srai_epi16(coeff0, 15);
|
||||
|
|
@ -147,20 +195,15 @@ void av1_quantize_fp_sse2(const int16_t *coeff_ptr, intptr_t n_coeffs,
|
|||
qcoeff0 = _mm_sub_epi16(qcoeff0, coeff0_sign);
|
||||
qcoeff1 = _mm_sub_epi16(qcoeff1, coeff1_sign);
|
||||
|
||||
_mm_store_si128((__m128i *)(qcoeff_ptr + n_coeffs), qcoeff0);
|
||||
_mm_store_si128((__m128i *)(qcoeff_ptr + n_coeffs) + 1, qcoeff1);
|
||||
write_qcoeff(&qcoeff0, &qcoeff1, qcoeff_ptr, n_coeffs);
|
||||
|
||||
coeff0 = _mm_mullo_epi16(qcoeff0, dequant);
|
||||
coeff1 = _mm_mullo_epi16(qcoeff1, dequant);
|
||||
|
||||
_mm_store_si128((__m128i *)(dqcoeff_ptr + n_coeffs), coeff0);
|
||||
_mm_store_si128((__m128i *)(dqcoeff_ptr + n_coeffs) + 1, coeff1);
|
||||
write_qcoeff(&coeff0, &coeff1, dqcoeff_ptr, n_coeffs);
|
||||
} else {
|
||||
_mm_store_si128((__m128i *)(qcoeff_ptr + n_coeffs), zero);
|
||||
_mm_store_si128((__m128i *)(qcoeff_ptr + n_coeffs) + 1, zero);
|
||||
|
||||
_mm_store_si128((__m128i *)(dqcoeff_ptr + n_coeffs), zero);
|
||||
_mm_store_si128((__m128i *)(dqcoeff_ptr + n_coeffs) + 1, zero);
|
||||
write_zero(qcoeff_ptr, n_coeffs);
|
||||
write_zero(dqcoeff_ptr, n_coeffs);
|
||||
}
|
||||
}
|
||||
|
||||
|
|
@ -200,10 +243,8 @@ void av1_quantize_fp_sse2(const int16_t *coeff_ptr, intptr_t n_coeffs,
|
|||
}
|
||||
} else {
|
||||
do {
|
||||
_mm_store_si128((__m128i *)(dqcoeff_ptr + n_coeffs), zero);
|
||||
_mm_store_si128((__m128i *)(dqcoeff_ptr + n_coeffs) + 1, zero);
|
||||
_mm_store_si128((__m128i *)(qcoeff_ptr + n_coeffs), zero);
|
||||
_mm_store_si128((__m128i *)(qcoeff_ptr + n_coeffs) + 1, zero);
|
||||
write_zero(dqcoeff_ptr, n_coeffs);
|
||||
write_zero(qcoeff_ptr, n_coeffs);
|
||||
n_coeffs += 8 * 2;
|
||||
} while (n_coeffs < 0);
|
||||
*eob_ptr = 0;
|
||||
|
|
|
|||
91
third_party/aom/av1/encoder/x86/corner_match_sse4.c
vendored
Normal file
91
third_party/aom/av1/encoder/x86/corner_match_sse4.c
vendored
Normal file
|
|
@ -0,0 +1,91 @@
|
|||
#include <stdlib.h>
|
||||
#include <memory.h>
|
||||
#include <math.h>
|
||||
#include <assert.h>
|
||||
|
||||
#include <smmintrin.h>
|
||||
|
||||
#include "./av1_rtcd.h"
|
||||
#include "aom_ports/mem.h"
|
||||
#include "av1/encoder/corner_match.h"
|
||||
|
||||
DECLARE_ALIGNED(16, static const uint8_t, byte_mask[16]) = {
|
||||
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 0, 0, 0
|
||||
};
|
||||
#if MATCH_SZ != 13
|
||||
#error "Need to change byte_mask in corner_match_sse4.c if MATCH_SZ != 13"
|
||||
#endif
|
||||
|
||||
/* Compute corr(im1, im2) * MATCH_SZ * stddev(im1), where the
|
||||
correlation/standard deviation are taken over MATCH_SZ by MATCH_SZ windows
|
||||
of each image, centered at (x1, y1) and (x2, y2) respectively.
|
||||
*/
|
||||
double compute_cross_correlation_sse4_1(unsigned char *im1, int stride1, int x1,
|
||||
int y1, unsigned char *im2, int stride2,
|
||||
int x2, int y2) {
|
||||
int i;
|
||||
// 2 16-bit partial sums in lanes 0, 4 (== 2 32-bit partial sums in lanes 0,
|
||||
// 2)
|
||||
__m128i sum1_vec = _mm_setzero_si128();
|
||||
__m128i sum2_vec = _mm_setzero_si128();
|
||||
// 4 32-bit partial sums of squares
|
||||
__m128i sumsq2_vec = _mm_setzero_si128();
|
||||
__m128i cross_vec = _mm_setzero_si128();
|
||||
|
||||
const __m128i mask = _mm_load_si128((__m128i *)byte_mask);
|
||||
const __m128i zero = _mm_setzero_si128();
|
||||
|
||||
im1 += (y1 - MATCH_SZ_BY2) * stride1 + (x1 - MATCH_SZ_BY2);
|
||||
im2 += (y2 - MATCH_SZ_BY2) * stride2 + (x2 - MATCH_SZ_BY2);
|
||||
|
||||
for (i = 0; i < MATCH_SZ; ++i) {
|
||||
const __m128i v1 =
|
||||
_mm_and_si128(_mm_loadu_si128((__m128i *)&im1[i * stride1]), mask);
|
||||
const __m128i v2 =
|
||||
_mm_and_si128(_mm_loadu_si128((__m128i *)&im2[i * stride2]), mask);
|
||||
|
||||
// Using the 'sad' intrinsic here is a bit faster than adding
|
||||
// v1_l + v1_r and v2_l + v2_r, plus it avoids the need for a 16->32 bit
|
||||
// conversion step later, for a net speedup of ~10%
|
||||
sum1_vec = _mm_add_epi16(sum1_vec, _mm_sad_epu8(v1, zero));
|
||||
sum2_vec = _mm_add_epi16(sum2_vec, _mm_sad_epu8(v2, zero));
|
||||
|
||||
const __m128i v1_l = _mm_cvtepu8_epi16(v1);
|
||||
const __m128i v1_r = _mm_cvtepu8_epi16(_mm_srli_si128(v1, 8));
|
||||
const __m128i v2_l = _mm_cvtepu8_epi16(v2);
|
||||
const __m128i v2_r = _mm_cvtepu8_epi16(_mm_srli_si128(v2, 8));
|
||||
|
||||
sumsq2_vec = _mm_add_epi32(
|
||||
sumsq2_vec,
|
||||
_mm_add_epi32(_mm_madd_epi16(v2_l, v2_l), _mm_madd_epi16(v2_r, v2_r)));
|
||||
cross_vec = _mm_add_epi32(
|
||||
cross_vec,
|
||||
_mm_add_epi32(_mm_madd_epi16(v1_l, v2_l), _mm_madd_epi16(v1_r, v2_r)));
|
||||
}
|
||||
|
||||
// Now we can treat the four registers (sum1_vec, sum2_vec, sumsq2_vec,
|
||||
// cross_vec)
|
||||
// as holding 4 32-bit elements each, which we want to sum horizontally.
|
||||
// We do this by transposing and then summing vertically.
|
||||
__m128i tmp_0 = _mm_unpacklo_epi32(sum1_vec, sum2_vec);
|
||||
__m128i tmp_1 = _mm_unpackhi_epi32(sum1_vec, sum2_vec);
|
||||
__m128i tmp_2 = _mm_unpacklo_epi32(sumsq2_vec, cross_vec);
|
||||
__m128i tmp_3 = _mm_unpackhi_epi32(sumsq2_vec, cross_vec);
|
||||
|
||||
__m128i tmp_4 = _mm_unpacklo_epi64(tmp_0, tmp_2);
|
||||
__m128i tmp_5 = _mm_unpackhi_epi64(tmp_0, tmp_2);
|
||||
__m128i tmp_6 = _mm_unpacklo_epi64(tmp_1, tmp_3);
|
||||
__m128i tmp_7 = _mm_unpackhi_epi64(tmp_1, tmp_3);
|
||||
|
||||
__m128i res =
|
||||
_mm_add_epi32(_mm_add_epi32(tmp_4, tmp_5), _mm_add_epi32(tmp_6, tmp_7));
|
||||
|
||||
int sum1 = _mm_extract_epi32(res, 0);
|
||||
int sum2 = _mm_extract_epi32(res, 1);
|
||||
int sumsq2 = _mm_extract_epi32(res, 2);
|
||||
int cross = _mm_extract_epi32(res, 3);
|
||||
|
||||
int var2 = sumsq2 * MATCH_SZ_SQ - sum2 * sum2;
|
||||
int cov = cross * MATCH_SZ_SQ - sum1 * sum2;
|
||||
return cov / sqrt((double)var2);
|
||||
}
|
||||
|
|
@ -13,7 +13,7 @@
|
|||
|
||||
#include "./av1_rtcd.h"
|
||||
#include "./aom_config.h"
|
||||
#include "av1/common/av1_fwd_txfm2d_cfg.h"
|
||||
#include "av1/common/av1_fwd_txfm1d_cfg.h"
|
||||
#include "av1/common/av1_txfm.h"
|
||||
#include "av1/common/x86/highbd_txfm_utility_sse4.h"
|
||||
#include "aom_dsp/txfm_common.h"
|
||||
|
|
@ -58,7 +58,7 @@ static INLINE void load_buffer_4x4(const int16_t *input, __m128i *in,
|
|||
// shift[1] is used in txfm_func_col()
|
||||
// shift[2] is used in txfm_func_row()
|
||||
static void fdct4x4_sse4_1(__m128i *in, int bit) {
|
||||
const int32_t *cospi = cospi_arr[bit - cos_bit_min];
|
||||
const int32_t *cospi = cospi_arr(bit);
|
||||
const __m128i cospi32 = _mm_set1_epi32(cospi[32]);
|
||||
const __m128i cospi48 = _mm_set1_epi32(cospi[48]);
|
||||
const __m128i cospi16 = _mm_set1_epi32(cospi[16]);
|
||||
|
|
@ -133,7 +133,7 @@ void av1_highbd_fht4x4_sse4_1(const int16_t *input, tran_low_t *output,
|
|||
}
|
||||
|
||||
static void fadst4x4_sse4_1(__m128i *in, int bit) {
|
||||
const int32_t *cospi = cospi_arr[bit - cos_bit_min];
|
||||
const int32_t *cospi = cospi_arr(bit);
|
||||
const __m128i cospi8 = _mm_set1_epi32(cospi[8]);
|
||||
const __m128i cospi56 = _mm_set1_epi32(cospi[56]);
|
||||
const __m128i cospi40 = _mm_set1_epi32(cospi[40]);
|
||||
|
|
@ -209,71 +209,81 @@ static void fadst4x4_sse4_1(__m128i *in, int bit) {
|
|||
void av1_fwd_txfm2d_4x4_sse4_1(const int16_t *input, int32_t *coeff,
|
||||
int input_stride, int tx_type, int bd) {
|
||||
__m128i in[4];
|
||||
const TXFM_2D_CFG *cfg = NULL;
|
||||
const TXFM_1D_CFG *row_cfg = NULL;
|
||||
const TXFM_1D_CFG *col_cfg = NULL;
|
||||
|
||||
switch (tx_type) {
|
||||
case DCT_DCT:
|
||||
cfg = &fwd_txfm_2d_cfg_dct_dct_4;
|
||||
load_buffer_4x4(input, in, input_stride, 0, 0, cfg->shift[0]);
|
||||
fdct4x4_sse4_1(in, cfg->cos_bit_col[2]);
|
||||
fdct4x4_sse4_1(in, cfg->cos_bit_row[2]);
|
||||
row_cfg = &fwd_txfm_1d_row_cfg_dct_4;
|
||||
col_cfg = &fwd_txfm_1d_col_cfg_dct_4;
|
||||
load_buffer_4x4(input, in, input_stride, 0, 0, row_cfg->shift[0]);
|
||||
fdct4x4_sse4_1(in, col_cfg->cos_bit[2]);
|
||||
fdct4x4_sse4_1(in, row_cfg->cos_bit[2]);
|
||||
write_buffer_4x4(in, coeff);
|
||||
break;
|
||||
case ADST_DCT:
|
||||
cfg = &fwd_txfm_2d_cfg_adst_dct_4;
|
||||
load_buffer_4x4(input, in, input_stride, 0, 0, cfg->shift[0]);
|
||||
fadst4x4_sse4_1(in, cfg->cos_bit_col[2]);
|
||||
fdct4x4_sse4_1(in, cfg->cos_bit_row[2]);
|
||||
row_cfg = &fwd_txfm_1d_row_cfg_dct_4;
|
||||
col_cfg = &fwd_txfm_1d_col_cfg_adst_4;
|
||||
load_buffer_4x4(input, in, input_stride, 0, 0, row_cfg->shift[0]);
|
||||
fadst4x4_sse4_1(in, col_cfg->cos_bit[2]);
|
||||
fdct4x4_sse4_1(in, row_cfg->cos_bit[2]);
|
||||
write_buffer_4x4(in, coeff);
|
||||
break;
|
||||
case DCT_ADST:
|
||||
cfg = &fwd_txfm_2d_cfg_dct_adst_4;
|
||||
load_buffer_4x4(input, in, input_stride, 0, 0, cfg->shift[0]);
|
||||
fdct4x4_sse4_1(in, cfg->cos_bit_col[2]);
|
||||
fadst4x4_sse4_1(in, cfg->cos_bit_row[2]);
|
||||
row_cfg = &fwd_txfm_1d_row_cfg_adst_4;
|
||||
col_cfg = &fwd_txfm_1d_col_cfg_dct_4;
|
||||
load_buffer_4x4(input, in, input_stride, 0, 0, row_cfg->shift[0]);
|
||||
fdct4x4_sse4_1(in, col_cfg->cos_bit[2]);
|
||||
fadst4x4_sse4_1(in, row_cfg->cos_bit[2]);
|
||||
write_buffer_4x4(in, coeff);
|
||||
break;
|
||||
case ADST_ADST:
|
||||
cfg = &fwd_txfm_2d_cfg_adst_adst_4;
|
||||
load_buffer_4x4(input, in, input_stride, 0, 0, cfg->shift[0]);
|
||||
fadst4x4_sse4_1(in, cfg->cos_bit_col[2]);
|
||||
fadst4x4_sse4_1(in, cfg->cos_bit_row[2]);
|
||||
row_cfg = &fwd_txfm_1d_row_cfg_adst_4;
|
||||
col_cfg = &fwd_txfm_1d_col_cfg_adst_4;
|
||||
load_buffer_4x4(input, in, input_stride, 0, 0, row_cfg->shift[0]);
|
||||
fadst4x4_sse4_1(in, col_cfg->cos_bit[2]);
|
||||
fadst4x4_sse4_1(in, row_cfg->cos_bit[2]);
|
||||
write_buffer_4x4(in, coeff);
|
||||
break;
|
||||
#if CONFIG_EXT_TX
|
||||
case FLIPADST_DCT:
|
||||
cfg = &fwd_txfm_2d_cfg_adst_dct_4;
|
||||
load_buffer_4x4(input, in, input_stride, 1, 0, cfg->shift[0]);
|
||||
fadst4x4_sse4_1(in, cfg->cos_bit_col[2]);
|
||||
fdct4x4_sse4_1(in, cfg->cos_bit_row[2]);
|
||||
row_cfg = &fwd_txfm_1d_row_cfg_dct_4;
|
||||
col_cfg = &fwd_txfm_1d_col_cfg_adst_4;
|
||||
load_buffer_4x4(input, in, input_stride, 1, 0, row_cfg->shift[0]);
|
||||
fadst4x4_sse4_1(in, col_cfg->cos_bit[2]);
|
||||
fdct4x4_sse4_1(in, row_cfg->cos_bit[2]);
|
||||
write_buffer_4x4(in, coeff);
|
||||
break;
|
||||
case DCT_FLIPADST:
|
||||
cfg = &fwd_txfm_2d_cfg_dct_adst_4;
|
||||
load_buffer_4x4(input, in, input_stride, 0, 1, cfg->shift[0]);
|
||||
fdct4x4_sse4_1(in, cfg->cos_bit_col[2]);
|
||||
fadst4x4_sse4_1(in, cfg->cos_bit_row[2]);
|
||||
row_cfg = &fwd_txfm_1d_row_cfg_adst_4;
|
||||
col_cfg = &fwd_txfm_1d_col_cfg_dct_4;
|
||||
load_buffer_4x4(input, in, input_stride, 0, 1, row_cfg->shift[0]);
|
||||
fdct4x4_sse4_1(in, col_cfg->cos_bit[2]);
|
||||
fadst4x4_sse4_1(in, row_cfg->cos_bit[2]);
|
||||
write_buffer_4x4(in, coeff);
|
||||
break;
|
||||
case FLIPADST_FLIPADST:
|
||||
cfg = &fwd_txfm_2d_cfg_adst_adst_4;
|
||||
load_buffer_4x4(input, in, input_stride, 1, 1, cfg->shift[0]);
|
||||
fadst4x4_sse4_1(in, cfg->cos_bit_col[2]);
|
||||
fadst4x4_sse4_1(in, cfg->cos_bit_row[2]);
|
||||
row_cfg = &fwd_txfm_1d_row_cfg_adst_4;
|
||||
col_cfg = &fwd_txfm_1d_col_cfg_adst_4;
|
||||
load_buffer_4x4(input, in, input_stride, 1, 1, row_cfg->shift[0]);
|
||||
fadst4x4_sse4_1(in, col_cfg->cos_bit[2]);
|
||||
fadst4x4_sse4_1(in, row_cfg->cos_bit[2]);
|
||||
write_buffer_4x4(in, coeff);
|
||||
break;
|
||||
case ADST_FLIPADST:
|
||||
cfg = &fwd_txfm_2d_cfg_adst_adst_4;
|
||||
load_buffer_4x4(input, in, input_stride, 0, 1, cfg->shift[0]);
|
||||
fadst4x4_sse4_1(in, cfg->cos_bit_col[2]);
|
||||
fadst4x4_sse4_1(in, cfg->cos_bit_row[2]);
|
||||
row_cfg = &fwd_txfm_1d_row_cfg_adst_4;
|
||||
col_cfg = &fwd_txfm_1d_col_cfg_adst_4;
|
||||
load_buffer_4x4(input, in, input_stride, 0, 1, row_cfg->shift[0]);
|
||||
fadst4x4_sse4_1(in, col_cfg->cos_bit[2]);
|
||||
fadst4x4_sse4_1(in, row_cfg->cos_bit[2]);
|
||||
write_buffer_4x4(in, coeff);
|
||||
break;
|
||||
case FLIPADST_ADST:
|
||||
cfg = &fwd_txfm_2d_cfg_adst_adst_4;
|
||||
load_buffer_4x4(input, in, input_stride, 1, 0, cfg->shift[0]);
|
||||
fadst4x4_sse4_1(in, cfg->cos_bit_col[2]);
|
||||
fadst4x4_sse4_1(in, cfg->cos_bit_row[2]);
|
||||
row_cfg = &fwd_txfm_1d_row_cfg_adst_4;
|
||||
col_cfg = &fwd_txfm_1d_col_cfg_adst_4;
|
||||
load_buffer_4x4(input, in, input_stride, 1, 0, row_cfg->shift[0]);
|
||||
fadst4x4_sse4_1(in, col_cfg->cos_bit[2]);
|
||||
fadst4x4_sse4_1(in, row_cfg->cos_bit[2]);
|
||||
write_buffer_4x4(in, coeff);
|
||||
break;
|
||||
#endif
|
||||
|
|
@ -429,7 +439,7 @@ static INLINE void write_buffer_8x8(const __m128i *res, tran_low_t *output) {
|
|||
}
|
||||
|
||||
static void fdct8x8_sse4_1(__m128i *in, __m128i *out, int bit) {
|
||||
const int32_t *cospi = cospi_arr[bit - cos_bit_min];
|
||||
const int32_t *cospi = cospi_arr(bit);
|
||||
const __m128i cospi32 = _mm_set1_epi32(cospi[32]);
|
||||
const __m128i cospim32 = _mm_set1_epi32(-cospi[32]);
|
||||
const __m128i cospi48 = _mm_set1_epi32(cospi[48]);
|
||||
|
|
@ -625,7 +635,7 @@ static void fdct8x8_sse4_1(__m128i *in, __m128i *out, int bit) {
|
|||
}
|
||||
|
||||
static void fadst8x8_sse4_1(__m128i *in, __m128i *out, int bit) {
|
||||
const int32_t *cospi = cospi_arr[bit - cos_bit_min];
|
||||
const int32_t *cospi = cospi_arr(bit);
|
||||
const __m128i cospi4 = _mm_set1_epi32(cospi[4]);
|
||||
const __m128i cospi60 = _mm_set1_epi32(cospi[60]);
|
||||
const __m128i cospi20 = _mm_set1_epi32(cospi[20]);
|
||||
|
|
@ -930,97 +940,107 @@ static void fadst8x8_sse4_1(__m128i *in, __m128i *out, int bit) {
|
|||
void av1_fwd_txfm2d_8x8_sse4_1(const int16_t *input, int32_t *coeff, int stride,
|
||||
int tx_type, int bd) {
|
||||
__m128i in[16], out[16];
|
||||
const TXFM_2D_CFG *cfg = NULL;
|
||||
const TXFM_1D_CFG *row_cfg = NULL;
|
||||
const TXFM_1D_CFG *col_cfg = NULL;
|
||||
|
||||
switch (tx_type) {
|
||||
case DCT_DCT:
|
||||
cfg = &fwd_txfm_2d_cfg_dct_dct_8;
|
||||
load_buffer_8x8(input, in, stride, 0, 0, cfg->shift[0]);
|
||||
fdct8x8_sse4_1(in, out, cfg->cos_bit_col[2]);
|
||||
col_txfm_8x8_rounding(out, -cfg->shift[1]);
|
||||
row_cfg = &fwd_txfm_1d_row_cfg_dct_8;
|
||||
col_cfg = &fwd_txfm_1d_col_cfg_dct_8;
|
||||
load_buffer_8x8(input, in, stride, 0, 0, row_cfg->shift[0]);
|
||||
fdct8x8_sse4_1(in, out, col_cfg->cos_bit[2]);
|
||||
col_txfm_8x8_rounding(out, -row_cfg->shift[1]);
|
||||
transpose_8x8(out, in);
|
||||
fdct8x8_sse4_1(in, out, cfg->cos_bit_row[2]);
|
||||
fdct8x8_sse4_1(in, out, row_cfg->cos_bit[2]);
|
||||
transpose_8x8(out, in);
|
||||
write_buffer_8x8(in, coeff);
|
||||
break;
|
||||
case ADST_DCT:
|
||||
cfg = &fwd_txfm_2d_cfg_adst_dct_8;
|
||||
load_buffer_8x8(input, in, stride, 0, 0, cfg->shift[0]);
|
||||
fadst8x8_sse4_1(in, out, cfg->cos_bit_col[2]);
|
||||
col_txfm_8x8_rounding(out, -cfg->shift[1]);
|
||||
row_cfg = &fwd_txfm_1d_row_cfg_dct_8;
|
||||
col_cfg = &fwd_txfm_1d_col_cfg_adst_8;
|
||||
load_buffer_8x8(input, in, stride, 0, 0, row_cfg->shift[0]);
|
||||
fadst8x8_sse4_1(in, out, col_cfg->cos_bit[2]);
|
||||
col_txfm_8x8_rounding(out, -row_cfg->shift[1]);
|
||||
transpose_8x8(out, in);
|
||||
fdct8x8_sse4_1(in, out, cfg->cos_bit_row[2]);
|
||||
fdct8x8_sse4_1(in, out, row_cfg->cos_bit[2]);
|
||||
transpose_8x8(out, in);
|
||||
write_buffer_8x8(in, coeff);
|
||||
break;
|
||||
case DCT_ADST:
|
||||
cfg = &fwd_txfm_2d_cfg_dct_adst_8;
|
||||
load_buffer_8x8(input, in, stride, 0, 0, cfg->shift[0]);
|
||||
fdct8x8_sse4_1(in, out, cfg->cos_bit_col[2]);
|
||||
col_txfm_8x8_rounding(out, -cfg->shift[1]);
|
||||
row_cfg = &fwd_txfm_1d_row_cfg_adst_8;
|
||||
col_cfg = &fwd_txfm_1d_col_cfg_dct_8;
|
||||
load_buffer_8x8(input, in, stride, 0, 0, row_cfg->shift[0]);
|
||||
fdct8x8_sse4_1(in, out, col_cfg->cos_bit[2]);
|
||||
col_txfm_8x8_rounding(out, -row_cfg->shift[1]);
|
||||
transpose_8x8(out, in);
|
||||
fadst8x8_sse4_1(in, out, cfg->cos_bit_row[2]);
|
||||
fadst8x8_sse4_1(in, out, row_cfg->cos_bit[2]);
|
||||
transpose_8x8(out, in);
|
||||
write_buffer_8x8(in, coeff);
|
||||
break;
|
||||
case ADST_ADST:
|
||||
cfg = &fwd_txfm_2d_cfg_adst_adst_8;
|
||||
load_buffer_8x8(input, in, stride, 0, 0, cfg->shift[0]);
|
||||
fadst8x8_sse4_1(in, out, cfg->cos_bit_col[2]);
|
||||
col_txfm_8x8_rounding(out, -cfg->shift[1]);
|
||||
row_cfg = &fwd_txfm_1d_row_cfg_adst_8;
|
||||
col_cfg = &fwd_txfm_1d_col_cfg_adst_8;
|
||||
load_buffer_8x8(input, in, stride, 0, 0, row_cfg->shift[0]);
|
||||
fadst8x8_sse4_1(in, out, col_cfg->cos_bit[2]);
|
||||
col_txfm_8x8_rounding(out, -row_cfg->shift[1]);
|
||||
transpose_8x8(out, in);
|
||||
fadst8x8_sse4_1(in, out, cfg->cos_bit_row[2]);
|
||||
fadst8x8_sse4_1(in, out, row_cfg->cos_bit[2]);
|
||||
transpose_8x8(out, in);
|
||||
write_buffer_8x8(in, coeff);
|
||||
break;
|
||||
#if CONFIG_EXT_TX
|
||||
case FLIPADST_DCT:
|
||||
cfg = &fwd_txfm_2d_cfg_adst_dct_8;
|
||||
load_buffer_8x8(input, in, stride, 1, 0, cfg->shift[0]);
|
||||
fadst8x8_sse4_1(in, out, cfg->cos_bit_col[2]);
|
||||
col_txfm_8x8_rounding(out, -cfg->shift[1]);
|
||||
row_cfg = &fwd_txfm_1d_row_cfg_dct_8;
|
||||
col_cfg = &fwd_txfm_1d_col_cfg_adst_8;
|
||||
load_buffer_8x8(input, in, stride, 1, 0, row_cfg->shift[0]);
|
||||
fadst8x8_sse4_1(in, out, col_cfg->cos_bit[2]);
|
||||
col_txfm_8x8_rounding(out, -row_cfg->shift[1]);
|
||||
transpose_8x8(out, in);
|
||||
fdct8x8_sse4_1(in, out, cfg->cos_bit_row[2]);
|
||||
fdct8x8_sse4_1(in, out, row_cfg->cos_bit[2]);
|
||||
transpose_8x8(out, in);
|
||||
write_buffer_8x8(in, coeff);
|
||||
break;
|
||||
case DCT_FLIPADST:
|
||||
cfg = &fwd_txfm_2d_cfg_dct_adst_8;
|
||||
load_buffer_8x8(input, in, stride, 0, 1, cfg->shift[0]);
|
||||
fdct8x8_sse4_1(in, out, cfg->cos_bit_col[2]);
|
||||
col_txfm_8x8_rounding(out, -cfg->shift[1]);
|
||||
row_cfg = &fwd_txfm_1d_row_cfg_adst_8;
|
||||
col_cfg = &fwd_txfm_1d_col_cfg_dct_8;
|
||||
load_buffer_8x8(input, in, stride, 0, 1, row_cfg->shift[0]);
|
||||
fdct8x8_sse4_1(in, out, col_cfg->cos_bit[2]);
|
||||
col_txfm_8x8_rounding(out, -row_cfg->shift[1]);
|
||||
transpose_8x8(out, in);
|
||||
fadst8x8_sse4_1(in, out, cfg->cos_bit_row[2]);
|
||||
fadst8x8_sse4_1(in, out, row_cfg->cos_bit[2]);
|
||||
transpose_8x8(out, in);
|
||||
write_buffer_8x8(in, coeff);
|
||||
break;
|
||||
case FLIPADST_FLIPADST:
|
||||
cfg = &fwd_txfm_2d_cfg_adst_adst_8;
|
||||
load_buffer_8x8(input, in, stride, 1, 1, cfg->shift[0]);
|
||||
fadst8x8_sse4_1(in, out, cfg->cos_bit_col[2]);
|
||||
col_txfm_8x8_rounding(out, -cfg->shift[1]);
|
||||
row_cfg = &fwd_txfm_1d_row_cfg_adst_8;
|
||||
col_cfg = &fwd_txfm_1d_col_cfg_adst_8;
|
||||
load_buffer_8x8(input, in, stride, 1, 1, row_cfg->shift[0]);
|
||||
fadst8x8_sse4_1(in, out, col_cfg->cos_bit[2]);
|
||||
col_txfm_8x8_rounding(out, -row_cfg->shift[1]);
|
||||
transpose_8x8(out, in);
|
||||
fadst8x8_sse4_1(in, out, cfg->cos_bit_row[2]);
|
||||
fadst8x8_sse4_1(in, out, row_cfg->cos_bit[2]);
|
||||
transpose_8x8(out, in);
|
||||
write_buffer_8x8(in, coeff);
|
||||
break;
|
||||
case ADST_FLIPADST:
|
||||
cfg = &fwd_txfm_2d_cfg_adst_adst_8;
|
||||
load_buffer_8x8(input, in, stride, 0, 1, cfg->shift[0]);
|
||||
fadst8x8_sse4_1(in, out, cfg->cos_bit_col[2]);
|
||||
col_txfm_8x8_rounding(out, -cfg->shift[1]);
|
||||
row_cfg = &fwd_txfm_1d_row_cfg_adst_8;
|
||||
col_cfg = &fwd_txfm_1d_col_cfg_adst_8;
|
||||
load_buffer_8x8(input, in, stride, 0, 1, row_cfg->shift[0]);
|
||||
fadst8x8_sse4_1(in, out, col_cfg->cos_bit[2]);
|
||||
col_txfm_8x8_rounding(out, -row_cfg->shift[1]);
|
||||
transpose_8x8(out, in);
|
||||
fadst8x8_sse4_1(in, out, cfg->cos_bit_row[2]);
|
||||
fadst8x8_sse4_1(in, out, row_cfg->cos_bit[2]);
|
||||
transpose_8x8(out, in);
|
||||
write_buffer_8x8(in, coeff);
|
||||
break;
|
||||
case FLIPADST_ADST:
|
||||
cfg = &fwd_txfm_2d_cfg_adst_adst_8;
|
||||
load_buffer_8x8(input, in, stride, 1, 0, cfg->shift[0]);
|
||||
fadst8x8_sse4_1(in, out, cfg->cos_bit_col[2]);
|
||||
col_txfm_8x8_rounding(out, -cfg->shift[1]);
|
||||
row_cfg = &fwd_txfm_1d_row_cfg_adst_8;
|
||||
col_cfg = &fwd_txfm_1d_col_cfg_adst_8;
|
||||
load_buffer_8x8(input, in, stride, 1, 0, row_cfg->shift[0]);
|
||||
fadst8x8_sse4_1(in, out, col_cfg->cos_bit[2]);
|
||||
col_txfm_8x8_rounding(out, -row_cfg->shift[1]);
|
||||
transpose_8x8(out, in);
|
||||
fadst8x8_sse4_1(in, out, cfg->cos_bit_row[2]);
|
||||
fadst8x8_sse4_1(in, out, row_cfg->cos_bit[2]);
|
||||
transpose_8x8(out, in);
|
||||
write_buffer_8x8(in, coeff);
|
||||
break;
|
||||
|
|
@ -1107,7 +1127,7 @@ static INLINE void load_buffer_16x16(const int16_t *input, __m128i *out,
|
|||
}
|
||||
|
||||
static void fdct16x16_sse4_1(__m128i *in, __m128i *out, int bit) {
|
||||
const int32_t *cospi = cospi_arr[bit - cos_bit_min];
|
||||
const int32_t *cospi = cospi_arr(bit);
|
||||
const __m128i cospi32 = _mm_set1_epi32(cospi[32]);
|
||||
const __m128i cospim32 = _mm_set1_epi32(-cospi[32]);
|
||||
const __m128i cospi48 = _mm_set1_epi32(cospi[48]);
|
||||
|
|
@ -1393,7 +1413,7 @@ static void fdct16x16_sse4_1(__m128i *in, __m128i *out, int bit) {
|
|||
}
|
||||
|
||||
static void fadst16x16_sse4_1(__m128i *in, __m128i *out, int bit) {
|
||||
const int32_t *cospi = cospi_arr[bit - cos_bit_min];
|
||||
const int32_t *cospi = cospi_arr(bit);
|
||||
const __m128i cospi2 = _mm_set1_epi32(cospi[2]);
|
||||
const __m128i cospi62 = _mm_set1_epi32(cospi[62]);
|
||||
const __m128i cospi10 = _mm_set1_epi32(cospi[10]);
|
||||
|
|
@ -1794,97 +1814,107 @@ static void write_buffer_16x16(const __m128i *in, tran_low_t *output) {
|
|||
void av1_fwd_txfm2d_16x16_sse4_1(const int16_t *input, int32_t *coeff,
|
||||
int stride, int tx_type, int bd) {
|
||||
__m128i in[64], out[64];
|
||||
const TXFM_2D_CFG *cfg = NULL;
|
||||
const TXFM_1D_CFG *row_cfg = NULL;
|
||||
const TXFM_1D_CFG *col_cfg = NULL;
|
||||
|
||||
switch (tx_type) {
|
||||
case DCT_DCT:
|
||||
cfg = &fwd_txfm_2d_cfg_dct_dct_16;
|
||||
load_buffer_16x16(input, in, stride, 0, 0, cfg->shift[0]);
|
||||
fdct16x16_sse4_1(in, out, cfg->cos_bit_col[0]);
|
||||
col_txfm_16x16_rounding(out, -cfg->shift[1]);
|
||||
row_cfg = &fwd_txfm_1d_row_cfg_dct_16;
|
||||
col_cfg = &fwd_txfm_1d_col_cfg_dct_16;
|
||||
load_buffer_16x16(input, in, stride, 0, 0, row_cfg->shift[0]);
|
||||
fdct16x16_sse4_1(in, out, col_cfg->cos_bit[0]);
|
||||
col_txfm_16x16_rounding(out, -row_cfg->shift[1]);
|
||||
transpose_16x16(out, in);
|
||||
fdct16x16_sse4_1(in, out, cfg->cos_bit_row[0]);
|
||||
fdct16x16_sse4_1(in, out, row_cfg->cos_bit[0]);
|
||||
transpose_16x16(out, in);
|
||||
write_buffer_16x16(in, coeff);
|
||||
break;
|
||||
case ADST_DCT:
|
||||
cfg = &fwd_txfm_2d_cfg_adst_dct_16;
|
||||
load_buffer_16x16(input, in, stride, 0, 0, cfg->shift[0]);
|
||||
fadst16x16_sse4_1(in, out, cfg->cos_bit_col[0]);
|
||||
col_txfm_16x16_rounding(out, -cfg->shift[1]);
|
||||
row_cfg = &fwd_txfm_1d_row_cfg_dct_16;
|
||||
col_cfg = &fwd_txfm_1d_col_cfg_adst_16;
|
||||
load_buffer_16x16(input, in, stride, 0, 0, row_cfg->shift[0]);
|
||||
fadst16x16_sse4_1(in, out, col_cfg->cos_bit[0]);
|
||||
col_txfm_16x16_rounding(out, -row_cfg->shift[1]);
|
||||
transpose_16x16(out, in);
|
||||
fdct16x16_sse4_1(in, out, cfg->cos_bit_row[0]);
|
||||
fdct16x16_sse4_1(in, out, row_cfg->cos_bit[0]);
|
||||
transpose_16x16(out, in);
|
||||
write_buffer_16x16(in, coeff);
|
||||
break;
|
||||
case DCT_ADST:
|
||||
cfg = &fwd_txfm_2d_cfg_dct_adst_16;
|
||||
load_buffer_16x16(input, in, stride, 0, 0, cfg->shift[0]);
|
||||
fdct16x16_sse4_1(in, out, cfg->cos_bit_col[0]);
|
||||
col_txfm_16x16_rounding(out, -cfg->shift[1]);
|
||||
row_cfg = &fwd_txfm_1d_row_cfg_adst_16;
|
||||
col_cfg = &fwd_txfm_1d_col_cfg_dct_16;
|
||||
load_buffer_16x16(input, in, stride, 0, 0, row_cfg->shift[0]);
|
||||
fdct16x16_sse4_1(in, out, col_cfg->cos_bit[0]);
|
||||
col_txfm_16x16_rounding(out, -row_cfg->shift[1]);
|
||||
transpose_16x16(out, in);
|
||||
fadst16x16_sse4_1(in, out, cfg->cos_bit_row[0]);
|
||||
fadst16x16_sse4_1(in, out, row_cfg->cos_bit[0]);
|
||||
transpose_16x16(out, in);
|
||||
write_buffer_16x16(in, coeff);
|
||||
break;
|
||||
case ADST_ADST:
|
||||
cfg = &fwd_txfm_2d_cfg_adst_adst_16;
|
||||
load_buffer_16x16(input, in, stride, 0, 0, cfg->shift[0]);
|
||||
fadst16x16_sse4_1(in, out, cfg->cos_bit_col[0]);
|
||||
col_txfm_16x16_rounding(out, -cfg->shift[1]);
|
||||
row_cfg = &fwd_txfm_1d_row_cfg_adst_16;
|
||||
col_cfg = &fwd_txfm_1d_col_cfg_adst_16;
|
||||
load_buffer_16x16(input, in, stride, 0, 0, row_cfg->shift[0]);
|
||||
fadst16x16_sse4_1(in, out, col_cfg->cos_bit[0]);
|
||||
col_txfm_16x16_rounding(out, -row_cfg->shift[1]);
|
||||
transpose_16x16(out, in);
|
||||
fadst16x16_sse4_1(in, out, cfg->cos_bit_row[0]);
|
||||
fadst16x16_sse4_1(in, out, row_cfg->cos_bit[0]);
|
||||
transpose_16x16(out, in);
|
||||
write_buffer_16x16(in, coeff);
|
||||
break;
|
||||
#if CONFIG_EXT_TX
|
||||
case FLIPADST_DCT:
|
||||
cfg = &fwd_txfm_2d_cfg_adst_dct_16;
|
||||
load_buffer_16x16(input, in, stride, 1, 0, cfg->shift[0]);
|
||||
fadst16x16_sse4_1(in, out, cfg->cos_bit_col[0]);
|
||||
col_txfm_16x16_rounding(out, -cfg->shift[1]);
|
||||
row_cfg = &fwd_txfm_1d_row_cfg_dct_16;
|
||||
col_cfg = &fwd_txfm_1d_col_cfg_adst_16;
|
||||
load_buffer_16x16(input, in, stride, 1, 0, row_cfg->shift[0]);
|
||||
fadst16x16_sse4_1(in, out, col_cfg->cos_bit[0]);
|
||||
col_txfm_16x16_rounding(out, -row_cfg->shift[1]);
|
||||
transpose_16x16(out, in);
|
||||
fdct16x16_sse4_1(in, out, cfg->cos_bit_row[0]);
|
||||
fdct16x16_sse4_1(in, out, row_cfg->cos_bit[0]);
|
||||
transpose_16x16(out, in);
|
||||
write_buffer_16x16(in, coeff);
|
||||
break;
|
||||
case DCT_FLIPADST:
|
||||
cfg = &fwd_txfm_2d_cfg_dct_adst_16;
|
||||
load_buffer_16x16(input, in, stride, 0, 1, cfg->shift[0]);
|
||||
fdct16x16_sse4_1(in, out, cfg->cos_bit_col[0]);
|
||||
col_txfm_16x16_rounding(out, -cfg->shift[1]);
|
||||
row_cfg = &fwd_txfm_1d_row_cfg_adst_16;
|
||||
col_cfg = &fwd_txfm_1d_col_cfg_dct_16;
|
||||
load_buffer_16x16(input, in, stride, 0, 1, row_cfg->shift[0]);
|
||||
fdct16x16_sse4_1(in, out, col_cfg->cos_bit[0]);
|
||||
col_txfm_16x16_rounding(out, -row_cfg->shift[1]);
|
||||
transpose_16x16(out, in);
|
||||
fadst16x16_sse4_1(in, out, cfg->cos_bit_row[0]);
|
||||
fadst16x16_sse4_1(in, out, row_cfg->cos_bit[0]);
|
||||
transpose_16x16(out, in);
|
||||
write_buffer_16x16(in, coeff);
|
||||
break;
|
||||
case FLIPADST_FLIPADST:
|
||||
cfg = &fwd_txfm_2d_cfg_adst_adst_16;
|
||||
load_buffer_16x16(input, in, stride, 1, 1, cfg->shift[0]);
|
||||
fadst16x16_sse4_1(in, out, cfg->cos_bit_col[0]);
|
||||
col_txfm_16x16_rounding(out, -cfg->shift[1]);
|
||||
row_cfg = &fwd_txfm_1d_row_cfg_adst_16;
|
||||
col_cfg = &fwd_txfm_1d_col_cfg_adst_16;
|
||||
load_buffer_16x16(input, in, stride, 1, 1, row_cfg->shift[0]);
|
||||
fadst16x16_sse4_1(in, out, col_cfg->cos_bit[0]);
|
||||
col_txfm_16x16_rounding(out, -row_cfg->shift[1]);
|
||||
transpose_16x16(out, in);
|
||||
fadst16x16_sse4_1(in, out, cfg->cos_bit_row[0]);
|
||||
fadst16x16_sse4_1(in, out, row_cfg->cos_bit[0]);
|
||||
transpose_16x16(out, in);
|
||||
write_buffer_16x16(in, coeff);
|
||||
break;
|
||||
case ADST_FLIPADST:
|
||||
cfg = &fwd_txfm_2d_cfg_adst_adst_16;
|
||||
load_buffer_16x16(input, in, stride, 0, 1, cfg->shift[0]);
|
||||
fadst16x16_sse4_1(in, out, cfg->cos_bit_col[0]);
|
||||
col_txfm_16x16_rounding(out, -cfg->shift[1]);
|
||||
row_cfg = &fwd_txfm_1d_row_cfg_adst_16;
|
||||
col_cfg = &fwd_txfm_1d_col_cfg_adst_16;
|
||||
load_buffer_16x16(input, in, stride, 0, 1, row_cfg->shift[0]);
|
||||
fadst16x16_sse4_1(in, out, col_cfg->cos_bit[0]);
|
||||
col_txfm_16x16_rounding(out, -row_cfg->shift[1]);
|
||||
transpose_16x16(out, in);
|
||||
fadst16x16_sse4_1(in, out, cfg->cos_bit_row[0]);
|
||||
fadst16x16_sse4_1(in, out, row_cfg->cos_bit[0]);
|
||||
transpose_16x16(out, in);
|
||||
write_buffer_16x16(in, coeff);
|
||||
break;
|
||||
case FLIPADST_ADST:
|
||||
cfg = &fwd_txfm_2d_cfg_adst_adst_16;
|
||||
load_buffer_16x16(input, in, stride, 1, 0, cfg->shift[0]);
|
||||
fadst16x16_sse4_1(in, out, cfg->cos_bit_col[0]);
|
||||
col_txfm_16x16_rounding(out, -cfg->shift[1]);
|
||||
row_cfg = &fwd_txfm_1d_row_cfg_adst_16;
|
||||
col_cfg = &fwd_txfm_1d_col_cfg_adst_16;
|
||||
load_buffer_16x16(input, in, stride, 1, 0, row_cfg->shift[0]);
|
||||
fadst16x16_sse4_1(in, out, col_cfg->cos_bit[0]);
|
||||
col_txfm_16x16_rounding(out, -row_cfg->shift[1]);
|
||||
transpose_16x16(out, in);
|
||||
fadst16x16_sse4_1(in, out, cfg->cos_bit_row[0]);
|
||||
fadst16x16_sse4_1(in, out, row_cfg->cos_bit[0]);
|
||||
transpose_16x16(out, in);
|
||||
write_buffer_16x16(in, coeff);
|
||||
break;
|
||||
|
|
|
|||
|
|
@ -269,8 +269,8 @@ static void fdct16_avx2(__m256i *in) {
|
|||
x0 = _mm256_unpacklo_epi16(v0, v1);
|
||||
x1 = _mm256_unpackhi_epi16(v0, v1);
|
||||
|
||||
t0 = butter_fly(x0, x1, cospi_p16_p16);
|
||||
t1 = butter_fly(x0, x1, cospi_p16_m16);
|
||||
t0 = butter_fly(&x0, &x1, &cospi_p16_p16);
|
||||
t1 = butter_fly(&x0, &x1, &cospi_p16_m16);
|
||||
|
||||
// 4, 12
|
||||
v0 = _mm256_sub_epi16(s1, s2);
|
||||
|
|
@ -279,8 +279,8 @@ static void fdct16_avx2(__m256i *in) {
|
|||
x0 = _mm256_unpacklo_epi16(v0, v1);
|
||||
x1 = _mm256_unpackhi_epi16(v0, v1);
|
||||
|
||||
t2 = butter_fly(x0, x1, cospi_p24_p08);
|
||||
t3 = butter_fly(x0, x1, cospi_m08_p24);
|
||||
t2 = butter_fly(&x0, &x1, &cospi_p24_p08);
|
||||
t3 = butter_fly(&x0, &x1, &cospi_m08_p24);
|
||||
|
||||
// 2, 6, 10, 14
|
||||
s0 = _mm256_sub_epi16(u3, u4);
|
||||
|
|
@ -294,8 +294,8 @@ static void fdct16_avx2(__m256i *in) {
|
|||
x0 = _mm256_unpacklo_epi16(s2, s1);
|
||||
x1 = _mm256_unpackhi_epi16(s2, s1);
|
||||
|
||||
v2 = butter_fly(x0, x1, cospi_p16_p16); // output[5]
|
||||
v1 = butter_fly(x0, x1, cospi_p16_m16); // output[6]
|
||||
v2 = butter_fly(&x0, &x1, &cospi_p16_p16); // output[5]
|
||||
v1 = butter_fly(&x0, &x1, &cospi_p16_m16); // output[6]
|
||||
|
||||
s0 = _mm256_add_epi16(v0, v1); // step[4]
|
||||
s1 = _mm256_sub_epi16(v0, v1); // step[5]
|
||||
|
|
@ -306,14 +306,14 @@ static void fdct16_avx2(__m256i *in) {
|
|||
x0 = _mm256_unpacklo_epi16(s0, s3);
|
||||
x1 = _mm256_unpackhi_epi16(s0, s3);
|
||||
|
||||
t4 = butter_fly(x0, x1, cospi_p28_p04);
|
||||
t5 = butter_fly(x0, x1, cospi_m04_p28);
|
||||
t4 = butter_fly(&x0, &x1, &cospi_p28_p04);
|
||||
t5 = butter_fly(&x0, &x1, &cospi_m04_p28);
|
||||
|
||||
// 10, 6
|
||||
x0 = _mm256_unpacklo_epi16(s1, s2);
|
||||
x1 = _mm256_unpackhi_epi16(s1, s2);
|
||||
t6 = butter_fly(x0, x1, cospi_p12_p20);
|
||||
t7 = butter_fly(x0, x1, cospi_m20_p12);
|
||||
t6 = butter_fly(&x0, &x1, &cospi_p12_p20);
|
||||
t7 = butter_fly(&x0, &x1, &cospi_m20_p12);
|
||||
|
||||
// 1, 3, 5, 7, 9, 11, 13, 15
|
||||
s0 = _mm256_sub_epi16(in[7], in[8]); // step[8]
|
||||
|
|
@ -337,14 +337,14 @@ static void fdct16_avx2(__m256i *in) {
|
|||
x0 = _mm256_unpacklo_epi16(u5, u2);
|
||||
x1 = _mm256_unpackhi_epi16(u5, u2);
|
||||
|
||||
s2 = butter_fly(x0, x1, cospi_p16_p16); // step[13]
|
||||
s5 = butter_fly(x0, x1, cospi_p16_m16); // step[10]
|
||||
s2 = butter_fly(&x0, &x1, &cospi_p16_p16); // step[13]
|
||||
s5 = butter_fly(&x0, &x1, &cospi_p16_m16); // step[10]
|
||||
|
||||
x0 = _mm256_unpacklo_epi16(u4, u3);
|
||||
x1 = _mm256_unpackhi_epi16(u4, u3);
|
||||
|
||||
s3 = butter_fly(x0, x1, cospi_p16_p16); // step[12]
|
||||
s4 = butter_fly(x0, x1, cospi_p16_m16); // step[11]
|
||||
s3 = butter_fly(&x0, &x1, &cospi_p16_p16); // step[12]
|
||||
s4 = butter_fly(&x0, &x1, &cospi_p16_m16); // step[11]
|
||||
|
||||
u0 = _mm256_add_epi16(s0, s4); // output[8]
|
||||
u1 = _mm256_add_epi16(s1, s5);
|
||||
|
|
@ -364,14 +364,14 @@ static void fdct16_avx2(__m256i *in) {
|
|||
x0 = _mm256_unpacklo_epi16(u1, u6);
|
||||
x1 = _mm256_unpackhi_epi16(u1, u6);
|
||||
|
||||
s1 = butter_fly(x0, x1, cospi_m08_p24);
|
||||
s6 = butter_fly(x0, x1, cospi_p24_p08);
|
||||
s1 = butter_fly(&x0, &x1, &cospi_m08_p24);
|
||||
s6 = butter_fly(&x0, &x1, &cospi_p24_p08);
|
||||
|
||||
x0 = _mm256_unpacklo_epi16(u2, u5);
|
||||
x1 = _mm256_unpackhi_epi16(u2, u5);
|
||||
|
||||
s2 = butter_fly(x0, x1, cospi_m24_m08);
|
||||
s5 = butter_fly(x0, x1, cospi_m08_p24);
|
||||
s2 = butter_fly(&x0, &x1, &cospi_m24_m08);
|
||||
s5 = butter_fly(&x0, &x1, &cospi_m08_p24);
|
||||
|
||||
// stage 5
|
||||
u0 = _mm256_add_epi16(s0, s1);
|
||||
|
|
@ -386,23 +386,23 @@ static void fdct16_avx2(__m256i *in) {
|
|||
// stage 6
|
||||
x0 = _mm256_unpacklo_epi16(u0, u7);
|
||||
x1 = _mm256_unpackhi_epi16(u0, u7);
|
||||
in[1] = butter_fly(x0, x1, cospi_p30_p02);
|
||||
in[15] = butter_fly(x0, x1, cospi_m02_p30);
|
||||
in[1] = butter_fly(&x0, &x1, &cospi_p30_p02);
|
||||
in[15] = butter_fly(&x0, &x1, &cospi_m02_p30);
|
||||
|
||||
x0 = _mm256_unpacklo_epi16(u1, u6);
|
||||
x1 = _mm256_unpackhi_epi16(u1, u6);
|
||||
in[9] = butter_fly(x0, x1, cospi_p14_p18);
|
||||
in[7] = butter_fly(x0, x1, cospi_m18_p14);
|
||||
in[9] = butter_fly(&x0, &x1, &cospi_p14_p18);
|
||||
in[7] = butter_fly(&x0, &x1, &cospi_m18_p14);
|
||||
|
||||
x0 = _mm256_unpacklo_epi16(u2, u5);
|
||||
x1 = _mm256_unpackhi_epi16(u2, u5);
|
||||
in[5] = butter_fly(x0, x1, cospi_p22_p10);
|
||||
in[11] = butter_fly(x0, x1, cospi_m10_p22);
|
||||
in[5] = butter_fly(&x0, &x1, &cospi_p22_p10);
|
||||
in[11] = butter_fly(&x0, &x1, &cospi_m10_p22);
|
||||
|
||||
x0 = _mm256_unpacklo_epi16(u3, u4);
|
||||
x1 = _mm256_unpackhi_epi16(u3, u4);
|
||||
in[13] = butter_fly(x0, x1, cospi_p06_p26);
|
||||
in[3] = butter_fly(x0, x1, cospi_m26_p06);
|
||||
in[13] = butter_fly(&x0, &x1, &cospi_p06_p26);
|
||||
in[3] = butter_fly(&x0, &x1, &cospi_m26_p06);
|
||||
}
|
||||
|
||||
void fadst16_avx2(__m256i *in) {
|
||||
|
|
@ -953,7 +953,9 @@ void fadst16_avx2(__m256i *in) {
|
|||
}
|
||||
|
||||
#if CONFIG_EXT_TX
|
||||
static void fidtx16_avx2(__m256i *in) { txfm_scaling16_avx2(Sqrt2, in); }
|
||||
static void fidtx16_avx2(__m256i *in) {
|
||||
txfm_scaling16_avx2((int16_t)Sqrt2, in);
|
||||
}
|
||||
#endif
|
||||
|
||||
void av1_fht16x16_avx2(const int16_t *input, tran_low_t *output, int stride,
|
||||
|
|
@ -964,28 +966,28 @@ void av1_fht16x16_avx2(const int16_t *input, tran_low_t *output, int stride,
|
|||
case DCT_DCT:
|
||||
load_buffer_16x16(input, stride, 0, 0, in);
|
||||
fdct16_avx2(in);
|
||||
mm256_transpose_16x16(in);
|
||||
mm256_transpose_16x16(in, in);
|
||||
right_shift_16x16(in);
|
||||
fdct16_avx2(in);
|
||||
break;
|
||||
case ADST_DCT:
|
||||
load_buffer_16x16(input, stride, 0, 0, in);
|
||||
fadst16_avx2(in);
|
||||
mm256_transpose_16x16(in);
|
||||
mm256_transpose_16x16(in, in);
|
||||
right_shift_16x16(in);
|
||||
fdct16_avx2(in);
|
||||
break;
|
||||
case DCT_ADST:
|
||||
load_buffer_16x16(input, stride, 0, 0, in);
|
||||
fdct16_avx2(in);
|
||||
mm256_transpose_16x16(in);
|
||||
mm256_transpose_16x16(in, in);
|
||||
right_shift_16x16(in);
|
||||
fadst16_avx2(in);
|
||||
break;
|
||||
case ADST_ADST:
|
||||
load_buffer_16x16(input, stride, 0, 0, in);
|
||||
fadst16_avx2(in);
|
||||
mm256_transpose_16x16(in);
|
||||
mm256_transpose_16x16(in, in);
|
||||
right_shift_16x16(in);
|
||||
fadst16_avx2(in);
|
||||
break;
|
||||
|
|
@ -993,91 +995,91 @@ void av1_fht16x16_avx2(const int16_t *input, tran_low_t *output, int stride,
|
|||
case FLIPADST_DCT:
|
||||
load_buffer_16x16(input, stride, 1, 0, in);
|
||||
fadst16_avx2(in);
|
||||
mm256_transpose_16x16(in);
|
||||
mm256_transpose_16x16(in, in);
|
||||
right_shift_16x16(in);
|
||||
fdct16_avx2(in);
|
||||
break;
|
||||
case DCT_FLIPADST:
|
||||
load_buffer_16x16(input, stride, 0, 1, in);
|
||||
fdct16_avx2(in);
|
||||
mm256_transpose_16x16(in);
|
||||
mm256_transpose_16x16(in, in);
|
||||
right_shift_16x16(in);
|
||||
fadst16_avx2(in);
|
||||
break;
|
||||
case FLIPADST_FLIPADST:
|
||||
load_buffer_16x16(input, stride, 1, 1, in);
|
||||
fadst16_avx2(in);
|
||||
mm256_transpose_16x16(in);
|
||||
mm256_transpose_16x16(in, in);
|
||||
right_shift_16x16(in);
|
||||
fadst16_avx2(in);
|
||||
break;
|
||||
case ADST_FLIPADST:
|
||||
load_buffer_16x16(input, stride, 0, 1, in);
|
||||
fadst16_avx2(in);
|
||||
mm256_transpose_16x16(in);
|
||||
mm256_transpose_16x16(in, in);
|
||||
right_shift_16x16(in);
|
||||
fadst16_avx2(in);
|
||||
break;
|
||||
case FLIPADST_ADST:
|
||||
load_buffer_16x16(input, stride, 1, 0, in);
|
||||
fadst16_avx2(in);
|
||||
mm256_transpose_16x16(in);
|
||||
mm256_transpose_16x16(in, in);
|
||||
right_shift_16x16(in);
|
||||
fadst16_avx2(in);
|
||||
break;
|
||||
case IDTX:
|
||||
load_buffer_16x16(input, stride, 0, 0, in);
|
||||
fidtx16_avx2(in);
|
||||
mm256_transpose_16x16(in);
|
||||
mm256_transpose_16x16(in, in);
|
||||
right_shift_16x16(in);
|
||||
fidtx16_avx2(in);
|
||||
break;
|
||||
case V_DCT:
|
||||
load_buffer_16x16(input, stride, 0, 0, in);
|
||||
fdct16_avx2(in);
|
||||
mm256_transpose_16x16(in);
|
||||
mm256_transpose_16x16(in, in);
|
||||
right_shift_16x16(in);
|
||||
fidtx16_avx2(in);
|
||||
break;
|
||||
case H_DCT:
|
||||
load_buffer_16x16(input, stride, 0, 0, in);
|
||||
fidtx16_avx2(in);
|
||||
mm256_transpose_16x16(in);
|
||||
mm256_transpose_16x16(in, in);
|
||||
right_shift_16x16(in);
|
||||
fdct16_avx2(in);
|
||||
break;
|
||||
case V_ADST:
|
||||
load_buffer_16x16(input, stride, 0, 0, in);
|
||||
fadst16_avx2(in);
|
||||
mm256_transpose_16x16(in);
|
||||
mm256_transpose_16x16(in, in);
|
||||
right_shift_16x16(in);
|
||||
fidtx16_avx2(in);
|
||||
break;
|
||||
case H_ADST:
|
||||
load_buffer_16x16(input, stride, 0, 0, in);
|
||||
fidtx16_avx2(in);
|
||||
mm256_transpose_16x16(in);
|
||||
mm256_transpose_16x16(in, in);
|
||||
right_shift_16x16(in);
|
||||
fadst16_avx2(in);
|
||||
break;
|
||||
case V_FLIPADST:
|
||||
load_buffer_16x16(input, stride, 1, 0, in);
|
||||
fadst16_avx2(in);
|
||||
mm256_transpose_16x16(in);
|
||||
mm256_transpose_16x16(in, in);
|
||||
right_shift_16x16(in);
|
||||
fidtx16_avx2(in);
|
||||
break;
|
||||
case H_FLIPADST:
|
||||
load_buffer_16x16(input, stride, 0, 1, in);
|
||||
fidtx16_avx2(in);
|
||||
mm256_transpose_16x16(in);
|
||||
mm256_transpose_16x16(in, in);
|
||||
right_shift_16x16(in);
|
||||
fadst16_avx2(in);
|
||||
break;
|
||||
#endif // CONFIG_EXT_TX
|
||||
default: assert(0); break;
|
||||
}
|
||||
mm256_transpose_16x16(in);
|
||||
mm256_transpose_16x16(in, in);
|
||||
write_buffer_16x16(in, output);
|
||||
_mm256_zeroupper();
|
||||
}
|
||||
|
|
@ -1110,10 +1112,10 @@ static void mm256_vectors_swap(__m256i *a0, __m256i *a1, const int size) {
|
|||
}
|
||||
|
||||
static void mm256_transpose_32x32(__m256i *in0, __m256i *in1) {
|
||||
mm256_transpose_16x16(in0);
|
||||
mm256_transpose_16x16(&in0[16]);
|
||||
mm256_transpose_16x16(in1);
|
||||
mm256_transpose_16x16(&in1[16]);
|
||||
mm256_transpose_16x16(in0, in0);
|
||||
mm256_transpose_16x16(&in0[16], &in0[16]);
|
||||
mm256_transpose_16x16(in1, in1);
|
||||
mm256_transpose_16x16(&in1[16], &in1[16]);
|
||||
mm256_vectors_swap(&in0[16], in1, 16);
|
||||
}
|
||||
|
||||
|
|
@ -1247,23 +1249,23 @@ static void fdct16_odd_avx2(__m256i *in) {
|
|||
|
||||
u0 = _mm256_unpacklo_epi16(in[4], in[11]);
|
||||
u1 = _mm256_unpackhi_epi16(in[4], in[11]);
|
||||
y4 = butter_fly(u0, u1, cospi_m16_p16);
|
||||
y11 = butter_fly(u0, u1, cospi_p16_p16);
|
||||
y4 = butter_fly(&u0, &u1, &cospi_m16_p16);
|
||||
y11 = butter_fly(&u0, &u1, &cospi_p16_p16);
|
||||
|
||||
u0 = _mm256_unpacklo_epi16(in[5], in[10]);
|
||||
u1 = _mm256_unpackhi_epi16(in[5], in[10]);
|
||||
y5 = butter_fly(u0, u1, cospi_m16_p16);
|
||||
y10 = butter_fly(u0, u1, cospi_p16_p16);
|
||||
y5 = butter_fly(&u0, &u1, &cospi_m16_p16);
|
||||
y10 = butter_fly(&u0, &u1, &cospi_p16_p16);
|
||||
|
||||
u0 = _mm256_unpacklo_epi16(in[6], in[9]);
|
||||
u1 = _mm256_unpackhi_epi16(in[6], in[9]);
|
||||
y6 = butter_fly(u0, u1, cospi_m16_p16);
|
||||
y9 = butter_fly(u0, u1, cospi_p16_p16);
|
||||
y6 = butter_fly(&u0, &u1, &cospi_m16_p16);
|
||||
y9 = butter_fly(&u0, &u1, &cospi_p16_p16);
|
||||
|
||||
u0 = _mm256_unpacklo_epi16(in[7], in[8]);
|
||||
u1 = _mm256_unpackhi_epi16(in[7], in[8]);
|
||||
y7 = butter_fly(u0, u1, cospi_m16_p16);
|
||||
y8 = butter_fly(u0, u1, cospi_p16_p16);
|
||||
y7 = butter_fly(&u0, &u1, &cospi_m16_p16);
|
||||
y8 = butter_fly(&u0, &u1, &cospi_p16_p16);
|
||||
|
||||
y12 = in[12];
|
||||
y13 = in[13];
|
||||
|
|
@ -1300,23 +1302,23 @@ static void fdct16_odd_avx2(__m256i *in) {
|
|||
|
||||
u0 = _mm256_unpacklo_epi16(x2, x13);
|
||||
u1 = _mm256_unpackhi_epi16(x2, x13);
|
||||
y2 = butter_fly(u0, u1, cospi_m08_p24);
|
||||
y13 = butter_fly(u0, u1, cospi_p24_p08);
|
||||
y2 = butter_fly(&u0, &u1, &cospi_m08_p24);
|
||||
y13 = butter_fly(&u0, &u1, &cospi_p24_p08);
|
||||
|
||||
u0 = _mm256_unpacklo_epi16(x3, x12);
|
||||
u1 = _mm256_unpackhi_epi16(x3, x12);
|
||||
y3 = butter_fly(u0, u1, cospi_m08_p24);
|
||||
y12 = butter_fly(u0, u1, cospi_p24_p08);
|
||||
y3 = butter_fly(&u0, &u1, &cospi_m08_p24);
|
||||
y12 = butter_fly(&u0, &u1, &cospi_p24_p08);
|
||||
|
||||
u0 = _mm256_unpacklo_epi16(x4, x11);
|
||||
u1 = _mm256_unpackhi_epi16(x4, x11);
|
||||
y4 = butter_fly(u0, u1, cospi_m24_m08);
|
||||
y11 = butter_fly(u0, u1, cospi_m08_p24);
|
||||
y4 = butter_fly(&u0, &u1, &cospi_m24_m08);
|
||||
y11 = butter_fly(&u0, &u1, &cospi_m08_p24);
|
||||
|
||||
u0 = _mm256_unpacklo_epi16(x5, x10);
|
||||
u1 = _mm256_unpackhi_epi16(x5, x10);
|
||||
y5 = butter_fly(u0, u1, cospi_m24_m08);
|
||||
y10 = butter_fly(u0, u1, cospi_m08_p24);
|
||||
y5 = butter_fly(&u0, &u1, &cospi_m24_m08);
|
||||
y10 = butter_fly(&u0, &u1, &cospi_m08_p24);
|
||||
|
||||
// stage 5
|
||||
x0 = _mm256_add_epi16(y0, y3);
|
||||
|
|
@ -1349,23 +1351,23 @@ static void fdct16_odd_avx2(__m256i *in) {
|
|||
|
||||
u0 = _mm256_unpacklo_epi16(x1, x14);
|
||||
u1 = _mm256_unpackhi_epi16(x1, x14);
|
||||
y1 = butter_fly(u0, u1, cospi_m04_p28);
|
||||
y14 = butter_fly(u0, u1, cospi_p28_p04);
|
||||
y1 = butter_fly(&u0, &u1, &cospi_m04_p28);
|
||||
y14 = butter_fly(&u0, &u1, &cospi_p28_p04);
|
||||
|
||||
u0 = _mm256_unpacklo_epi16(x2, x13);
|
||||
u1 = _mm256_unpackhi_epi16(x2, x13);
|
||||
y2 = butter_fly(u0, u1, cospi_m28_m04);
|
||||
y13 = butter_fly(u0, u1, cospi_m04_p28);
|
||||
y2 = butter_fly(&u0, &u1, &cospi_m28_m04);
|
||||
y13 = butter_fly(&u0, &u1, &cospi_m04_p28);
|
||||
|
||||
u0 = _mm256_unpacklo_epi16(x5, x10);
|
||||
u1 = _mm256_unpackhi_epi16(x5, x10);
|
||||
y5 = butter_fly(u0, u1, cospi_m20_p12);
|
||||
y10 = butter_fly(u0, u1, cospi_p12_p20);
|
||||
y5 = butter_fly(&u0, &u1, &cospi_m20_p12);
|
||||
y10 = butter_fly(&u0, &u1, &cospi_p12_p20);
|
||||
|
||||
u0 = _mm256_unpacklo_epi16(x6, x9);
|
||||
u1 = _mm256_unpackhi_epi16(x6, x9);
|
||||
y6 = butter_fly(u0, u1, cospi_m12_m20);
|
||||
y9 = butter_fly(u0, u1, cospi_m20_p12);
|
||||
y6 = butter_fly(&u0, &u1, &cospi_m12_m20);
|
||||
y9 = butter_fly(&u0, &u1, &cospi_m20_p12);
|
||||
|
||||
// stage 7
|
||||
x0 = _mm256_add_epi16(y0, y1);
|
||||
|
|
@ -1389,43 +1391,43 @@ static void fdct16_odd_avx2(__m256i *in) {
|
|||
// stage 8
|
||||
u0 = _mm256_unpacklo_epi16(x0, x15);
|
||||
u1 = _mm256_unpackhi_epi16(x0, x15);
|
||||
in[0] = butter_fly(u0, u1, cospi_p31_p01);
|
||||
in[15] = butter_fly(u0, u1, cospi_m01_p31);
|
||||
in[0] = butter_fly(&u0, &u1, &cospi_p31_p01);
|
||||
in[15] = butter_fly(&u0, &u1, &cospi_m01_p31);
|
||||
|
||||
u0 = _mm256_unpacklo_epi16(x1, x14);
|
||||
u1 = _mm256_unpackhi_epi16(x1, x14);
|
||||
in[1] = butter_fly(u0, u1, cospi_p15_p17);
|
||||
in[14] = butter_fly(u0, u1, cospi_m17_p15);
|
||||
in[1] = butter_fly(&u0, &u1, &cospi_p15_p17);
|
||||
in[14] = butter_fly(&u0, &u1, &cospi_m17_p15);
|
||||
|
||||
u0 = _mm256_unpacklo_epi16(x2, x13);
|
||||
u1 = _mm256_unpackhi_epi16(x2, x13);
|
||||
in[2] = butter_fly(u0, u1, cospi_p23_p09);
|
||||
in[13] = butter_fly(u0, u1, cospi_m09_p23);
|
||||
in[2] = butter_fly(&u0, &u1, &cospi_p23_p09);
|
||||
in[13] = butter_fly(&u0, &u1, &cospi_m09_p23);
|
||||
|
||||
u0 = _mm256_unpacklo_epi16(x3, x12);
|
||||
u1 = _mm256_unpackhi_epi16(x3, x12);
|
||||
in[3] = butter_fly(u0, u1, cospi_p07_p25);
|
||||
in[12] = butter_fly(u0, u1, cospi_m25_p07);
|
||||
in[3] = butter_fly(&u0, &u1, &cospi_p07_p25);
|
||||
in[12] = butter_fly(&u0, &u1, &cospi_m25_p07);
|
||||
|
||||
u0 = _mm256_unpacklo_epi16(x4, x11);
|
||||
u1 = _mm256_unpackhi_epi16(x4, x11);
|
||||
in[4] = butter_fly(u0, u1, cospi_p27_p05);
|
||||
in[11] = butter_fly(u0, u1, cospi_m05_p27);
|
||||
in[4] = butter_fly(&u0, &u1, &cospi_p27_p05);
|
||||
in[11] = butter_fly(&u0, &u1, &cospi_m05_p27);
|
||||
|
||||
u0 = _mm256_unpacklo_epi16(x5, x10);
|
||||
u1 = _mm256_unpackhi_epi16(x5, x10);
|
||||
in[5] = butter_fly(u0, u1, cospi_p11_p21);
|
||||
in[10] = butter_fly(u0, u1, cospi_m21_p11);
|
||||
in[5] = butter_fly(&u0, &u1, &cospi_p11_p21);
|
||||
in[10] = butter_fly(&u0, &u1, &cospi_m21_p11);
|
||||
|
||||
u0 = _mm256_unpacklo_epi16(x6, x9);
|
||||
u1 = _mm256_unpackhi_epi16(x6, x9);
|
||||
in[6] = butter_fly(u0, u1, cospi_p19_p13);
|
||||
in[9] = butter_fly(u0, u1, cospi_m13_p19);
|
||||
in[6] = butter_fly(&u0, &u1, &cospi_p19_p13);
|
||||
in[9] = butter_fly(&u0, &u1, &cospi_m13_p19);
|
||||
|
||||
u0 = _mm256_unpacklo_epi16(x7, x8);
|
||||
u1 = _mm256_unpackhi_epi16(x7, x8);
|
||||
in[7] = butter_fly(u0, u1, cospi_p03_p29);
|
||||
in[8] = butter_fly(u0, u1, cospi_m29_p03);
|
||||
in[7] = butter_fly(&u0, &u1, &cospi_p03_p29);
|
||||
in[8] = butter_fly(&u0, &u1, &cospi_m29_p03);
|
||||
}
|
||||
|
||||
static void fdct32_avx2(__m256i *in0, __m256i *in1) {
|
||||
|
|
@ -1464,7 +1466,7 @@ static INLINE void write_buffer_32x32(const __m256i *in0, const __m256i *in1,
|
|||
static void fhalfright32_16col_avx2(__m256i *in) {
|
||||
int i = 0;
|
||||
const __m256i zero = _mm256_setzero_si256();
|
||||
const __m256i sqrt2 = _mm256_set1_epi16(Sqrt2);
|
||||
const __m256i sqrt2 = _mm256_set1_epi16((int16_t)Sqrt2);
|
||||
const __m256i dct_rounding = _mm256_set1_epi32(DCT_CONST_ROUNDING);
|
||||
__m256i x0, x1;
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue