mirror of
https://repo.dactyloidae.xyz/Dactyloidae/UXP.git
synced 2026-10-09 00:37:32 +09:00
Update aom to commit id f5bdeac22930ff4c6b219be49c843db35970b918
This commit is contained in:
parent
e329bc4331
commit
2ab8fcbfd1
370 changed files with 56185 additions and 32026 deletions
|
|
@ -352,10 +352,7 @@ void av1_cyclic_refresh_check_golden_update(AV1_COMP *const cpi) {
|
|||
// For video conference clips, if the background has high motion in current
|
||||
// frame because of the camera movement, set this frame as the golden frame.
|
||||
// Use 70% and 5% as the thresholds for golden frame refreshing.
|
||||
// Also, force this frame as a golden update frame if this frame will change
|
||||
// the resolution (av1_resize_pending != 0).
|
||||
if (av1_resize_pending(cpi) ||
|
||||
(cnt1 * 10 > (70 * rows * cols) && cnt2 * 20 < cnt1)) {
|
||||
if (cnt1 * 10 > (70 * rows * cols) && cnt2 * 20 < cnt1) {
|
||||
av1_cyclic_refresh_set_golden_update(cpi);
|
||||
rc->frames_till_gf_update_due = rc->baseline_gf_interval;
|
||||
|
||||
|
|
|
|||
45
third_party/aom/av1/encoder/av1_quantize.c
vendored
45
third_party/aom/av1/encoder/av1_quantize.c
vendored
|
|
@ -845,7 +845,6 @@ void av1_quantize_dc_nuq_facade(const tran_low_t *coeff_ptr, intptr_t n_coeffs,
|
|||
}
|
||||
#endif // CONFIG_NEW_QUANT
|
||||
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
void av1_highbd_quantize_fp_facade(const tran_low_t *coeff_ptr,
|
||||
intptr_t n_coeffs, const MACROBLOCK_PLANE *p,
|
||||
tran_low_t *qcoeff_ptr,
|
||||
|
|
@ -899,14 +898,29 @@ void av1_highbd_quantize_b_facade(const tran_low_t *coeff_ptr,
|
|||
|
||||
switch (qparam->log_scale) {
|
||||
case 0:
|
||||
aom_highbd_quantize_b(coeff_ptr, n_coeffs, skip_block, p->zbin, p->round,
|
||||
p->quant, p->quant_shift, qcoeff_ptr, dqcoeff_ptr,
|
||||
pd->dequant, eob_ptr, sc->scan, sc->iscan
|
||||
if (LIKELY(n_coeffs >= 8)) {
|
||||
aom_highbd_quantize_b(coeff_ptr, n_coeffs, skip_block, p->zbin,
|
||||
p->round, p->quant, p->quant_shift, qcoeff_ptr,
|
||||
dqcoeff_ptr, pd->dequant, eob_ptr, sc->scan,
|
||||
sc->iscan
|
||||
#if CONFIG_AOM_QM
|
||||
,
|
||||
qm_ptr, iqm_ptr
|
||||
,
|
||||
qm_ptr, iqm_ptr
|
||||
#endif
|
||||
);
|
||||
);
|
||||
} else {
|
||||
// TODO(luoyi): Need SIMD (e.g. sse2) for smaller block size
|
||||
// quantization
|
||||
aom_highbd_quantize_b_c(coeff_ptr, n_coeffs, skip_block, p->zbin,
|
||||
p->round, p->quant, p->quant_shift, qcoeff_ptr,
|
||||
dqcoeff_ptr, pd->dequant, eob_ptr, sc->scan,
|
||||
sc->iscan
|
||||
#if CONFIG_AOM_QM
|
||||
,
|
||||
qm_ptr, iqm_ptr
|
||||
#endif
|
||||
);
|
||||
}
|
||||
break;
|
||||
case 1:
|
||||
aom_highbd_quantize_b_32x32(coeff_ptr, n_coeffs, skip_block, p->zbin,
|
||||
|
|
@ -936,7 +950,6 @@ void av1_highbd_quantize_b_facade(const tran_low_t *coeff_ptr,
|
|||
}
|
||||
}
|
||||
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
static INLINE void highbd_quantize_dc(
|
||||
const tran_low_t *coeff_ptr, int n_coeffs, int skip_block,
|
||||
const int16_t *round_ptr, const int16_t quant, tran_low_t *qcoeff_ptr,
|
||||
|
|
@ -958,14 +971,13 @@ static INLINE void highbd_quantize_dc(
|
|||
const int coeff_sign = (coeff >> 31);
|
||||
const int abs_coeff = (coeff ^ coeff_sign) - coeff_sign;
|
||||
const int64_t tmp = abs_coeff + round_ptr[0];
|
||||
const uint32_t abs_qcoeff = (uint32_t)((tmp * quant) >> (16 - log_scale));
|
||||
const int abs_qcoeff = (int)((tmp * quant) >> (16 - log_scale));
|
||||
qcoeff_ptr[0] = (tran_low_t)((abs_qcoeff ^ coeff_sign) - coeff_sign);
|
||||
dqcoeff_ptr[0] = qcoeff_ptr[0] * dequant_ptr / (1 << log_scale);
|
||||
if (abs_qcoeff) eob = 0;
|
||||
}
|
||||
*eob_ptr = eob + 1;
|
||||
}
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
|
||||
void av1_highbd_quantize_dc_facade(const tran_low_t *coeff_ptr,
|
||||
intptr_t n_coeffs, const MACROBLOCK_PLANE *p,
|
||||
|
|
@ -1504,9 +1516,7 @@ void av1_highbd_quantize_dc_nuq_facade(
|
|||
}
|
||||
}
|
||||
#endif // CONFIG_NEW_QUANT
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
void av1_highbd_quantize_fp_c(const tran_low_t *coeff_ptr, intptr_t count,
|
||||
int skip_block, const int16_t *zbin_ptr,
|
||||
const int16_t *round_ptr,
|
||||
|
|
@ -1547,15 +1557,14 @@ void av1_highbd_quantize_fp_c(const tran_low_t *coeff_ptr, intptr_t count,
|
|||
#endif
|
||||
const int coeff_sign = (coeff >> 31);
|
||||
const int abs_coeff = (coeff ^ coeff_sign) - coeff_sign;
|
||||
const int64_t tmp = abs_coeff + round_ptr[rc != 0];
|
||||
const int64_t tmp = abs_coeff + (round_ptr[rc != 0] >> log_scale);
|
||||
#if CONFIG_AOM_QM
|
||||
const uint32_t abs_qcoeff =
|
||||
(uint32_t)((tmp * quant_ptr[rc != 0] * wt) >> (shift + AOM_QM_BITS));
|
||||
const int abs_qcoeff =
|
||||
(int)((tmp * quant_ptr[rc != 0] * wt) >> (shift + AOM_QM_BITS));
|
||||
qcoeff_ptr[rc] = (tran_low_t)((abs_qcoeff ^ coeff_sign) - coeff_sign);
|
||||
dqcoeff_ptr[rc] = qcoeff_ptr[rc] * dequant / scale;
|
||||
#else
|
||||
const uint32_t abs_qcoeff =
|
||||
(uint32_t)((tmp * quant_ptr[rc != 0]) >> shift);
|
||||
const int abs_qcoeff = (int)((tmp * quant_ptr[rc != 0]) >> shift);
|
||||
qcoeff_ptr[rc] = (tran_low_t)((abs_qcoeff ^ coeff_sign) - coeff_sign);
|
||||
dqcoeff_ptr[rc] = qcoeff_ptr[rc] * dequant_ptr[rc != 0] / scale;
|
||||
#endif
|
||||
|
|
@ -1565,8 +1574,6 @@ void av1_highbd_quantize_fp_c(const tran_low_t *coeff_ptr, intptr_t count,
|
|||
*eob_ptr = eob + 1;
|
||||
}
|
||||
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
|
||||
static void invert_quant(int16_t *quant, int16_t *shift, int d) {
|
||||
uint32_t t;
|
||||
int l, m;
|
||||
|
|
|
|||
2
third_party/aom/av1/encoder/av1_quantize.h
vendored
2
third_party/aom/av1/encoder/av1_quantize.h
vendored
|
|
@ -146,7 +146,6 @@ void av1_quantize_dc_nuq_facade(const tran_low_t *coeff_ptr, intptr_t n_coeffs,
|
|||
const QUANT_PARAM *qparam);
|
||||
#endif // CONFIG_NEW_QUANT
|
||||
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
void av1_highbd_quantize_fp_facade(const tran_low_t *coeff_ptr,
|
||||
intptr_t n_coeffs, const MACROBLOCK_PLANE *p,
|
||||
tran_low_t *qcoeff_ptr,
|
||||
|
|
@ -190,7 +189,6 @@ void av1_highbd_quantize_dc_nuq_facade(
|
|||
tran_low_t *dqcoeff_ptr, uint16_t *eob_ptr, const SCAN_ORDER *sc,
|
||||
const QUANT_PARAM *qparam);
|
||||
#endif // CONFIG_NEW_QUANT
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
|
||||
#ifdef __cplusplus
|
||||
} // extern "C"
|
||||
|
|
|
|||
748
third_party/aom/av1/encoder/bgsprite.c
vendored
Normal file
748
third_party/aom/av1/encoder/bgsprite.c
vendored
Normal file
|
|
@ -0,0 +1,748 @@
|
|||
/*
|
||||
* Copyright (c) 2017, Alliance for Open Media. All rights reserved
|
||||
*
|
||||
* This source code is subject to the terms of the BSD 2 Clause License and
|
||||
* the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
|
||||
* was not distributed with this source code in the LICENSE file, you can
|
||||
* obtain it at www.aomedia.org/license/software. If the Alliance for Open
|
||||
* Media Patent License 1.0 was not distributed with this source code in the
|
||||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#define _POSIX_C_SOURCE 200112L // rand_r()
|
||||
#include <assert.h>
|
||||
#include <float.h>
|
||||
#include <limits.h>
|
||||
#include <math.h>
|
||||
#include <stdlib.h>
|
||||
#include <time.h>
|
||||
|
||||
#include "av1/encoder/bgsprite.h"
|
||||
|
||||
#include "aom_mem/aom_mem.h"
|
||||
#include "./aom_scale_rtcd.h"
|
||||
#include "av1/common/mv.h"
|
||||
#include "av1/common/warped_motion.h"
|
||||
#include "av1/encoder/encoder.h"
|
||||
#include "av1/encoder/global_motion.h"
|
||||
#include "av1/encoder/mathutils.h"
|
||||
#include "av1/encoder/temporal_filter.h"
|
||||
|
||||
/* Blending Modes:
|
||||
* 0 = Median
|
||||
* 1 = Mean
|
||||
*/
|
||||
#define BGSPRITE_BLENDING_MODE 1
|
||||
|
||||
/* Interpolation for panorama alignment sampling:
|
||||
* 0 = Nearest neighbor
|
||||
* 1 = Bilinear
|
||||
*/
|
||||
#define BGSPRITE_INTERPOLATION 0
|
||||
|
||||
#define TRANSFORM_MAT_DIM 3
|
||||
|
||||
typedef struct {
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
uint16_t y;
|
||||
uint16_t u;
|
||||
uint16_t v;
|
||||
#else
|
||||
uint8_t y;
|
||||
uint8_t u;
|
||||
uint8_t v;
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
} YuvPixel;
|
||||
|
||||
// Maps to convert from matrix form to param vector form.
|
||||
static const int params_to_matrix_map[] = { 2, 3, 0, 4, 5, 1, 6, 7 };
|
||||
static const int matrix_to_params_map[] = { 2, 5, 0, 1, 3, 4, 6, 7 };
|
||||
|
||||
// Convert the parameter array to a 3x3 matrix form.
|
||||
static void params_to_matrix(const double *const params, double *target) {
|
||||
for (int i = 0; i < MAX_PARAMDIM - 1; i++) {
|
||||
assert(params_to_matrix_map[i] < MAX_PARAMDIM - 1);
|
||||
target[i] = params[params_to_matrix_map[i]];
|
||||
}
|
||||
target[8] = 1;
|
||||
}
|
||||
|
||||
// Convert a 3x3 matrix to a parameter array form.
|
||||
static void matrix_to_params(const double *const matrix, double *target) {
|
||||
for (int i = 0; i < MAX_PARAMDIM - 1; i++) {
|
||||
assert(matrix_to_params_map[i] < MAX_PARAMDIM - 1);
|
||||
target[i] = matrix[matrix_to_params_map[i]];
|
||||
}
|
||||
}
|
||||
|
||||
// Do matrix multiplication on params.
|
||||
static void multiply_params(double *const m1, double *const m2,
|
||||
double *target) {
|
||||
double m1_matrix[MAX_PARAMDIM];
|
||||
double m2_matrix[MAX_PARAMDIM];
|
||||
double result[MAX_PARAMDIM];
|
||||
|
||||
params_to_matrix(m1, m1_matrix);
|
||||
params_to_matrix(m2, m2_matrix);
|
||||
multiply_mat(m2_matrix, m1_matrix, result, TRANSFORM_MAT_DIM,
|
||||
TRANSFORM_MAT_DIM, TRANSFORM_MAT_DIM);
|
||||
matrix_to_params(result, target);
|
||||
}
|
||||
|
||||
// Finds x and y limits of a single transformed image.
|
||||
// Width and height are the size of the input video.
|
||||
static void find_frame_limit(int width, int height,
|
||||
const double *const transform, int *x_min,
|
||||
int *x_max, int *y_min, int *y_max) {
|
||||
double transform_matrix[MAX_PARAMDIM];
|
||||
double xy_matrix[3] = { 0, 0, 1 };
|
||||
double uv_matrix[3] = { 0 };
|
||||
// Macro used to update frame limits based on transformed coordinates.
|
||||
#define UPDATELIMITS(u, v, x_min, x_max, y_min, y_max) \
|
||||
{ \
|
||||
if ((int)ceil(u) > *x_max) { \
|
||||
*x_max = (int)ceil(u); \
|
||||
} \
|
||||
if ((int)floor(u) < *x_min) { \
|
||||
*x_min = (int)floor(u); \
|
||||
} \
|
||||
if ((int)ceil(v) > *y_max) { \
|
||||
*y_max = (int)ceil(v); \
|
||||
} \
|
||||
if ((int)floor(v) < *y_min) { \
|
||||
*y_min = (int)floor(v); \
|
||||
} \
|
||||
}
|
||||
|
||||
params_to_matrix(transform, transform_matrix);
|
||||
xy_matrix[0] = 0;
|
||||
xy_matrix[1] = 0;
|
||||
multiply_mat(transform_matrix, xy_matrix, uv_matrix, TRANSFORM_MAT_DIM,
|
||||
TRANSFORM_MAT_DIM, 1);
|
||||
*x_max = (int)ceil(uv_matrix[0]);
|
||||
*x_min = (int)floor(uv_matrix[0]);
|
||||
*y_max = (int)ceil(uv_matrix[1]);
|
||||
*y_min = (int)floor(uv_matrix[1]);
|
||||
|
||||
xy_matrix[0] = width;
|
||||
xy_matrix[1] = 0;
|
||||
multiply_mat(transform_matrix, xy_matrix, uv_matrix, TRANSFORM_MAT_DIM,
|
||||
TRANSFORM_MAT_DIM, 1);
|
||||
UPDATELIMITS(uv_matrix[0], uv_matrix[1], x_min, x_max, y_min, y_max);
|
||||
|
||||
xy_matrix[0] = width;
|
||||
xy_matrix[1] = height;
|
||||
multiply_mat(transform_matrix, xy_matrix, uv_matrix, TRANSFORM_MAT_DIM,
|
||||
TRANSFORM_MAT_DIM, 1);
|
||||
UPDATELIMITS(uv_matrix[0], uv_matrix[1], x_min, x_max, y_min, y_max);
|
||||
|
||||
xy_matrix[0] = 0;
|
||||
xy_matrix[1] = height;
|
||||
multiply_mat(transform_matrix, xy_matrix, uv_matrix, TRANSFORM_MAT_DIM,
|
||||
TRANSFORM_MAT_DIM, 1);
|
||||
UPDATELIMITS(uv_matrix[0], uv_matrix[1], x_min, x_max, y_min, y_max);
|
||||
|
||||
#undef UPDATELIMITS
|
||||
}
|
||||
|
||||
// Finds x and y limits for arrays. Also finds the overall max and minimums
|
||||
static void find_limits(int width, int height, const double **const params,
|
||||
int num_frames, int *x_min, int *x_max, int *y_min,
|
||||
int *y_max, int *pano_x_min, int *pano_x_max,
|
||||
int *pano_y_min, int *pano_y_max) {
|
||||
*pano_x_max = INT_MIN;
|
||||
*pano_x_min = INT_MAX;
|
||||
*pano_y_max = INT_MIN;
|
||||
*pano_y_min = INT_MAX;
|
||||
for (int i = 0; i < num_frames; ++i) {
|
||||
find_frame_limit(width, height, (const double *const)params[i], &x_min[i],
|
||||
&x_max[i], &y_min[i], &y_max[i]);
|
||||
if (x_max[i] > *pano_x_max) {
|
||||
*pano_x_max = x_max[i];
|
||||
}
|
||||
if (x_min[i] < *pano_x_min) {
|
||||
*pano_x_min = x_min[i];
|
||||
}
|
||||
if (y_max[i] > *pano_y_max) {
|
||||
*pano_y_max = y_max[i];
|
||||
}
|
||||
if (y_min[i] < *pano_y_min) {
|
||||
*pano_y_min = y_min[i];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Inverts a 3x3 matrix that is in the parameter form.
|
||||
static void invert_params(const double *const params, double *target) {
|
||||
double temp[MAX_PARAMDIM] = { 0 };
|
||||
params_to_matrix(params, temp);
|
||||
|
||||
// Find determinant of matrix (expansion by minors).
|
||||
const double det = temp[0] * ((temp[4] * temp[8]) - (temp[5] * temp[7])) -
|
||||
temp[1] * ((temp[3] * temp[8]) - (temp[5] * temp[6])) +
|
||||
temp[2] * ((temp[3] * temp[7]) - (temp[4] * temp[6]));
|
||||
assert(det != 0);
|
||||
|
||||
// inverse is transpose of cofactor * 1/det.
|
||||
double inverse[MAX_PARAMDIM] = { 0 };
|
||||
inverse[0] = (temp[4] * temp[8] - temp[7] * temp[5]) / det;
|
||||
inverse[1] = (temp[2] * temp[7] - temp[1] * temp[8]) / det;
|
||||
inverse[2] = (temp[1] * temp[5] - temp[2] * temp[4]) / det;
|
||||
inverse[3] = (temp[5] * temp[6] - temp[3] * temp[8]) / det;
|
||||
inverse[4] = (temp[0] * temp[8] - temp[2] * temp[6]) / det;
|
||||
inverse[5] = (temp[3] * temp[2] - temp[0] * temp[5]) / det;
|
||||
inverse[6] = (temp[3] * temp[7] - temp[6] * temp[4]) / det;
|
||||
inverse[7] = (temp[6] * temp[1] - temp[0] * temp[7]) / det;
|
||||
inverse[8] = (temp[0] * temp[4] - temp[3] * temp[1]) / det;
|
||||
|
||||
matrix_to_params(inverse, target);
|
||||
}
|
||||
|
||||
#if BGSPRITE_BLENDING_MODE == 0
|
||||
// swaps two YuvPixels.
|
||||
static void swap_yuv(YuvPixel *a, YuvPixel *b) {
|
||||
const YuvPixel temp = *b;
|
||||
*b = *a;
|
||||
*a = temp;
|
||||
}
|
||||
|
||||
// Partitions array to find pivot index in qselect.
|
||||
static int partition(YuvPixel arr[], int left, int right, int pivot_idx) {
|
||||
YuvPixel pivot = arr[pivot_idx];
|
||||
|
||||
// Move pivot to the end.
|
||||
swap_yuv(&arr[pivot_idx], &arr[right]);
|
||||
|
||||
int p_idx = left;
|
||||
for (int i = left; i < right; ++i) {
|
||||
if (arr[i].y <= pivot.y) {
|
||||
swap_yuv(&arr[i], &arr[p_idx]);
|
||||
p_idx++;
|
||||
}
|
||||
}
|
||||
|
||||
swap_yuv(&arr[p_idx], &arr[right]);
|
||||
|
||||
return p_idx;
|
||||
}
|
||||
|
||||
// Returns the kth element in array, partially sorted in place (quickselect).
|
||||
static YuvPixel qselect(YuvPixel arr[], int left, int right, int k) {
|
||||
if (left >= right) {
|
||||
return arr[left];
|
||||
}
|
||||
unsigned int seed = (int)time(NULL);
|
||||
int pivot_idx = left + rand_r(&seed) % (right - left + 1);
|
||||
pivot_idx = partition(arr, left, right, pivot_idx);
|
||||
|
||||
if (k == pivot_idx) {
|
||||
return arr[k];
|
||||
} else if (k < pivot_idx) {
|
||||
return qselect(arr, left, pivot_idx - 1, k);
|
||||
} else {
|
||||
return qselect(arr, pivot_idx + 1, right, k);
|
||||
}
|
||||
}
|
||||
#endif // BGSPRITE_BLENDING_MODE == 0
|
||||
|
||||
// Stitches images together to create ARF and stores it in 'panorama'.
|
||||
static void stitch_images(YV12_BUFFER_CONFIG **const frames,
|
||||
const int num_frames, const int center_idx,
|
||||
const double **const params, const int *const x_min,
|
||||
const int *const x_max, const int *const y_min,
|
||||
const int *const y_max, int pano_x_min,
|
||||
int pano_x_max, int pano_y_min, int pano_y_max,
|
||||
YV12_BUFFER_CONFIG *panorama) {
|
||||
const int width = pano_x_max - pano_x_min + 1;
|
||||
const int height = pano_y_max - pano_y_min + 1;
|
||||
|
||||
// Create temp_pano[y][x][num_frames] stack of pixel values
|
||||
YuvPixel ***temp_pano = aom_malloc(height * sizeof(*temp_pano));
|
||||
for (int i = 0; i < height; ++i) {
|
||||
temp_pano[i] = aom_malloc(width * sizeof(**temp_pano));
|
||||
for (int j = 0; j < width; ++j) {
|
||||
temp_pano[i][j] = aom_malloc(num_frames * sizeof(***temp_pano));
|
||||
}
|
||||
}
|
||||
// Create count[y][x] to count how many values in stack for median filtering
|
||||
int **count = aom_malloc(height * sizeof(*count));
|
||||
for (int i = 0; i < height; ++i) {
|
||||
count[i] = aom_calloc(width, sizeof(**count)); // counts initialized to 0
|
||||
}
|
||||
|
||||
// Re-sample images onto panorama (pre-median filtering).
|
||||
const int x_offset = -pano_x_min;
|
||||
const int y_offset = -pano_y_min;
|
||||
const int frame_width = frames[0]->y_width;
|
||||
const int frame_height = frames[0]->y_height;
|
||||
for (int i = 0; i < num_frames; ++i) {
|
||||
// Find transforms from panorama coordinate system back to single image
|
||||
// coordinate system for sampling.
|
||||
int transformed_width = x_max[i] - x_min[i] + 1;
|
||||
int transformed_height = y_max[i] - y_min[i] + 1;
|
||||
|
||||
double transform_matrix[MAX_PARAMDIM];
|
||||
double transform_params[MAX_PARAMDIM - 1];
|
||||
invert_params(params[i], transform_params);
|
||||
params_to_matrix(transform_params, transform_matrix);
|
||||
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
const uint16_t *y_buffer16 = CONVERT_TO_SHORTPTR(frames[i]->y_buffer);
|
||||
const uint16_t *u_buffer16 = CONVERT_TO_SHORTPTR(frames[i]->u_buffer);
|
||||
const uint16_t *v_buffer16 = CONVERT_TO_SHORTPTR(frames[i]->v_buffer);
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
|
||||
for (int y = 0; y < transformed_height; ++y) {
|
||||
for (int x = 0; x < transformed_width; ++x) {
|
||||
// Do transform.
|
||||
double xy_matrix[3] = { x + x_min[i], y + y_min[i], 1 };
|
||||
double uv_matrix[3] = { 0 };
|
||||
multiply_mat(transform_matrix, xy_matrix, uv_matrix, TRANSFORM_MAT_DIM,
|
||||
TRANSFORM_MAT_DIM, 1);
|
||||
|
||||
// Coordinates used for nearest neighbor interpolation.
|
||||
int image_x = (int)round(uv_matrix[0]);
|
||||
int image_y = (int)round(uv_matrix[1]);
|
||||
|
||||
// Temporary values for bilinear interpolation
|
||||
double interpolated_yvalue = 0.0;
|
||||
double interpolated_uvalue = 0.0;
|
||||
double interpolated_vvalue = 0.0;
|
||||
double interpolated_fraction = 0.0;
|
||||
int interpolation_count = 0;
|
||||
|
||||
#if BGSPRITE_INTERPOLATION == 1
|
||||
// Coordintes used for bilinear interpolation.
|
||||
double x_base;
|
||||
double y_base;
|
||||
double x_decimal = modf(uv_matrix[0], &x_base);
|
||||
double y_decimal = modf(uv_matrix[1], &y_base);
|
||||
|
||||
if ((x_decimal > 0.2 && x_decimal < 0.8) ||
|
||||
(y_decimal > 0.2 && y_decimal < 0.8)) {
|
||||
for (int u = 0; u < 2; ++u) {
|
||||
for (int v = 0; v < 2; ++v) {
|
||||
int interp_x = (int)x_base + u;
|
||||
int interp_y = (int)y_base + v;
|
||||
if (interp_x >= 0 && interp_x < frame_width && interp_y >= 0 &&
|
||||
interp_y < frame_height) {
|
||||
interpolation_count++;
|
||||
|
||||
interpolated_fraction +=
|
||||
fabs(u - x_decimal) * fabs(v - y_decimal);
|
||||
int ychannel_idx = interp_y * frames[i]->y_stride + interp_x;
|
||||
int uvchannel_idx = (interp_y >> frames[i]->subsampling_y) *
|
||||
frames[i]->uv_stride +
|
||||
(interp_x >> frames[i]->subsampling_x);
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
if (frames[i]->flags & YV12_FLAG_HIGHBITDEPTH) {
|
||||
interpolated_yvalue += (1 - fabs(u - x_decimal)) *
|
||||
(1 - fabs(v - y_decimal)) *
|
||||
y_buffer16[ychannel_idx];
|
||||
interpolated_uvalue += (1 - fabs(u - x_decimal)) *
|
||||
(1 - fabs(v - y_decimal)) *
|
||||
u_buffer16[uvchannel_idx];
|
||||
interpolated_vvalue += (1 - fabs(u - x_decimal)) *
|
||||
(1 - fabs(v - y_decimal)) *
|
||||
v_buffer16[uvchannel_idx];
|
||||
} else {
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
interpolated_yvalue += (1 - fabs(u - x_decimal)) *
|
||||
(1 - fabs(v - y_decimal)) *
|
||||
frames[i]->y_buffer[ychannel_idx];
|
||||
interpolated_uvalue += (1 - fabs(u - x_decimal)) *
|
||||
(1 - fabs(v - y_decimal)) *
|
||||
frames[i]->u_buffer[uvchannel_idx];
|
||||
interpolated_vvalue += (1 - fabs(u - x_decimal)) *
|
||||
(1 - fabs(v - y_decimal)) *
|
||||
frames[i]->v_buffer[uvchannel_idx];
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
}
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
#endif // BGSPRITE_INTERPOLATION == 1
|
||||
|
||||
if (BGSPRITE_INTERPOLATION && interpolation_count > 2) {
|
||||
if (interpolation_count != 4) {
|
||||
interpolated_yvalue /= interpolated_fraction;
|
||||
interpolated_uvalue /= interpolated_fraction;
|
||||
interpolated_vvalue /= interpolated_fraction;
|
||||
}
|
||||
int pano_x = x + x_min[i] + x_offset;
|
||||
int pano_y = y + y_min[i] + y_offset;
|
||||
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
if (frames[i]->flags & YV12_FLAG_HIGHBITDEPTH) {
|
||||
temp_pano[pano_y][pano_x][count[pano_y][pano_x]].y =
|
||||
(uint16_t)interpolated_yvalue;
|
||||
temp_pano[pano_y][pano_x][count[pano_y][pano_x]].u =
|
||||
(uint16_t)interpolated_uvalue;
|
||||
temp_pano[pano_y][pano_x][count[pano_y][pano_x]].v =
|
||||
(uint16_t)interpolated_vvalue;
|
||||
} else {
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
temp_pano[pano_y][pano_x][count[pano_y][pano_x]].y =
|
||||
(uint8_t)interpolated_yvalue;
|
||||
temp_pano[pano_y][pano_x][count[pano_y][pano_x]].u =
|
||||
(uint8_t)interpolated_uvalue;
|
||||
temp_pano[pano_y][pano_x][count[pano_y][pano_x]].v =
|
||||
(uint8_t)interpolated_vvalue;
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
}
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
++count[pano_y][pano_x];
|
||||
} else if (image_x >= 0 && image_x < frame_width && image_y >= 0 &&
|
||||
image_y < frame_height) {
|
||||
// Place in panorama stack.
|
||||
int pano_x = x + x_min[i] + x_offset;
|
||||
int pano_y = y + y_min[i] + y_offset;
|
||||
|
||||
int ychannel_idx = image_y * frames[i]->y_stride + image_x;
|
||||
int uvchannel_idx =
|
||||
(image_y >> frames[i]->subsampling_y) * frames[i]->uv_stride +
|
||||
(image_x >> frames[i]->subsampling_x);
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
if (frames[i]->flags & YV12_FLAG_HIGHBITDEPTH) {
|
||||
temp_pano[pano_y][pano_x][count[pano_y][pano_x]].y =
|
||||
y_buffer16[ychannel_idx];
|
||||
temp_pano[pano_y][pano_x][count[pano_y][pano_x]].u =
|
||||
u_buffer16[uvchannel_idx];
|
||||
temp_pano[pano_y][pano_x][count[pano_y][pano_x]].v =
|
||||
v_buffer16[uvchannel_idx];
|
||||
} else {
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
temp_pano[pano_y][pano_x][count[pano_y][pano_x]].y =
|
||||
frames[i]->y_buffer[ychannel_idx];
|
||||
temp_pano[pano_y][pano_x][count[pano_y][pano_x]].u =
|
||||
frames[i]->u_buffer[uvchannel_idx];
|
||||
temp_pano[pano_y][pano_x][count[pano_y][pano_x]].v =
|
||||
frames[i]->v_buffer[uvchannel_idx];
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
}
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
++count[pano_y][pano_x];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#if BGSPRITE_BLENDING_MODE == 1
|
||||
// Apply mean filtering and store result in temp_pano[y][x][0].
|
||||
for (int y = 0; y < height; ++y) {
|
||||
for (int x = 0; x < width; ++x) {
|
||||
if (count[y][x] == 0) {
|
||||
// Just make the pixel black.
|
||||
// TODO(toddnguyen): Color the pixel with nearest neighbor
|
||||
} else {
|
||||
// Find
|
||||
uint32_t y_sum = 0;
|
||||
uint32_t u_sum = 0;
|
||||
uint32_t v_sum = 0;
|
||||
for (int i = 0; i < count[y][x]; ++i) {
|
||||
y_sum += temp_pano[y][x][i].y;
|
||||
u_sum += temp_pano[y][x][i].u;
|
||||
v_sum += temp_pano[y][x][i].v;
|
||||
}
|
||||
|
||||
const uint32_t unsigned_count = (uint32_t)count[y][x];
|
||||
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
if (panorama->flags & YV12_FLAG_HIGHBITDEPTH) {
|
||||
temp_pano[y][x][0].y = (uint16_t)OD_DIVU(y_sum, unsigned_count);
|
||||
temp_pano[y][x][0].u = (uint16_t)OD_DIVU(u_sum, unsigned_count);
|
||||
temp_pano[y][x][0].v = (uint16_t)OD_DIVU(v_sum, unsigned_count);
|
||||
} else {
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
temp_pano[y][x][0].y = (uint8_t)OD_DIVU(y_sum, unsigned_count);
|
||||
temp_pano[y][x][0].u = (uint8_t)OD_DIVU(u_sum, unsigned_count);
|
||||
temp_pano[y][x][0].v = (uint8_t)OD_DIVU(v_sum, unsigned_count);
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
}
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
}
|
||||
}
|
||||
}
|
||||
#else
|
||||
// Apply median filtering using quickselect.
|
||||
for (int y = 0; y < height; ++y) {
|
||||
for (int x = 0; x < width; ++x) {
|
||||
if (count[y][x] == 0) {
|
||||
// Just make the pixel black.
|
||||
// TODO(toddnguyen): Color the pixel with nearest neighbor
|
||||
} else {
|
||||
// Find
|
||||
const int median_idx = (int)floor(count[y][x] / 2);
|
||||
YuvPixel median =
|
||||
qselect(temp_pano[y][x], 0, count[y][x] - 1, median_idx);
|
||||
|
||||
// Make the median value the 0th index for UV subsampling later
|
||||
temp_pano[y][x][0] = median;
|
||||
assert(median.y == temp_pano[y][x][0].y &&
|
||||
median.u == temp_pano[y][x][0].u &&
|
||||
median.v == temp_pano[y][x][0].v);
|
||||
}
|
||||
}
|
||||
}
|
||||
#endif // BGSPRITE_BLENDING_MODE == 1
|
||||
|
||||
// NOTE(toddnguyen): Right now the ARF in the cpi struct is fixed size at
|
||||
// the same size as the frames. For now, we crop the generated panorama.
|
||||
// assert(panorama->y_width < width && panorama->y_height < height);
|
||||
const int crop_x_offset = x_min[center_idx] + x_offset;
|
||||
const int crop_y_offset = y_min[center_idx] + y_offset;
|
||||
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
if (panorama->flags & YV12_FLAG_HIGHBITDEPTH) {
|
||||
// Use median Y value.
|
||||
uint16_t *pano_y_buffer16 = CONVERT_TO_SHORTPTR(panorama->y_buffer);
|
||||
for (int y = 0; y < panorama->y_height; ++y) {
|
||||
for (int x = 0; x < panorama->y_width; ++x) {
|
||||
const int ychannel_idx = y * panorama->y_stride + x;
|
||||
if (count[y + crop_y_offset][x + crop_x_offset] > 0) {
|
||||
pano_y_buffer16[ychannel_idx] =
|
||||
temp_pano[y + crop_y_offset][x + crop_x_offset][0].y;
|
||||
} else {
|
||||
pano_y_buffer16[ychannel_idx] = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// UV subsampling with median UV values
|
||||
uint16_t *pano_u_buffer16 = CONVERT_TO_SHORTPTR(panorama->u_buffer);
|
||||
uint16_t *pano_v_buffer16 = CONVERT_TO_SHORTPTR(panorama->v_buffer);
|
||||
|
||||
for (int y = 0; y < panorama->uv_height; ++y) {
|
||||
for (int x = 0; x < panorama->uv_width; ++x) {
|
||||
uint32_t avg_count = 0;
|
||||
uint32_t u_sum = 0;
|
||||
uint32_t v_sum = 0;
|
||||
|
||||
// Look at surrounding pixels for subsampling
|
||||
for (int s_x = 0; s_x < panorama->subsampling_x + 1; ++s_x) {
|
||||
for (int s_y = 0; s_y < panorama->subsampling_y + 1; ++s_y) {
|
||||
int y_sample = crop_y_offset + (y << panorama->subsampling_y) + s_y;
|
||||
int x_sample = crop_x_offset + (x << panorama->subsampling_x) + s_x;
|
||||
if (y_sample > 0 && y_sample < height && x_sample > 0 &&
|
||||
x_sample < width && count[y_sample][x_sample] > 0) {
|
||||
u_sum += temp_pano[y_sample][x_sample][0].u;
|
||||
v_sum += temp_pano[y_sample][x_sample][0].v;
|
||||
avg_count++;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const int uvchannel_idx = y * panorama->uv_stride + x;
|
||||
if (avg_count != 0) {
|
||||
pano_u_buffer16[uvchannel_idx] = (uint16_t)OD_DIVU(u_sum, avg_count);
|
||||
pano_v_buffer16[uvchannel_idx] = (uint16_t)OD_DIVU(v_sum, avg_count);
|
||||
} else {
|
||||
pano_u_buffer16[uvchannel_idx] = 0;
|
||||
pano_v_buffer16[uvchannel_idx] = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
} else {
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
// Use median Y value.
|
||||
for (int y = 0; y < panorama->y_height; ++y) {
|
||||
for (int x = 0; x < panorama->y_width; ++x) {
|
||||
const int ychannel_idx = y * panorama->y_stride + x;
|
||||
if (count[y + crop_y_offset][x + crop_x_offset] > 0) {
|
||||
panorama->y_buffer[ychannel_idx] =
|
||||
temp_pano[y + crop_y_offset][x + crop_x_offset][0].y;
|
||||
} else {
|
||||
panorama->y_buffer[ychannel_idx] = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// UV subsampling with median UV values
|
||||
for (int y = 0; y < panorama->uv_height; ++y) {
|
||||
for (int x = 0; x < panorama->uv_width; ++x) {
|
||||
uint16_t avg_count = 0;
|
||||
uint16_t u_sum = 0;
|
||||
uint16_t v_sum = 0;
|
||||
|
||||
// Look at surrounding pixels for subsampling
|
||||
for (int s_x = 0; s_x < panorama->subsampling_x + 1; ++s_x) {
|
||||
for (int s_y = 0; s_y < panorama->subsampling_y + 1; ++s_y) {
|
||||
int y_sample = crop_y_offset + (y << panorama->subsampling_y) + s_y;
|
||||
int x_sample = crop_x_offset + (x << panorama->subsampling_x) + s_x;
|
||||
if (y_sample > 0 && y_sample < height && x_sample > 0 &&
|
||||
x_sample < width && count[y_sample][x_sample] > 0) {
|
||||
u_sum += temp_pano[y_sample][x_sample][0].u;
|
||||
v_sum += temp_pano[y_sample][x_sample][0].v;
|
||||
avg_count++;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const int uvchannel_idx = y * panorama->uv_stride + x;
|
||||
if (avg_count != 0) {
|
||||
panorama->u_buffer[uvchannel_idx] =
|
||||
(uint8_t)OD_DIVU(u_sum, avg_count);
|
||||
panorama->v_buffer[uvchannel_idx] =
|
||||
(uint8_t)OD_DIVU(v_sum, avg_count);
|
||||
} else {
|
||||
panorama->u_buffer[uvchannel_idx] = 0;
|
||||
panorama->v_buffer[uvchannel_idx] = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
}
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
|
||||
for (int i = 0; i < height; ++i) {
|
||||
for (int j = 0; j < width; ++j) {
|
||||
aom_free(temp_pano[i][j]);
|
||||
}
|
||||
aom_free(temp_pano[i]);
|
||||
aom_free(count[i]);
|
||||
}
|
||||
aom_free(count);
|
||||
aom_free(temp_pano);
|
||||
}
|
||||
|
||||
int av1_background_sprite(AV1_COMP *cpi, int distance) {
|
||||
YV12_BUFFER_CONFIG *frames[MAX_LAG_BUFFERS] = { NULL };
|
||||
static const double identity_params[MAX_PARAMDIM - 1] = {
|
||||
0.0, 0.0, 1.0, 0.0, 0.0, 1.0, 0.0, 0.0
|
||||
};
|
||||
|
||||
const int frames_after_arf =
|
||||
av1_lookahead_depth(cpi->lookahead) - distance - 1;
|
||||
int frames_fwd = (cpi->oxcf.arnr_max_frames - 1) >> 1;
|
||||
int frames_bwd;
|
||||
|
||||
// Define the forward and backwards filter limits for this arnr group.
|
||||
if (frames_fwd > frames_after_arf) frames_fwd = frames_after_arf;
|
||||
if (frames_fwd > distance) frames_fwd = distance;
|
||||
frames_bwd = frames_fwd;
|
||||
|
||||
#if CONFIG_EXT_REFS
|
||||
const GF_GROUP *const gf_group = &cpi->twopass.gf_group;
|
||||
if (gf_group->rf_level[gf_group->index] == GF_ARF_LOW) {
|
||||
cpi->alt_ref_buffer = av1_lookahead_peek(cpi->lookahead, distance)->img;
|
||||
cpi->is_arf_filter_off[gf_group->arf_update_idx[gf_group->index]] = 1;
|
||||
frames_fwd = 0;
|
||||
frames_bwd = 0;
|
||||
} else {
|
||||
cpi->is_arf_filter_off[gf_group->arf_update_idx[gf_group->index]] = 0;
|
||||
}
|
||||
#endif // CONFIG_EXT_REFS
|
||||
|
||||
const int start_frame = distance + frames_fwd;
|
||||
const int frames_to_stitch = frames_bwd + 1 + frames_fwd;
|
||||
|
||||
// Get frames to be included in background sprite.
|
||||
for (int frame = 0; frame < frames_to_stitch; ++frame) {
|
||||
const int which_buffer = start_frame - frame;
|
||||
struct lookahead_entry *buf =
|
||||
av1_lookahead_peek(cpi->lookahead, which_buffer);
|
||||
frames[frames_to_stitch - 1 - frame] = &buf->img;
|
||||
}
|
||||
|
||||
YV12_BUFFER_CONFIG temp_bg;
|
||||
memset(&temp_bg, 0, sizeof(temp_bg));
|
||||
aom_alloc_frame_buffer(&temp_bg, frames[0]->y_width, frames[0]->y_height,
|
||||
frames[0]->subsampling_x, frames[0]->subsampling_y,
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
frames[0]->flags & YV12_FLAG_HIGHBITDEPTH,
|
||||
#endif
|
||||
frames[0]->border, 0);
|
||||
aom_yv12_copy_frame(frames[0], &temp_bg);
|
||||
temp_bg.bit_depth = frames[0]->bit_depth;
|
||||
|
||||
// Allocate empty arrays for parameters between frames.
|
||||
double **params = aom_malloc(frames_to_stitch * sizeof(*params));
|
||||
for (int i = 0; i < frames_to_stitch; ++i) {
|
||||
params[i] = aom_malloc(sizeof(identity_params));
|
||||
memcpy(params[i], identity_params, sizeof(identity_params));
|
||||
}
|
||||
|
||||
// Use global motion to find affine transformations between frames.
|
||||
// params[i] will have the transform from frame[i] to frame[i-1].
|
||||
// params[0] will have the identity matrix because it has no previous frame.
|
||||
TransformationType model = AFFINE;
|
||||
int inliers_by_motion[RANSAC_NUM_MOTIONS];
|
||||
for (int frame = 0; frame < frames_to_stitch - 1; ++frame) {
|
||||
const int global_motion_ret = compute_global_motion_feature_based(
|
||||
model, frames[frame + 1], frames[frame],
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
cpi->common.bit_depth,
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
inliers_by_motion, params[frame + 1], RANSAC_NUM_MOTIONS);
|
||||
|
||||
// Quit if global motion had an error.
|
||||
if (global_motion_ret == 0) {
|
||||
for (int i = 0; i < frames_to_stitch; ++i) {
|
||||
aom_free(params[i]);
|
||||
}
|
||||
aom_free(params);
|
||||
return 1;
|
||||
}
|
||||
}
|
||||
|
||||
// Compound the transformation parameters.
|
||||
for (int i = 1; i < frames_to_stitch; ++i) {
|
||||
multiply_params(params[i - 1], params[i], params[i]);
|
||||
}
|
||||
|
||||
// Compute frame limits for final stitched images.
|
||||
int pano_x_max = INT_MIN;
|
||||
int pano_x_min = INT_MAX;
|
||||
int pano_y_max = INT_MIN;
|
||||
int pano_y_min = INT_MAX;
|
||||
int *x_max = aom_malloc(frames_to_stitch * sizeof(*x_max));
|
||||
int *x_min = aom_malloc(frames_to_stitch * sizeof(*x_min));
|
||||
int *y_max = aom_malloc(frames_to_stitch * sizeof(*y_max));
|
||||
int *y_min = aom_malloc(frames_to_stitch * sizeof(*y_min));
|
||||
|
||||
find_limits(cpi->initial_width, cpi->initial_height,
|
||||
(const double **const)params, frames_to_stitch, x_min, x_max,
|
||||
y_min, y_max, &pano_x_min, &pano_x_max, &pano_y_min, &pano_y_max);
|
||||
|
||||
// Center panorama on the ARF.
|
||||
const int center_idx = frames_bwd;
|
||||
assert(center_idx >= 0 && center_idx < frames_to_stitch);
|
||||
|
||||
// Recompute transformations to adjust to center image.
|
||||
// Invert center image's transform.
|
||||
double inverse[MAX_PARAMDIM - 1] = { 0 };
|
||||
invert_params(params[center_idx], inverse);
|
||||
|
||||
// Multiply the inverse to all transformation parameters.
|
||||
for (int i = 0; i < frames_to_stitch; ++i) {
|
||||
multiply_params(inverse, params[i], params[i]);
|
||||
}
|
||||
|
||||
// Recompute frame limits for new adjusted center.
|
||||
find_limits(cpi->initial_width, cpi->initial_height,
|
||||
(const double **const)params, frames_to_stitch, x_min, x_max,
|
||||
y_min, y_max, &pano_x_min, &pano_x_max, &pano_y_min, &pano_y_max);
|
||||
|
||||
// Stitch Images.
|
||||
stitch_images(frames, frames_to_stitch, center_idx,
|
||||
(const double **const)params, x_min, x_max, y_min, y_max,
|
||||
pano_x_min, pano_x_max, pano_y_min, pano_y_max, &temp_bg);
|
||||
|
||||
// Apply temporal filter.
|
||||
av1_temporal_filter(cpi, &temp_bg, distance);
|
||||
|
||||
// Free memory.
|
||||
aom_free_frame_buffer(&temp_bg);
|
||||
for (int i = 0; i < frames_to_stitch; ++i) {
|
||||
aom_free(params[i]);
|
||||
}
|
||||
aom_free(params);
|
||||
aom_free(x_max);
|
||||
aom_free(x_min);
|
||||
aom_free(y_max);
|
||||
aom_free(y_min);
|
||||
|
||||
return 0;
|
||||
}
|
||||
30
third_party/aom/av1/encoder/bgsprite.h
vendored
Normal file
30
third_party/aom/av1/encoder/bgsprite.h
vendored
Normal file
|
|
@ -0,0 +1,30 @@
|
|||
/*
|
||||
* Copyright (c) 2017, Alliance for Open Media. All rights reserved
|
||||
*
|
||||
* This source code is subject to the terms of the BSD 2 Clause License and
|
||||
* the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
|
||||
* was not distributed with this source code in the LICENSE file, you can
|
||||
* obtain it at www.aomedia.org/license/software. If the Alliance for Open
|
||||
* Media Patent License 1.0 was not distributed with this source code in the
|
||||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#ifndef AV1_ENCODER_BGSPRITE_H_
|
||||
#define AV1_ENCODER_BGSPRITE_H_
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
|
||||
#include "av1/encoder/encoder.h"
|
||||
|
||||
// Creates alternate reference frame staring from source image + frames up to
|
||||
// 'distance' past source frame.
|
||||
// Returns 0 on success and 1 on failure.
|
||||
int av1_background_sprite(AV1_COMP *cpi, int distance);
|
||||
|
||||
#ifdef __cplusplus
|
||||
} // extern "C"
|
||||
#endif
|
||||
|
||||
#endif // AV1_ENCODER_BGSPRITE_H_
|
||||
2678
third_party/aom/av1/encoder/bitstream.c
vendored
2678
third_party/aom/av1/encoder/bitstream.c
vendored
File diff suppressed because it is too large
Load diff
9
third_party/aom/av1/encoder/bitstream.h
vendored
9
third_party/aom/av1/encoder/bitstream.h
vendored
|
|
@ -19,7 +19,11 @@ extern "C" {
|
|||
#include "av1/encoder/encoder.h"
|
||||
|
||||
#if CONFIG_REFERENCE_BUFFER
|
||||
void write_sequence_header(SequenceHeader *seq_params);
|
||||
void write_sequence_header(
|
||||
#if CONFIG_EXT_TILE
|
||||
AV1_COMMON *const cm,
|
||||
#endif // CONFIG_EXT_TILE
|
||||
SequenceHeader *seq_params);
|
||||
#endif
|
||||
|
||||
void av1_pack_bitstream(AV1_COMP *const cpi, uint8_t *dest, size_t *size);
|
||||
|
|
@ -42,7 +46,8 @@ void av1_write_tx_type(const AV1_COMMON *const cm, const MACROBLOCKD *xd,
|
|||
const int supertx_enabled,
|
||||
#endif
|
||||
#if CONFIG_TXK_SEL
|
||||
int block, int plane,
|
||||
int blk_row, int blk_col, int block, int plane,
|
||||
TX_SIZE tx_size,
|
||||
#endif
|
||||
aom_writer *w);
|
||||
|
||||
|
|
|
|||
14
third_party/aom/av1/encoder/block.h
vendored
14
third_party/aom/av1/encoder/block.h
vendored
|
|
@ -116,7 +116,6 @@ struct macroblock {
|
|||
// The equivalend SAD error of one (whole) bit at the current quantizer
|
||||
// for sub-8x8 blocks.
|
||||
int sadperbit4;
|
||||
int rddiv;
|
||||
int rdmult;
|
||||
int mb_energy;
|
||||
int *m_search_count_ptr;
|
||||
|
|
@ -206,16 +205,15 @@ struct macroblock {
|
|||
int pvq_speed;
|
||||
int pvq_coded; // Indicates whether pvq_info needs be stored to tokenize
|
||||
#endif
|
||||
#if CONFIG_DAALA_DIST
|
||||
// Keep rate of each 4x4 block in the current macroblock during RDO
|
||||
// This is needed when using the 8x8 Daala distortion metric during RDO,
|
||||
// because it evaluates distortion in a different order than the underlying
|
||||
// 4x4 blocks are coded.
|
||||
int rate_4x4[MAX_SB_SQUARE / (TX_SIZE_W_MIN * TX_SIZE_H_MIN)];
|
||||
#if CONFIG_DIST_8X8
|
||||
#if CONFIG_CB4X4
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
DECLARE_ALIGNED(16, uint16_t, decoded_8x8[8 * 8]);
|
||||
#else
|
||||
DECLARE_ALIGNED(16, uint8_t, decoded_8x8[8 * 8]);
|
||||
#endif
|
||||
#endif // CONFIG_CB4X4
|
||||
#endif // CONFIG_DAALA_DIST
|
||||
#endif // CONFIG_DIST_8X8
|
||||
#if CONFIG_CFL
|
||||
// Whether luma needs to be stored during RDO.
|
||||
int cfl_store_y;
|
||||
|
|
|
|||
26
third_party/aom/av1/encoder/context_tree.c
vendored
26
third_party/aom/av1/encoder/context_tree.c
vendored
|
|
@ -65,12 +65,10 @@ static void alloc_mode_context(AV1_COMMON *cm, int num_4x4_blk,
|
|||
}
|
||||
|
||||
#if CONFIG_PALETTE
|
||||
if (cm->allow_screen_content_tools) {
|
||||
for (i = 0; i < 2; ++i) {
|
||||
CHECK_MEM_ERROR(
|
||||
cm, ctx->color_index_map[i],
|
||||
aom_memalign(32, num_pix * sizeof(*ctx->color_index_map[i])));
|
||||
}
|
||||
for (i = 0; i < 2; ++i) {
|
||||
CHECK_MEM_ERROR(
|
||||
cm, ctx->color_index_map[i],
|
||||
aom_memalign(32, num_pix * sizeof(*ctx->color_index_map[i])));
|
||||
}
|
||||
#endif // CONFIG_PALETTE
|
||||
}
|
||||
|
|
@ -141,7 +139,13 @@ static void alloc_tree_contexts(AV1_COMMON *cm, PC_TREE *tree,
|
|||
&tree->verticalb[1]);
|
||||
alloc_mode_context(cm, num_4x4_blk / 4, PARTITION_VERT_B,
|
||||
&tree->verticalb[2]);
|
||||
#ifdef CONFIG_SUPERTX
|
||||
for (int i = 0; i < 4; ++i) {
|
||||
alloc_mode_context(cm, num_4x4_blk / 4, PARTITION_HORZ_4,
|
||||
&tree->horizontal4[i]);
|
||||
alloc_mode_context(cm, num_4x4_blk / 4, PARTITION_HORZ_4,
|
||||
&tree->vertical4[i]);
|
||||
}
|
||||
#if CONFIG_SUPERTX
|
||||
alloc_mode_context(cm, num_4x4_blk, PARTITION_HORZ,
|
||||
&tree->horizontal_supertx);
|
||||
alloc_mode_context(cm, num_4x4_blk, PARTITION_VERT, &tree->vertical_supertx);
|
||||
|
|
@ -159,7 +163,7 @@ static void alloc_tree_contexts(AV1_COMMON *cm, PC_TREE *tree,
|
|||
alloc_mode_context(cm, num_4x4_blk, &tree->none);
|
||||
alloc_mode_context(cm, num_4x4_blk / 2, &tree->horizontal[0]);
|
||||
alloc_mode_context(cm, num_4x4_blk / 2, &tree->vertical[0]);
|
||||
#ifdef CONFIG_SUPERTX
|
||||
#if CONFIG_SUPERTX
|
||||
alloc_mode_context(cm, num_4x4_blk, &tree->horizontal_supertx);
|
||||
alloc_mode_context(cm, num_4x4_blk, &tree->vertical_supertx);
|
||||
alloc_mode_context(cm, num_4x4_blk, &tree->split_supertx);
|
||||
|
|
@ -184,13 +188,17 @@ static void free_tree_contexts(PC_TREE *tree) {
|
|||
free_mode_context(&tree->verticala[i]);
|
||||
free_mode_context(&tree->verticalb[i]);
|
||||
}
|
||||
for (i = 0; i < 4; ++i) {
|
||||
free_mode_context(&tree->horizontal4[i]);
|
||||
free_mode_context(&tree->vertical4[i]);
|
||||
}
|
||||
#endif // CONFIG_EXT_PARTITION_TYPES
|
||||
free_mode_context(&tree->none);
|
||||
free_mode_context(&tree->horizontal[0]);
|
||||
free_mode_context(&tree->horizontal[1]);
|
||||
free_mode_context(&tree->vertical[0]);
|
||||
free_mode_context(&tree->vertical[1]);
|
||||
#ifdef CONFIG_SUPERTX
|
||||
#if CONFIG_SUPERTX
|
||||
free_mode_context(&tree->horizontal_supertx);
|
||||
free_mode_context(&tree->vertical_supertx);
|
||||
free_mode_context(&tree->split_supertx);
|
||||
|
|
|
|||
4
third_party/aom/av1/encoder/context_tree.h
vendored
4
third_party/aom/av1/encoder/context_tree.h
vendored
|
|
@ -81,12 +81,14 @@ typedef struct PC_TREE {
|
|||
PICK_MODE_CONTEXT horizontalb[3];
|
||||
PICK_MODE_CONTEXT verticala[3];
|
||||
PICK_MODE_CONTEXT verticalb[3];
|
||||
PICK_MODE_CONTEXT horizontal4[4];
|
||||
PICK_MODE_CONTEXT vertical4[4];
|
||||
#endif
|
||||
union {
|
||||
struct PC_TREE *split[4];
|
||||
PICK_MODE_CONTEXT *leaf_split[4];
|
||||
};
|
||||
#ifdef CONFIG_SUPERTX
|
||||
#if CONFIG_SUPERTX
|
||||
PICK_MODE_CONTEXT horizontal_supertx;
|
||||
PICK_MODE_CONTEXT vertical_supertx;
|
||||
PICK_MODE_CONTEXT split_supertx;
|
||||
|
|
|
|||
18
third_party/aom/av1/encoder/cost.c
vendored
18
third_party/aom/av1/encoder/cost.c
vendored
|
|
@ -65,3 +65,21 @@ void av1_cost_tokens_skip(int *costs, const aom_prob *probs, aom_tree tree) {
|
|||
costs[-tree[0]] = av1_cost_bit(probs[0], 0);
|
||||
cost(costs, tree, probs, 2, 0);
|
||||
}
|
||||
|
||||
void av1_cost_tokens_from_cdf(int *costs, const aom_cdf_prob *cdf,
|
||||
const int *inv_map) {
|
||||
int i;
|
||||
aom_cdf_prob prev_cdf = 0;
|
||||
for (i = 0;; ++i) {
|
||||
const aom_cdf_prob p15 = AOM_ICDF(cdf[i]) - prev_cdf;
|
||||
prev_cdf = AOM_ICDF(cdf[i]);
|
||||
|
||||
if (inv_map)
|
||||
costs[inv_map[i]] = av1_cost_symbol(p15);
|
||||
else
|
||||
costs[i] = av1_cost_symbol(p15);
|
||||
|
||||
// Stop once we reach the end of the CDF
|
||||
if (cdf[i] == AOM_ICDF(CDF_PROB_TOP)) break;
|
||||
}
|
||||
}
|
||||
|
|
|
|||
10
third_party/aom/av1/encoder/cost.h
vendored
10
third_party/aom/av1/encoder/cost.h
vendored
|
|
@ -34,6 +34,14 @@ extern const uint16_t av1_prob_cost[256];
|
|||
// for each bit.
|
||||
#define av1_cost_literal(n) ((n) * (1 << AV1_PROB_COST_SHIFT))
|
||||
|
||||
// Calculate the cost of a symbol with probability p15 / 2^15
|
||||
static INLINE int av1_cost_symbol(aom_cdf_prob p15) {
|
||||
assert(0 < p15 && p15 < CDF_PROB_TOP);
|
||||
const int shift = CDF_PROB_BITS - 1 - get_msb(p15);
|
||||
return av1_cost_zero(get_prob(p15 << shift, CDF_PROB_TOP)) +
|
||||
av1_cost_literal(shift);
|
||||
}
|
||||
|
||||
static INLINE unsigned int cost_branch256(const unsigned int ct[2],
|
||||
aom_prob p) {
|
||||
return ct[0] * av1_cost_zero(p) + ct[1] * av1_cost_one(p);
|
||||
|
|
@ -55,6 +63,8 @@ static INLINE int treed_cost(aom_tree tree, const aom_prob *probs, int bits,
|
|||
|
||||
void av1_cost_tokens(int *costs, const aom_prob *probs, aom_tree tree);
|
||||
void av1_cost_tokens_skip(int *costs, const aom_prob *probs, aom_tree tree);
|
||||
void av1_cost_tokens_from_cdf(int *costs, const aom_cdf_prob *cdf,
|
||||
const int *inv_map);
|
||||
|
||||
#ifdef __cplusplus
|
||||
} // extern "C"
|
||||
|
|
|
|||
606
third_party/aom/av1/encoder/dct.c
vendored
606
third_party/aom/av1/encoder/dct.c
vendored
|
|
@ -21,6 +21,9 @@
|
|||
#include "av1/common/av1_fwd_txfm1d.h"
|
||||
#include "av1/common/av1_fwd_txfm1d_cfg.h"
|
||||
#include "av1/common/idct.h"
|
||||
#if CONFIG_DAALA_DCT4 || CONFIG_DAALA_DCT8
|
||||
#include "av1/common/daala_tx.h"
|
||||
#endif
|
||||
|
||||
static INLINE void range_check(const tran_low_t *input, const int size,
|
||||
const int bit) {
|
||||
|
|
@ -39,6 +42,18 @@ static INLINE void range_check(const tran_low_t *input, const int size,
|
|||
#endif
|
||||
}
|
||||
|
||||
#if CONFIG_DAALA_DCT4
|
||||
static void fdct4(const tran_low_t *input, tran_low_t *output) {
|
||||
int i;
|
||||
od_coeff x[4];
|
||||
od_coeff y[4];
|
||||
for (i = 0; i < 4; i++) x[i] = (od_coeff)input[i];
|
||||
od_bin_fdct4(y, x, 1);
|
||||
for (i = 0; i < 4; i++) output[i] = (tran_low_t)y[i];
|
||||
}
|
||||
|
||||
#else
|
||||
|
||||
static void fdct4(const tran_low_t *input, tran_low_t *output) {
|
||||
tran_high_t temp;
|
||||
tran_low_t step[4];
|
||||
|
|
@ -74,6 +89,19 @@ static void fdct4(const tran_low_t *input, tran_low_t *output) {
|
|||
|
||||
range_check(output, 4, 16);
|
||||
}
|
||||
#endif
|
||||
|
||||
#if CONFIG_DAALA_DCT8
|
||||
static void fdct8(const tran_low_t *input, tran_low_t *output) {
|
||||
int i;
|
||||
od_coeff x[8];
|
||||
od_coeff y[8];
|
||||
for (i = 0; i < 8; i++) x[i] = (od_coeff)input[i];
|
||||
od_bin_fdct8(y, x, 1);
|
||||
for (i = 0; i < 8; i++) output[i] = (tran_low_t)y[i];
|
||||
}
|
||||
|
||||
#else
|
||||
|
||||
static void fdct8(const tran_low_t *input, tran_low_t *output) {
|
||||
tran_high_t temp;
|
||||
|
|
@ -152,6 +180,7 @@ static void fdct8(const tran_low_t *input, tran_low_t *output) {
|
|||
|
||||
range_check(output, 8, 16);
|
||||
}
|
||||
#endif
|
||||
|
||||
static void fdct16(const tran_low_t *input, tran_low_t *output) {
|
||||
tran_high_t temp;
|
||||
|
|
@ -767,6 +796,18 @@ static void fadst4(const tran_low_t *input, tran_low_t *output) {
|
|||
output[3] = (tran_low_t)fdct_round_shift(s3);
|
||||
}
|
||||
|
||||
#if CONFIG_DAALA_DCT8
|
||||
static void fadst8(const tran_low_t *input, tran_low_t *output) {
|
||||
int i;
|
||||
od_coeff x[8];
|
||||
od_coeff y[8];
|
||||
for (i = 0; i < 8; i++) x[i] = (od_coeff)input[i];
|
||||
od_bin_fdst8(y, x, 1);
|
||||
for (i = 0; i < 8; i++) output[i] = (tran_low_t)y[i];
|
||||
}
|
||||
|
||||
#else
|
||||
|
||||
static void fadst8(const tran_low_t *input, tran_low_t *output) {
|
||||
tran_high_t s0, s1, s2, s3, s4, s5, s6, s7;
|
||||
|
||||
|
|
@ -837,6 +878,7 @@ static void fadst8(const tran_low_t *input, tran_low_t *output) {
|
|||
output[6] = (tran_low_t)x5;
|
||||
output[7] = (tran_low_t)-x1;
|
||||
}
|
||||
#endif
|
||||
|
||||
static void fadst16(const tran_low_t *input, tran_low_t *output) {
|
||||
tran_high_t s0, s1, s2, s3, s4, s5, s6, s7, s8;
|
||||
|
|
@ -1021,6 +1063,83 @@ static void fhalfright32(const tran_low_t *input, tran_low_t *output) {
|
|||
// Note overall scaling factor is 4 times orthogonal
|
||||
}
|
||||
|
||||
#if CONFIG_MRC_TX
|
||||
static void get_masked_residual32(const int16_t **input, int *input_stride,
|
||||
const uint8_t *pred, int pred_stride,
|
||||
int16_t *masked_input) {
|
||||
int mrc_mask[32 * 32];
|
||||
get_mrc_mask(pred, pred_stride, mrc_mask, 32, 32, 32);
|
||||
int32_t sum = 0;
|
||||
int16_t avg;
|
||||
// Get the masked average of the prediction
|
||||
for (int i = 0; i < 32; ++i) {
|
||||
for (int j = 0; j < 32; ++j) {
|
||||
sum += mrc_mask[i * 32 + j] * (*input)[i * (*input_stride) + j];
|
||||
}
|
||||
}
|
||||
avg = ROUND_POWER_OF_TWO_SIGNED(sum, 10);
|
||||
// Replace all of the unmasked pixels in the prediction with the average
|
||||
// of the masked pixels
|
||||
for (int i = 0; i < 32; ++i) {
|
||||
for (int j = 0; j < 32; ++j)
|
||||
masked_input[i * 32 + j] =
|
||||
(mrc_mask[i * 32 + j]) ? (*input)[i * (*input_stride) + j] : avg;
|
||||
}
|
||||
*input = masked_input;
|
||||
*input_stride = 32;
|
||||
}
|
||||
#endif // CONFIG_MRC_TX
|
||||
|
||||
#if CONFIG_LGT
|
||||
static void flgt4(const tran_low_t *input, tran_low_t *output,
|
||||
const tran_high_t *lgtmtx) {
|
||||
if (!(input[0] | input[1] | input[2] | input[3])) {
|
||||
output[0] = output[1] = output[2] = output[3] = 0;
|
||||
return;
|
||||
}
|
||||
|
||||
// evaluate s[j] = sum of all lgtmtx[j][i]*input[i] over i=1,...,4
|
||||
tran_high_t s[4] = { 0 };
|
||||
for (int i = 0; i < 4; ++i)
|
||||
for (int j = 0; j < 4; ++j) s[j] += lgtmtx[j * 4 + i] * input[i];
|
||||
|
||||
for (int i = 0; i < 4; ++i) output[i] = (tran_low_t)fdct_round_shift(s[i]);
|
||||
}
|
||||
|
||||
static void flgt8(const tran_low_t *input, tran_low_t *output,
|
||||
const tran_high_t *lgtmtx) {
|
||||
// evaluate s[j] = sum of all lgtmtx[j][i]*input[i] over i=1,...,8
|
||||
tran_high_t s[8] = { 0 };
|
||||
for (int i = 0; i < 8; ++i)
|
||||
for (int j = 0; j < 8; ++j) s[j] += lgtmtx[j * 8 + i] * input[i];
|
||||
|
||||
for (int i = 0; i < 8; ++i) output[i] = (tran_low_t)fdct_round_shift(s[i]);
|
||||
}
|
||||
|
||||
// The get_fwd_lgt functions return 1 if LGT is chosen to apply, and 0 otherwise
|
||||
int get_fwd_lgt4(transform_1d tx_orig, TxfmParam *txfm_param,
|
||||
const tran_high_t *lgtmtx[], int ntx) {
|
||||
// inter/intra split
|
||||
if (tx_orig == &fadst4) {
|
||||
for (int i = 0; i < ntx; ++i)
|
||||
lgtmtx[i] = txfm_param->is_inter ? &lgt4_170[0][0] : &lgt4_140[0][0];
|
||||
return 1;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
int get_fwd_lgt8(transform_1d tx_orig, TxfmParam *txfm_param,
|
||||
const tran_high_t *lgtmtx[], int ntx) {
|
||||
// inter/intra split
|
||||
if (tx_orig == &fadst8) {
|
||||
for (int i = 0; i < ntx; ++i)
|
||||
lgtmtx[i] = txfm_param->is_inter ? &lgt8_170[0][0] : &lgt8_150[0][0];
|
||||
return 1;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
#endif // CONFIG_LGT
|
||||
|
||||
#if CONFIG_EXT_TX
|
||||
// TODO(sarahparker) these functions will be removed once the highbitdepth
|
||||
// codepath works properly for rectangular transforms. They have almost
|
||||
|
|
@ -1028,13 +1147,24 @@ static void fhalfright32(const tran_low_t *input, tran_low_t *output) {
|
|||
// being used for square transforms.
|
||||
static void fidtx4(const tran_low_t *input, tran_low_t *output) {
|
||||
int i;
|
||||
for (i = 0; i < 4; ++i)
|
||||
for (i = 0; i < 4; ++i) {
|
||||
#if CONFIG_DAALA_DCT4
|
||||
output[i] = input[i];
|
||||
#else
|
||||
output[i] = (tran_low_t)fdct_round_shift(input[i] * Sqrt2);
|
||||
#endif
|
||||
}
|
||||
}
|
||||
|
||||
static void fidtx8(const tran_low_t *input, tran_low_t *output) {
|
||||
int i;
|
||||
for (i = 0; i < 8; ++i) output[i] = input[i] * 2;
|
||||
for (i = 0; i < 8; ++i) {
|
||||
#if CONFIG_DAALA_DCT8
|
||||
output[i] = input[i];
|
||||
#else
|
||||
output[i] = input[i] * 2;
|
||||
#endif
|
||||
}
|
||||
}
|
||||
|
||||
static void fidtx16(const tran_low_t *input, tran_low_t *output) {
|
||||
|
|
@ -1110,6 +1240,9 @@ static void copy_fliplrud(const int16_t *src, int src_stride, int l, int w,
|
|||
static void maybe_flip_input(const int16_t **src, int *src_stride, int l, int w,
|
||||
int16_t *buff, int tx_type) {
|
||||
switch (tx_type) {
|
||||
#if CONFIG_MRC_TX
|
||||
case MRC_DCT:
|
||||
#endif // CONFIG_MRC_TX
|
||||
case DCT_DCT:
|
||||
case ADST_DCT:
|
||||
case DCT_ADST:
|
||||
|
|
@ -1144,10 +1277,21 @@ static void maybe_flip_input(const int16_t **src, int *src_stride, int l, int w,
|
|||
#endif // CONFIG_EXT_TX
|
||||
|
||||
void av1_fht4x4_c(const int16_t *input, tran_low_t *output, int stride,
|
||||
int tx_type) {
|
||||
TxfmParam *txfm_param) {
|
||||
int tx_type = txfm_param->tx_type;
|
||||
#if CONFIG_MRC_TX
|
||||
assert(tx_type != MRC_DCT && "Invalid tx type for tx size");
|
||||
#endif
|
||||
#if CONFIG_DCT_ONLY
|
||||
assert(tx_type == DCT_DCT);
|
||||
#endif
|
||||
#if !CONFIG_DAALA_DCT4
|
||||
if (tx_type == DCT_DCT) {
|
||||
aom_fdct4x4_c(input, output, stride);
|
||||
} else {
|
||||
return;
|
||||
}
|
||||
#endif
|
||||
{
|
||||
static const transform_2d FHT[] = {
|
||||
{ fdct4, fdct4 }, // DCT_DCT
|
||||
{ fadst4, fdct4 }, // ADST_DCT
|
||||
|
|
@ -1166,7 +1310,7 @@ void av1_fht4x4_c(const int16_t *input, tran_low_t *output, int stride,
|
|||
{ fidtx4, fadst4 }, // H_ADST
|
||||
{ fadst4, fidtx4 }, // V_FLIPADST
|
||||
{ fidtx4, fadst4 }, // H_FLIPADST
|
||||
#endif // CONFIG_EXT_TX
|
||||
#endif
|
||||
};
|
||||
const transform_2d ht = FHT[tx_type];
|
||||
tran_low_t out[4 * 4];
|
||||
|
|
@ -1178,25 +1322,60 @@ void av1_fht4x4_c(const int16_t *input, tran_low_t *output, int stride,
|
|||
maybe_flip_input(&input, &stride, 4, 4, flipped_input, tx_type);
|
||||
#endif
|
||||
|
||||
#if CONFIG_LGT
|
||||
// Choose LGT adaptive to the prediction. We may apply different LGTs for
|
||||
// different rows/columns, indicated by the pointers to 2D arrays
|
||||
const tran_high_t *lgtmtx_col[4];
|
||||
const tran_high_t *lgtmtx_row[4];
|
||||
int use_lgt_col = get_fwd_lgt4(ht.cols, txfm_param, lgtmtx_col, 4);
|
||||
int use_lgt_row = get_fwd_lgt4(ht.rows, txfm_param, lgtmtx_row, 4);
|
||||
#endif
|
||||
|
||||
// Columns
|
||||
for (i = 0; i < 4; ++i) {
|
||||
/* A C99-safe upshift by 4 for both Daala and VPx TX. */
|
||||
for (j = 0; j < 4; ++j) temp_in[j] = input[j * stride + i] * 16;
|
||||
#if !CONFIG_DAALA_DCT4
|
||||
if (i == 0 && temp_in[0]) temp_in[0] += 1;
|
||||
ht.cols(temp_in, temp_out);
|
||||
#endif
|
||||
#if CONFIG_LGT
|
||||
if (use_lgt_col)
|
||||
flgt4(temp_in, temp_out, lgtmtx_col[i]);
|
||||
else
|
||||
#endif
|
||||
ht.cols(temp_in, temp_out);
|
||||
for (j = 0; j < 4; ++j) out[j * 4 + i] = temp_out[j];
|
||||
}
|
||||
|
||||
// Rows
|
||||
for (i = 0; i < 4; ++i) {
|
||||
for (j = 0; j < 4; ++j) temp_in[j] = out[j + i * 4];
|
||||
ht.rows(temp_in, temp_out);
|
||||
#if CONFIG_LGT
|
||||
if (use_lgt_row)
|
||||
flgt4(temp_in, temp_out, lgtmtx_row[i]);
|
||||
else
|
||||
#endif
|
||||
ht.rows(temp_in, temp_out);
|
||||
#if CONFIG_DAALA_DCT4
|
||||
/* Daala TX has orthonormal scaling; shift down by only 1 to achieve
|
||||
the usual VPx coefficient left-shift of 3. */
|
||||
for (j = 0; j < 4; ++j) output[j + i * 4] = temp_out[j] >> 1;
|
||||
#else
|
||||
for (j = 0; j < 4; ++j) output[j + i * 4] = (temp_out[j] + 1) >> 2;
|
||||
#endif
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void av1_fht4x8_c(const int16_t *input, tran_low_t *output, int stride,
|
||||
int tx_type) {
|
||||
TxfmParam *txfm_param) {
|
||||
int tx_type = txfm_param->tx_type;
|
||||
#if CONFIG_MRC_TX
|
||||
assert(tx_type != MRC_DCT && "Invalid tx type for tx size");
|
||||
#endif // CONFIG_MRC_TX
|
||||
#if CONFIG_DCT_ONLY
|
||||
assert(tx_type == DCT_DCT);
|
||||
#endif
|
||||
static const transform_2d FHT[] = {
|
||||
{ fdct8, fdct4 }, // DCT_DCT
|
||||
{ fadst8, fdct4 }, // ADST_DCT
|
||||
|
|
@ -1228,19 +1407,36 @@ void av1_fht4x8_c(const int16_t *input, tran_low_t *output, int stride,
|
|||
maybe_flip_input(&input, &stride, n2, n, flipped_input, tx_type);
|
||||
#endif
|
||||
|
||||
#if CONFIG_LGT
|
||||
const tran_high_t *lgtmtx_col[4];
|
||||
const tran_high_t *lgtmtx_row[8];
|
||||
int use_lgt_col = get_fwd_lgt8(ht.cols, txfm_param, lgtmtx_col, 4);
|
||||
int use_lgt_row = get_fwd_lgt4(ht.rows, txfm_param, lgtmtx_row, 8);
|
||||
#endif
|
||||
|
||||
// Rows
|
||||
for (i = 0; i < n2; ++i) {
|
||||
for (j = 0; j < n; ++j)
|
||||
temp_in[j] =
|
||||
(tran_low_t)fdct_round_shift(input[i * stride + j] * 4 * Sqrt2);
|
||||
ht.rows(temp_in, temp_out);
|
||||
#if CONFIG_LGT
|
||||
if (use_lgt_row)
|
||||
flgt4(temp_in, temp_out, lgtmtx_row[i]);
|
||||
else
|
||||
#endif
|
||||
ht.rows(temp_in, temp_out);
|
||||
for (j = 0; j < n; ++j) out[j * n2 + i] = temp_out[j];
|
||||
}
|
||||
|
||||
// Columns
|
||||
for (i = 0; i < n; ++i) {
|
||||
for (j = 0; j < n2; ++j) temp_in[j] = out[j + i * n2];
|
||||
ht.cols(temp_in, temp_out);
|
||||
#if CONFIG_LGT
|
||||
if (use_lgt_col)
|
||||
flgt8(temp_in, temp_out, lgtmtx_col[i]);
|
||||
else
|
||||
#endif
|
||||
ht.cols(temp_in, temp_out);
|
||||
for (j = 0; j < n2; ++j)
|
||||
output[i + j * n] = (temp_out[j] + (temp_out[j] < 0)) >> 1;
|
||||
}
|
||||
|
|
@ -1248,7 +1444,14 @@ void av1_fht4x8_c(const int16_t *input, tran_low_t *output, int stride,
|
|||
}
|
||||
|
||||
void av1_fht8x4_c(const int16_t *input, tran_low_t *output, int stride,
|
||||
int tx_type) {
|
||||
TxfmParam *txfm_param) {
|
||||
int tx_type = txfm_param->tx_type;
|
||||
#if CONFIG_MRC_TX
|
||||
assert(tx_type != MRC_DCT && "Invalid tx type for tx size");
|
||||
#endif // CONFIG_MRC_TX
|
||||
#if CONFIG_DCT_ONLY
|
||||
assert(tx_type == DCT_DCT);
|
||||
#endif
|
||||
static const transform_2d FHT[] = {
|
||||
{ fdct4, fdct8 }, // DCT_DCT
|
||||
{ fadst4, fdct8 }, // ADST_DCT
|
||||
|
|
@ -1280,19 +1483,36 @@ void av1_fht8x4_c(const int16_t *input, tran_low_t *output, int stride,
|
|||
maybe_flip_input(&input, &stride, n, n2, flipped_input, tx_type);
|
||||
#endif
|
||||
|
||||
#if CONFIG_LGT
|
||||
const tran_high_t *lgtmtx_col[8];
|
||||
const tran_high_t *lgtmtx_row[4];
|
||||
int use_lgt_col = get_fwd_lgt4(ht.cols, txfm_param, lgtmtx_col, 8);
|
||||
int use_lgt_row = get_fwd_lgt8(ht.rows, txfm_param, lgtmtx_row, 4);
|
||||
#endif
|
||||
|
||||
// Columns
|
||||
for (i = 0; i < n2; ++i) {
|
||||
for (j = 0; j < n; ++j)
|
||||
temp_in[j] =
|
||||
(tran_low_t)fdct_round_shift(input[j * stride + i] * 4 * Sqrt2);
|
||||
ht.cols(temp_in, temp_out);
|
||||
#if CONFIG_LGT
|
||||
if (use_lgt_col)
|
||||
flgt4(temp_in, temp_out, lgtmtx_col[i]);
|
||||
else
|
||||
#endif
|
||||
ht.cols(temp_in, temp_out);
|
||||
for (j = 0; j < n; ++j) out[j * n2 + i] = temp_out[j];
|
||||
}
|
||||
|
||||
// Rows
|
||||
for (i = 0; i < n; ++i) {
|
||||
for (j = 0; j < n2; ++j) temp_in[j] = out[j + i * n2];
|
||||
ht.rows(temp_in, temp_out);
|
||||
#if CONFIG_LGT
|
||||
if (use_lgt_row)
|
||||
flgt8(temp_in, temp_out, lgtmtx_row[i]);
|
||||
else
|
||||
#endif
|
||||
ht.rows(temp_in, temp_out);
|
||||
for (j = 0; j < n2; ++j)
|
||||
output[j + i * n2] = (temp_out[j] + (temp_out[j] < 0)) >> 1;
|
||||
}
|
||||
|
|
@ -1300,7 +1520,14 @@ void av1_fht8x4_c(const int16_t *input, tran_low_t *output, int stride,
|
|||
}
|
||||
|
||||
void av1_fht4x16_c(const int16_t *input, tran_low_t *output, int stride,
|
||||
int tx_type) {
|
||||
TxfmParam *txfm_param) {
|
||||
int tx_type = txfm_param->tx_type;
|
||||
#if CONFIG_MRC_TX
|
||||
assert(tx_type != MRC_DCT && "Invalid tx type for tx size");
|
||||
#endif // CONFIG_MRC_TX
|
||||
#if CONFIG_DCT_ONLY
|
||||
assert(tx_type == DCT_DCT);
|
||||
#endif
|
||||
static const transform_2d FHT[] = {
|
||||
{ fdct16, fdct4 }, // DCT_DCT
|
||||
{ fadst16, fdct4 }, // ADST_DCT
|
||||
|
|
@ -1332,10 +1559,20 @@ void av1_fht4x16_c(const int16_t *input, tran_low_t *output, int stride,
|
|||
maybe_flip_input(&input, &stride, n4, n, flipped_input, tx_type);
|
||||
#endif
|
||||
|
||||
#if CONFIG_LGT
|
||||
const tran_high_t *lgtmtx_row[16];
|
||||
int use_lgt_row = get_fwd_lgt4(ht.rows, txfm_param, lgtmtx_row, 16);
|
||||
#endif
|
||||
|
||||
// Rows
|
||||
for (i = 0; i < n4; ++i) {
|
||||
for (j = 0; j < n; ++j) temp_in[j] = input[i * stride + j] * 4;
|
||||
ht.rows(temp_in, temp_out);
|
||||
#if CONFIG_LGT
|
||||
if (use_lgt_row)
|
||||
flgt4(temp_in, temp_out, lgtmtx_row[i]);
|
||||
else
|
||||
#endif
|
||||
ht.rows(temp_in, temp_out);
|
||||
for (j = 0; j < n; ++j) out[j * n4 + i] = temp_out[j];
|
||||
}
|
||||
|
||||
|
|
@ -1350,7 +1587,14 @@ void av1_fht4x16_c(const int16_t *input, tran_low_t *output, int stride,
|
|||
}
|
||||
|
||||
void av1_fht16x4_c(const int16_t *input, tran_low_t *output, int stride,
|
||||
int tx_type) {
|
||||
TxfmParam *txfm_param) {
|
||||
int tx_type = txfm_param->tx_type;
|
||||
#if CONFIG_MRC_TX
|
||||
assert(tx_type != MRC_DCT && "Invalid tx type for tx size");
|
||||
#endif // CONFIG_MRC_TX
|
||||
#if CONFIG_DCT_ONLY
|
||||
assert(tx_type == DCT_DCT);
|
||||
#endif
|
||||
static const transform_2d FHT[] = {
|
||||
{ fdct4, fdct16 }, // DCT_DCT
|
||||
{ fadst4, fdct16 }, // ADST_DCT
|
||||
|
|
@ -1382,10 +1626,20 @@ void av1_fht16x4_c(const int16_t *input, tran_low_t *output, int stride,
|
|||
maybe_flip_input(&input, &stride, n, n4, flipped_input, tx_type);
|
||||
#endif
|
||||
|
||||
#if CONFIG_LGT
|
||||
const tran_high_t *lgtmtx_col[16];
|
||||
int use_lgt_col = get_fwd_lgt4(ht.cols, txfm_param, lgtmtx_col, 16);
|
||||
#endif
|
||||
|
||||
// Columns
|
||||
for (i = 0; i < n4; ++i) {
|
||||
for (j = 0; j < n; ++j) temp_in[j] = input[j * stride + i] * 4;
|
||||
ht.cols(temp_in, temp_out);
|
||||
#if CONFIG_LGT
|
||||
if (use_lgt_col)
|
||||
flgt4(temp_in, temp_out, lgtmtx_col[i]);
|
||||
else
|
||||
#endif
|
||||
ht.cols(temp_in, temp_out);
|
||||
for (j = 0; j < n; ++j) out[j * n4 + i] = temp_out[j];
|
||||
}
|
||||
|
||||
|
|
@ -1400,7 +1654,14 @@ void av1_fht16x4_c(const int16_t *input, tran_low_t *output, int stride,
|
|||
}
|
||||
|
||||
void av1_fht8x16_c(const int16_t *input, tran_low_t *output, int stride,
|
||||
int tx_type) {
|
||||
TxfmParam *txfm_param) {
|
||||
int tx_type = txfm_param->tx_type;
|
||||
#if CONFIG_MRC_TX
|
||||
assert(tx_type != MRC_DCT && "Invalid tx type for tx size");
|
||||
#endif // CONFIG_MRC_TX
|
||||
#if CONFIG_DCT_ONLY
|
||||
assert(tx_type == DCT_DCT);
|
||||
#endif
|
||||
static const transform_2d FHT[] = {
|
||||
{ fdct16, fdct8 }, // DCT_DCT
|
||||
{ fadst16, fdct8 }, // ADST_DCT
|
||||
|
|
@ -1432,12 +1693,22 @@ void av1_fht8x16_c(const int16_t *input, tran_low_t *output, int stride,
|
|||
maybe_flip_input(&input, &stride, n2, n, flipped_input, tx_type);
|
||||
#endif
|
||||
|
||||
#if CONFIG_LGT
|
||||
const tran_high_t *lgtmtx_row[16];
|
||||
int use_lgt_row = get_fwd_lgt8(ht.rows, txfm_param, lgtmtx_row, 16);
|
||||
#endif
|
||||
|
||||
// Rows
|
||||
for (i = 0; i < n2; ++i) {
|
||||
for (j = 0; j < n; ++j)
|
||||
temp_in[j] =
|
||||
(tran_low_t)fdct_round_shift(input[i * stride + j] * 4 * Sqrt2);
|
||||
ht.rows(temp_in, temp_out);
|
||||
#if CONFIG_LGT
|
||||
if (use_lgt_row)
|
||||
flgt8(temp_in, temp_out, lgtmtx_row[i]);
|
||||
else
|
||||
#endif
|
||||
ht.rows(temp_in, temp_out);
|
||||
for (j = 0; j < n; ++j)
|
||||
out[j * n2 + i] = ROUND_POWER_OF_TWO_SIGNED(temp_out[j], 2);
|
||||
}
|
||||
|
|
@ -1452,7 +1723,14 @@ void av1_fht8x16_c(const int16_t *input, tran_low_t *output, int stride,
|
|||
}
|
||||
|
||||
void av1_fht16x8_c(const int16_t *input, tran_low_t *output, int stride,
|
||||
int tx_type) {
|
||||
TxfmParam *txfm_param) {
|
||||
int tx_type = txfm_param->tx_type;
|
||||
#if CONFIG_MRC_TX
|
||||
assert(tx_type != MRC_DCT && "Invalid tx type for tx size");
|
||||
#endif // CONFIG_MRC_TX
|
||||
#if CONFIG_DCT_ONLY
|
||||
assert(tx_type == DCT_DCT);
|
||||
#endif
|
||||
static const transform_2d FHT[] = {
|
||||
{ fdct8, fdct16 }, // DCT_DCT
|
||||
{ fadst8, fdct16 }, // ADST_DCT
|
||||
|
|
@ -1484,12 +1762,22 @@ void av1_fht16x8_c(const int16_t *input, tran_low_t *output, int stride,
|
|||
maybe_flip_input(&input, &stride, n, n2, flipped_input, tx_type);
|
||||
#endif
|
||||
|
||||
#if CONFIG_LGT
|
||||
const tran_high_t *lgtmtx_col[16];
|
||||
int use_lgt_col = get_fwd_lgt8(ht.cols, txfm_param, lgtmtx_col, 16);
|
||||
#endif
|
||||
|
||||
// Columns
|
||||
for (i = 0; i < n2; ++i) {
|
||||
for (j = 0; j < n; ++j)
|
||||
temp_in[j] =
|
||||
(tran_low_t)fdct_round_shift(input[j * stride + i] * 4 * Sqrt2);
|
||||
ht.cols(temp_in, temp_out);
|
||||
#if CONFIG_LGT
|
||||
if (use_lgt_col)
|
||||
flgt8(temp_in, temp_out, lgtmtx_col[i]);
|
||||
else
|
||||
#endif
|
||||
ht.cols(temp_in, temp_out);
|
||||
for (j = 0; j < n; ++j)
|
||||
out[j * n2 + i] = ROUND_POWER_OF_TWO_SIGNED(temp_out[j], 2);
|
||||
}
|
||||
|
|
@ -1504,7 +1792,14 @@ void av1_fht16x8_c(const int16_t *input, tran_low_t *output, int stride,
|
|||
}
|
||||
|
||||
void av1_fht8x32_c(const int16_t *input, tran_low_t *output, int stride,
|
||||
int tx_type) {
|
||||
TxfmParam *txfm_param) {
|
||||
int tx_type = txfm_param->tx_type;
|
||||
#if CONFIG_MRC_TX
|
||||
assert(tx_type != MRC_DCT && "Invalid tx type for tx size");
|
||||
#endif // CONFIG_MRC_TX
|
||||
#if CONFIG_DCT_ONLY
|
||||
assert(tx_type == DCT_DCT);
|
||||
#endif
|
||||
static const transform_2d FHT[] = {
|
||||
{ fdct32, fdct8 }, // DCT_DCT
|
||||
{ fhalfright32, fdct8 }, // ADST_DCT
|
||||
|
|
@ -1536,10 +1831,20 @@ void av1_fht8x32_c(const int16_t *input, tran_low_t *output, int stride,
|
|||
maybe_flip_input(&input, &stride, n4, n, flipped_input, tx_type);
|
||||
#endif
|
||||
|
||||
#if CONFIG_LGT
|
||||
const tran_high_t *lgtmtx_row[32];
|
||||
int use_lgt_row = get_fwd_lgt8(ht.rows, txfm_param, lgtmtx_row, 32);
|
||||
#endif
|
||||
|
||||
// Rows
|
||||
for (i = 0; i < n4; ++i) {
|
||||
for (j = 0; j < n; ++j) temp_in[j] = input[i * stride + j] * 4;
|
||||
ht.rows(temp_in, temp_out);
|
||||
#if CONFIG_LGT
|
||||
if (use_lgt_row)
|
||||
flgt8(temp_in, temp_out, lgtmtx_row[i]);
|
||||
else
|
||||
#endif
|
||||
ht.rows(temp_in, temp_out);
|
||||
for (j = 0; j < n; ++j) out[j * n4 + i] = temp_out[j];
|
||||
}
|
||||
|
||||
|
|
@ -1554,7 +1859,14 @@ void av1_fht8x32_c(const int16_t *input, tran_low_t *output, int stride,
|
|||
}
|
||||
|
||||
void av1_fht32x8_c(const int16_t *input, tran_low_t *output, int stride,
|
||||
int tx_type) {
|
||||
TxfmParam *txfm_param) {
|
||||
int tx_type = txfm_param->tx_type;
|
||||
#if CONFIG_MRC_TX
|
||||
assert(tx_type != MRC_DCT && "Invalid tx type for tx size");
|
||||
#endif // CONFIG_MRC_TX
|
||||
#if CONFIG_DCT_ONLY
|
||||
assert(tx_type == DCT_DCT);
|
||||
#endif
|
||||
static const transform_2d FHT[] = {
|
||||
{ fdct8, fdct32 }, // DCT_DCT
|
||||
{ fadst8, fdct32 }, // ADST_DCT
|
||||
|
|
@ -1586,10 +1898,20 @@ void av1_fht32x8_c(const int16_t *input, tran_low_t *output, int stride,
|
|||
maybe_flip_input(&input, &stride, n, n4, flipped_input, tx_type);
|
||||
#endif
|
||||
|
||||
#if CONFIG_LGT
|
||||
const tran_high_t *lgtmtx_col[32];
|
||||
int use_lgt_col = get_fwd_lgt8(ht.cols, txfm_param, lgtmtx_col, 32);
|
||||
#endif
|
||||
|
||||
// Columns
|
||||
for (i = 0; i < n4; ++i) {
|
||||
for (j = 0; j < n; ++j) temp_in[j] = input[j * stride + i] * 4;
|
||||
ht.cols(temp_in, temp_out);
|
||||
#if CONFIG_LGT
|
||||
if (use_lgt_col)
|
||||
flgt8(temp_in, temp_out, lgtmtx_col[i]);
|
||||
else
|
||||
#endif
|
||||
ht.cols(temp_in, temp_out);
|
||||
for (j = 0; j < n; ++j) out[j * n4 + i] = temp_out[j];
|
||||
}
|
||||
|
||||
|
|
@ -1604,7 +1926,14 @@ void av1_fht32x8_c(const int16_t *input, tran_low_t *output, int stride,
|
|||
}
|
||||
|
||||
void av1_fht16x32_c(const int16_t *input, tran_low_t *output, int stride,
|
||||
int tx_type) {
|
||||
TxfmParam *txfm_param) {
|
||||
int tx_type = txfm_param->tx_type;
|
||||
#if CONFIG_MRC_TX
|
||||
assert(tx_type != MRC_DCT && "Invalid tx type for tx size");
|
||||
#endif // CONFIG_MRC_TX
|
||||
#if CONFIG_DCT_ONLY
|
||||
assert(tx_type == DCT_DCT);
|
||||
#endif
|
||||
static const transform_2d FHT[] = {
|
||||
{ fdct32, fdct16 }, // DCT_DCT
|
||||
{ fhalfright32, fdct16 }, // ADST_DCT
|
||||
|
|
@ -1656,7 +1985,14 @@ void av1_fht16x32_c(const int16_t *input, tran_low_t *output, int stride,
|
|||
}
|
||||
|
||||
void av1_fht32x16_c(const int16_t *input, tran_low_t *output, int stride,
|
||||
int tx_type) {
|
||||
TxfmParam *txfm_param) {
|
||||
int tx_type = txfm_param->tx_type;
|
||||
#if CONFIG_MRC_TX
|
||||
assert(tx_type != MRC_DCT && "Invalid tx type for tx size");
|
||||
#endif // CONFIG_MRC_TX
|
||||
#if CONFIG_DCT_ONLY
|
||||
assert(tx_type == DCT_DCT);
|
||||
#endif
|
||||
static const transform_2d FHT[] = {
|
||||
{ fdct16, fdct32 }, // DCT_DCT
|
||||
{ fadst16, fdct32 }, // ADST_DCT
|
||||
|
|
@ -1833,10 +2169,21 @@ void av1_fdct8x8_quant_c(const int16_t *input, int stride,
|
|||
}
|
||||
|
||||
void av1_fht8x8_c(const int16_t *input, tran_low_t *output, int stride,
|
||||
int tx_type) {
|
||||
TxfmParam *txfm_param) {
|
||||
int tx_type = txfm_param->tx_type;
|
||||
#if CONFIG_MRC_TX
|
||||
assert(tx_type != MRC_DCT && "Invalid tx type for tx size");
|
||||
#endif // CONFIG_MRC_TX
|
||||
#if CONFIG_DCT_ONLY
|
||||
assert(tx_type == DCT_DCT);
|
||||
#endif
|
||||
#if !CONFIG_DAALA_DCT8
|
||||
if (tx_type == DCT_DCT) {
|
||||
aom_fdct8x8_c(input, output, stride);
|
||||
} else {
|
||||
return;
|
||||
}
|
||||
#endif
|
||||
{
|
||||
static const transform_2d FHT[] = {
|
||||
{ fdct8, fdct8 }, // DCT_DCT
|
||||
{ fadst8, fdct8 }, // ADST_DCT
|
||||
|
|
@ -1855,7 +2202,7 @@ void av1_fht8x8_c(const int16_t *input, tran_low_t *output, int stride,
|
|||
{ fidtx8, fadst8 }, // H_ADST
|
||||
{ fadst8, fidtx8 }, // V_FLIPADST
|
||||
{ fidtx8, fadst8 }, // H_FLIPADST
|
||||
#endif // CONFIG_EXT_TX
|
||||
#endif
|
||||
};
|
||||
const transform_2d ht = FHT[tx_type];
|
||||
tran_low_t out[64];
|
||||
|
|
@ -1867,19 +2214,45 @@ void av1_fht8x8_c(const int16_t *input, tran_low_t *output, int stride,
|
|||
maybe_flip_input(&input, &stride, 8, 8, flipped_input, tx_type);
|
||||
#endif
|
||||
|
||||
#if CONFIG_LGT
|
||||
const tran_high_t *lgtmtx_col[8];
|
||||
const tran_high_t *lgtmtx_row[8];
|
||||
int use_lgt_col = get_fwd_lgt8(ht.cols, txfm_param, lgtmtx_col, 8);
|
||||
int use_lgt_row = get_fwd_lgt8(ht.rows, txfm_param, lgtmtx_row, 8);
|
||||
#endif
|
||||
|
||||
// Columns
|
||||
for (i = 0; i < 8; ++i) {
|
||||
#if CONFIG_DAALA_DCT8
|
||||
for (j = 0; j < 8; ++j) temp_in[j] = input[j * stride + i] * 16;
|
||||
#else
|
||||
for (j = 0; j < 8; ++j) temp_in[j] = input[j * stride + i] * 4;
|
||||
ht.cols(temp_in, temp_out);
|
||||
#endif
|
||||
#if CONFIG_LGT
|
||||
if (use_lgt_col)
|
||||
flgt8(temp_in, temp_out, lgtmtx_col[i]);
|
||||
else
|
||||
#endif
|
||||
ht.cols(temp_in, temp_out);
|
||||
for (j = 0; j < 8; ++j) out[j * 8 + i] = temp_out[j];
|
||||
}
|
||||
|
||||
// Rows
|
||||
for (i = 0; i < 8; ++i) {
|
||||
for (j = 0; j < 8; ++j) temp_in[j] = out[j + i * 8];
|
||||
ht.rows(temp_in, temp_out);
|
||||
#if CONFIG_LGT
|
||||
if (use_lgt_row)
|
||||
flgt8(temp_in, temp_out, lgtmtx_row[i]);
|
||||
else
|
||||
#endif
|
||||
ht.rows(temp_in, temp_out);
|
||||
#if CONFIG_DAALA_DCT8
|
||||
for (j = 0; j < 8; ++j)
|
||||
output[j + i * 8] = (temp_out[j] + (temp_out[j] < 0)) >> 1;
|
||||
#else
|
||||
for (j = 0; j < 8; ++j)
|
||||
output[j + i * 8] = (temp_out[j] + (temp_out[j] < 0)) >> 1;
|
||||
#endif
|
||||
}
|
||||
}
|
||||
}
|
||||
|
|
@ -1941,7 +2314,14 @@ void av1_fwht4x4_c(const int16_t *input, tran_low_t *output, int stride) {
|
|||
}
|
||||
|
||||
void av1_fht16x16_c(const int16_t *input, tran_low_t *output, int stride,
|
||||
int tx_type) {
|
||||
TxfmParam *txfm_param) {
|
||||
int tx_type = txfm_param->tx_type;
|
||||
#if CONFIG_MRC_TX
|
||||
assert(tx_type != MRC_DCT && "Invalid tx type for tx size");
|
||||
#endif // CONFIG_MRC_TX
|
||||
#if CONFIG_DCT_ONLY
|
||||
assert(tx_type == DCT_DCT);
|
||||
#endif
|
||||
static const transform_2d FHT[] = {
|
||||
{ fdct16, fdct16 }, // DCT_DCT
|
||||
{ fadst16, fdct16 }, // ADST_DCT
|
||||
|
|
@ -1960,9 +2340,8 @@ void av1_fht16x16_c(const int16_t *input, tran_low_t *output, int stride,
|
|||
{ fidtx16, fadst16 }, // H_ADST
|
||||
{ fadst16, fidtx16 }, // V_FLIPADST
|
||||
{ fidtx16, fadst16 }, // H_FLIPADST
|
||||
#endif // CONFIG_EXT_TX
|
||||
#endif
|
||||
};
|
||||
|
||||
const transform_2d ht = FHT[tx_type];
|
||||
tran_low_t out[256];
|
||||
int i, j;
|
||||
|
|
@ -1989,80 +2368,17 @@ void av1_fht16x16_c(const int16_t *input, tran_low_t *output, int stride,
|
|||
}
|
||||
}
|
||||
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
void av1_highbd_fht4x4_c(const int16_t *input, tran_low_t *output, int stride,
|
||||
int tx_type) {
|
||||
av1_fht4x4_c(input, output, stride, tx_type);
|
||||
}
|
||||
|
||||
void av1_highbd_fht4x8_c(const int16_t *input, tran_low_t *output, int stride,
|
||||
int tx_type) {
|
||||
av1_fht4x8_c(input, output, stride, tx_type);
|
||||
}
|
||||
|
||||
void av1_highbd_fht8x4_c(const int16_t *input, tran_low_t *output, int stride,
|
||||
int tx_type) {
|
||||
av1_fht8x4_c(input, output, stride, tx_type);
|
||||
}
|
||||
|
||||
void av1_highbd_fht8x16_c(const int16_t *input, tran_low_t *output, int stride,
|
||||
int tx_type) {
|
||||
av1_fht8x16_c(input, output, stride, tx_type);
|
||||
}
|
||||
|
||||
void av1_highbd_fht16x8_c(const int16_t *input, tran_low_t *output, int stride,
|
||||
int tx_type) {
|
||||
av1_fht16x8_c(input, output, stride, tx_type);
|
||||
}
|
||||
|
||||
void av1_highbd_fht16x32_c(const int16_t *input, tran_low_t *output, int stride,
|
||||
int tx_type) {
|
||||
av1_fht16x32_c(input, output, stride, tx_type);
|
||||
}
|
||||
|
||||
void av1_highbd_fht32x16_c(const int16_t *input, tran_low_t *output, int stride,
|
||||
int tx_type) {
|
||||
av1_fht32x16_c(input, output, stride, tx_type);
|
||||
}
|
||||
|
||||
void av1_highbd_fht4x16_c(const int16_t *input, tran_low_t *output, int stride,
|
||||
int tx_type) {
|
||||
av1_fht4x16_c(input, output, stride, tx_type);
|
||||
}
|
||||
|
||||
void av1_highbd_fht16x4_c(const int16_t *input, tran_low_t *output, int stride,
|
||||
int tx_type) {
|
||||
av1_fht16x4_c(input, output, stride, tx_type);
|
||||
}
|
||||
|
||||
void av1_highbd_fht8x32_c(const int16_t *input, tran_low_t *output, int stride,
|
||||
int tx_type) {
|
||||
av1_fht8x32_c(input, output, stride, tx_type);
|
||||
}
|
||||
|
||||
void av1_highbd_fht32x8_c(const int16_t *input, tran_low_t *output, int stride,
|
||||
int tx_type) {
|
||||
av1_fht32x8_c(input, output, stride, tx_type);
|
||||
}
|
||||
|
||||
void av1_highbd_fht8x8_c(const int16_t *input, tran_low_t *output, int stride,
|
||||
int tx_type) {
|
||||
av1_fht8x8_c(input, output, stride, tx_type);
|
||||
}
|
||||
|
||||
void av1_highbd_fwht4x4_c(const int16_t *input, tran_low_t *output,
|
||||
int stride) {
|
||||
av1_fwht4x4_c(input, output, stride);
|
||||
}
|
||||
|
||||
void av1_highbd_fht16x16_c(const int16_t *input, tran_low_t *output, int stride,
|
||||
int tx_type) {
|
||||
av1_fht16x16_c(input, output, stride, tx_type);
|
||||
}
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
|
||||
void av1_fht32x32_c(const int16_t *input, tran_low_t *output, int stride,
|
||||
int tx_type) {
|
||||
TxfmParam *txfm_param) {
|
||||
int tx_type = txfm_param->tx_type;
|
||||
#if CONFIG_DCT_ONLY
|
||||
assert(tx_type == DCT_DCT);
|
||||
#endif
|
||||
static const transform_2d FHT[] = {
|
||||
{ fdct32, fdct32 }, // DCT_DCT
|
||||
#if CONFIG_EXT_TX
|
||||
|
|
@ -2082,6 +2398,9 @@ void av1_fht32x32_c(const int16_t *input, tran_low_t *output, int stride,
|
|||
{ fhalfright32, fidtx32 }, // V_FLIPADST
|
||||
{ fidtx32, fhalfright32 }, // H_FLIPADST
|
||||
#endif
|
||||
#if CONFIG_MRC_TX
|
||||
{ fdct32, fdct32 }, // MRC_TX
|
||||
#endif // CONFIG_MRC_TX
|
||||
};
|
||||
const transform_2d ht = FHT[tx_type];
|
||||
tran_low_t out[1024];
|
||||
|
|
@ -2093,6 +2412,14 @@ void av1_fht32x32_c(const int16_t *input, tran_low_t *output, int stride,
|
|||
maybe_flip_input(&input, &stride, 32, 32, flipped_input, tx_type);
|
||||
#endif
|
||||
|
||||
#if CONFIG_MRC_TX
|
||||
if (tx_type == MRC_DCT) {
|
||||
int16_t masked_input[32 * 32];
|
||||
get_masked_residual32(&input, &stride, txfm_param->dst, txfm_param->stride,
|
||||
masked_input);
|
||||
}
|
||||
#endif // CONFIG_MRC_TX
|
||||
|
||||
// Columns
|
||||
for (i = 0; i < 32; ++i) {
|
||||
for (j = 0; j < 32; ++j) temp_in[j] = input[j * stride + i] * 4;
|
||||
|
|
@ -2150,7 +2477,14 @@ static void fdct64_row(const tran_low_t *input, tran_low_t *output) {
|
|||
}
|
||||
|
||||
void av1_fht64x64_c(const int16_t *input, tran_low_t *output, int stride,
|
||||
int tx_type) {
|
||||
TxfmParam *txfm_param) {
|
||||
int tx_type = txfm_param->tx_type;
|
||||
#if CONFIG_MRC_TX
|
||||
assert(tx_type != MRC_DCT && "Invalid tx type for tx size");
|
||||
#endif // CONFIG_MRC_TX
|
||||
#if CONFIG_DCT_ONLY
|
||||
assert(tx_type == DCT_DCT);
|
||||
#endif
|
||||
static const transform_2d FHT[] = {
|
||||
{ fdct64_col, fdct64_row }, // DCT_DCT
|
||||
#if CONFIG_EXT_TX
|
||||
|
|
@ -2179,6 +2513,7 @@ void av1_fht64x64_c(const int16_t *input, tran_low_t *output, int stride,
|
|||
int16_t flipped_input[64 * 64];
|
||||
maybe_flip_input(&input, &stride, 64, 64, flipped_input, tx_type);
|
||||
#endif
|
||||
|
||||
// Columns
|
||||
for (i = 0; i < 64; ++i) {
|
||||
for (j = 0; j < 64; ++j) temp_in[j] = input[j * stride + i];
|
||||
|
|
@ -2214,20 +2549,6 @@ void av1_fwd_idtx_c(const int16_t *src_diff, tran_low_t *coeff, int stride,
|
|||
}
|
||||
#endif // CONFIG_EXT_TX
|
||||
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
void av1_highbd_fht32x32_c(const int16_t *input, tran_low_t *output, int stride,
|
||||
int tx_type) {
|
||||
av1_fht32x32_c(input, output, stride, tx_type);
|
||||
}
|
||||
|
||||
#if CONFIG_TX64X64
|
||||
void av1_highbd_fht64x64_c(const int16_t *input, tran_low_t *output, int stride,
|
||||
int tx_type) {
|
||||
av1_fht64x64_c(input, output, stride, tx_type);
|
||||
}
|
||||
#endif // CONFIG_TX64X64
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
|
||||
#if CONFIG_DPCM_INTRA
|
||||
void av1_dpcm_ft4_c(const int16_t *input, int stride, TX_TYPE_1D tx_type,
|
||||
tran_low_t *output) {
|
||||
|
|
@ -2271,5 +2592,54 @@ void av1_dpcm_ft32_c(const int16_t *input, int stride, TX_TYPE_1D tx_type,
|
|||
for (int i = 0; i < 32; ++i) temp_in[i] = input[i * stride];
|
||||
ft(temp_in, output);
|
||||
}
|
||||
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
void av1_hbd_dpcm_ft4_c(const int16_t *input, int stride, TX_TYPE_1D tx_type,
|
||||
tran_low_t *output, int dir) {
|
||||
(void)dir;
|
||||
assert(tx_type < TX_TYPES_1D);
|
||||
static const transform_1d FHT[] = { fdct4, fadst4, fadst4, fidtx4 };
|
||||
const transform_1d ft = FHT[tx_type];
|
||||
tran_low_t temp_in[4];
|
||||
for (int i = 0; i < 4; ++i)
|
||||
temp_in[i] = (tran_low_t)fdct_round_shift(input[i * stride] * 4 * Sqrt2);
|
||||
ft(temp_in, output);
|
||||
}
|
||||
|
||||
void av1_hbd_dpcm_ft8_c(const int16_t *input, int stride, TX_TYPE_1D tx_type,
|
||||
tran_low_t *output, int dir) {
|
||||
(void)dir;
|
||||
assert(tx_type < TX_TYPES_1D);
|
||||
static const transform_1d FHT[] = { fdct8, fadst8, fadst8, fidtx8 };
|
||||
const transform_1d ft = FHT[tx_type];
|
||||
tran_low_t temp_in[8];
|
||||
for (int i = 0; i < 8; ++i) temp_in[i] = input[i * stride] * 4;
|
||||
ft(temp_in, output);
|
||||
}
|
||||
|
||||
void av1_hbd_dpcm_ft16_c(const int16_t *input, int stride, TX_TYPE_1D tx_type,
|
||||
tran_low_t *output, int dir) {
|
||||
(void)dir;
|
||||
assert(tx_type < TX_TYPES_1D);
|
||||
static const transform_1d FHT[] = { fdct16, fadst16, fadst16, fidtx16 };
|
||||
const transform_1d ft = FHT[tx_type];
|
||||
tran_low_t temp_in[16];
|
||||
for (int i = 0; i < 16; ++i)
|
||||
temp_in[i] = (tran_low_t)fdct_round_shift(input[i * stride] * 2 * Sqrt2);
|
||||
ft(temp_in, output);
|
||||
}
|
||||
|
||||
void av1_hbd_dpcm_ft32_c(const int16_t *input, int stride, TX_TYPE_1D tx_type,
|
||||
tran_low_t *output, int dir) {
|
||||
(void)dir;
|
||||
assert(tx_type < TX_TYPES_1D);
|
||||
static const transform_1d FHT[] = { fdct32, fhalfright32, fhalfright32,
|
||||
fidtx32 };
|
||||
const transform_1d ft = FHT[tx_type];
|
||||
tran_low_t temp_in[32];
|
||||
for (int i = 0; i < 32; ++i) temp_in[i] = input[i * stride];
|
||||
ft(temp_in, output);
|
||||
}
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
#endif // CONFIG_DPCM_INTRA
|
||||
#endif // !AV1_DCT_GTEST
|
||||
|
|
|
|||
1348
third_party/aom/av1/encoder/encodeframe.c
vendored
1348
third_party/aom/av1/encoder/encodeframe.c
vendored
File diff suppressed because it is too large
Load diff
2
third_party/aom/av1/encoder/encodeframe.h
vendored
2
third_party/aom/av1/encoder/encodeframe.h
vendored
|
|
@ -37,7 +37,7 @@ void av1_encode_tile(struct AV1_COMP *cpi, struct ThreadData *td, int tile_row,
|
|||
|
||||
void av1_update_tx_type_count(const struct AV1Common *cm, MACROBLOCKD *xd,
|
||||
#if CONFIG_TXK_SEL
|
||||
int block, int plane,
|
||||
int blk_row, int blk_col, int block, int plane,
|
||||
#endif
|
||||
BLOCK_SIZE bsize, TX_SIZE tx_size,
|
||||
FRAME_COUNTS *counts);
|
||||
|
|
|
|||
980
third_party/aom/av1/encoder/encodemb.c
vendored
980
third_party/aom/av1/encoder/encodemb.c
vendored
File diff suppressed because it is too large
Load diff
15
third_party/aom/av1/encoder/encodemb.h
vendored
15
third_party/aom/av1/encoder/encodemb.h
vendored
|
|
@ -53,9 +53,10 @@ void av1_xform_quant(const AV1_COMMON *cm, MACROBLOCK *x, int plane, int block,
|
|||
int blk_row, int blk_col, BLOCK_SIZE plane_bsize,
|
||||
TX_SIZE tx_size, int ctx, AV1_XFORM_QUANT xform_quant_idx);
|
||||
|
||||
int av1_optimize_b(const AV1_COMMON *cm, MACROBLOCK *mb, int plane, int block,
|
||||
BLOCK_SIZE plane_bsize, TX_SIZE tx_size,
|
||||
const ENTROPY_CONTEXT *a, const ENTROPY_CONTEXT *l);
|
||||
int av1_optimize_b(const AV1_COMMON *cm, MACROBLOCK *mb, int plane, int blk_row,
|
||||
int blk_col, int block, BLOCK_SIZE plane_bsize,
|
||||
TX_SIZE tx_size, const ENTROPY_CONTEXT *a,
|
||||
const ENTROPY_CONTEXT *l);
|
||||
|
||||
void av1_subtract_txb(MACROBLOCK *x, int plane, BLOCK_SIZE plane_bsize,
|
||||
int blk_col, int blk_row, TX_SIZE tx_size);
|
||||
|
|
@ -86,14 +87,6 @@ void av1_store_pvq_enc_info(PVQ_INFO *pvq_info, int *qg, int *theta, int *k,
|
|||
int *size, int skip_rest, int skip_dir, int bs);
|
||||
#endif
|
||||
|
||||
#if CONFIG_CFL
|
||||
void av1_predict_intra_block_encoder_facade(MACROBLOCK *x,
|
||||
FRAME_CONTEXT *ec_ctx, int plane,
|
||||
int block_idx, int blk_col,
|
||||
int blk_row, TX_SIZE tx_size,
|
||||
BLOCK_SIZE plane_bsize);
|
||||
#endif
|
||||
|
||||
#if CONFIG_DPCM_INTRA
|
||||
void av1_encode_block_intra_dpcm(const AV1_COMMON *cm, MACROBLOCK *x,
|
||||
PREDICTION_MODE mode, int plane, int block,
|
||||
|
|
|
|||
143
third_party/aom/av1/encoder/encodemv.c
vendored
143
third_party/aom/av1/encoder/encodemv.c
vendored
|
|
@ -31,7 +31,7 @@ void av1_entropy_mv_init(void) {
|
|||
}
|
||||
|
||||
static void encode_mv_component(aom_writer *w, int comp, nmv_component *mvcomp,
|
||||
int usehp) {
|
||||
MvSubpelPrecision precision) {
|
||||
int offset;
|
||||
const int sign = comp < 0;
|
||||
const int mag = sign ? -comp : comp;
|
||||
|
|
@ -42,34 +42,53 @@ static void encode_mv_component(aom_writer *w, int comp, nmv_component *mvcomp,
|
|||
|
||||
assert(comp != 0);
|
||||
|
||||
// Sign
|
||||
// Sign
|
||||
#if CONFIG_NEW_MULTISYMBOL
|
||||
aom_write_bit(w, sign);
|
||||
#else
|
||||
aom_write(w, sign, mvcomp->sign);
|
||||
#endif
|
||||
|
||||
// Class
|
||||
aom_write_symbol(w, mv_class, mvcomp->class_cdf, MV_CLASSES);
|
||||
|
||||
// Integer bits
|
||||
if (mv_class == MV_CLASS_0) {
|
||||
#if CONFIG_NEW_MULTISYMBOL
|
||||
aom_write_symbol(w, d, mvcomp->class0_cdf, CLASS0_SIZE);
|
||||
#else
|
||||
aom_write(w, d, mvcomp->class0[0]);
|
||||
#endif
|
||||
} else {
|
||||
int i;
|
||||
const int n = mv_class + CLASS0_BITS - 1; // number of bits
|
||||
for (i = 0; i < n; ++i) aom_write(w, (d >> i) & 1, mvcomp->bits[i]);
|
||||
}
|
||||
|
||||
// Fractional bits
|
||||
aom_write_symbol(
|
||||
w, fr, mv_class == MV_CLASS_0 ? mvcomp->class0_fp_cdf[d] : mvcomp->fp_cdf,
|
||||
MV_FP_SIZE);
|
||||
// Fractional bits
|
||||
#if CONFIG_INTRABC
|
||||
if (precision > MV_SUBPEL_NONE)
|
||||
#endif // CONFIG_INTRABC
|
||||
{
|
||||
aom_write_symbol(w, fr, mv_class == MV_CLASS_0 ? mvcomp->class0_fp_cdf[d]
|
||||
: mvcomp->fp_cdf,
|
||||
MV_FP_SIZE);
|
||||
}
|
||||
|
||||
// High precision bit
|
||||
if (usehp)
|
||||
if (precision > MV_SUBPEL_LOW_PRECISION)
|
||||
#if CONFIG_NEW_MULTISYMBOL
|
||||
aom_write_symbol(
|
||||
w, hp, mv_class == MV_CLASS_0 ? mvcomp->class0_hp_cdf : mvcomp->hp_cdf,
|
||||
2);
|
||||
#else
|
||||
aom_write(w, hp, mv_class == MV_CLASS_0 ? mvcomp->class0_hp : mvcomp->hp);
|
||||
#endif
|
||||
}
|
||||
|
||||
static void build_nmv_component_cost_table(int *mvcost,
|
||||
const nmv_component *const mvcomp,
|
||||
int usehp) {
|
||||
MvSubpelPrecision precision) {
|
||||
int i, v;
|
||||
int sign_cost[2], class_cost[MV_CLASSES], class0_cost[CLASS0_SIZE];
|
||||
int bits_cost[MV_OFFSET_BITS][2];
|
||||
|
|
@ -89,7 +108,7 @@ static void build_nmv_component_cost_table(int *mvcost,
|
|||
av1_cost_tokens(class0_fp_cost[i], mvcomp->class0_fp[i], av1_mv_fp_tree);
|
||||
av1_cost_tokens(fp_cost, mvcomp->fp, av1_mv_fp_tree);
|
||||
|
||||
if (usehp) {
|
||||
if (precision > MV_SUBPEL_LOW_PRECISION) {
|
||||
class0_hp_cost[0] = av1_cost_zero(mvcomp->class0_hp);
|
||||
class0_hp_cost[1] = av1_cost_one(mvcomp->class0_hp);
|
||||
hp_cost[0] = av1_cost_zero(mvcomp->hp);
|
||||
|
|
@ -110,16 +129,21 @@ static void build_nmv_component_cost_table(int *mvcost,
|
|||
const int b = c + CLASS0_BITS - 1; /* number of bits */
|
||||
for (i = 0; i < b; ++i) cost += bits_cost[i][((d >> i) & 1)];
|
||||
}
|
||||
if (c == MV_CLASS_0) {
|
||||
cost += class0_fp_cost[d][f];
|
||||
} else {
|
||||
cost += fp_cost[f];
|
||||
}
|
||||
if (usehp) {
|
||||
#if CONFIG_INTRABC
|
||||
if (precision > MV_SUBPEL_NONE)
|
||||
#endif // CONFIG_INTRABC
|
||||
{
|
||||
if (c == MV_CLASS_0) {
|
||||
cost += class0_hp_cost[e];
|
||||
cost += class0_fp_cost[d][f];
|
||||
} else {
|
||||
cost += hp_cost[e];
|
||||
cost += fp_cost[f];
|
||||
}
|
||||
if (precision > MV_SUBPEL_LOW_PRECISION) {
|
||||
if (c == MV_CLASS_0) {
|
||||
cost += class0_hp_cost[e];
|
||||
} else {
|
||||
cost += hp_cost[e];
|
||||
}
|
||||
}
|
||||
}
|
||||
mvcost[v] = cost + sign_cost[0];
|
||||
|
|
@ -127,36 +151,16 @@ static void build_nmv_component_cost_table(int *mvcost,
|
|||
}
|
||||
}
|
||||
|
||||
#if !CONFIG_NEW_MULTISYMBOL
|
||||
static void update_mv(aom_writer *w, const unsigned int ct[2], aom_prob *cur_p,
|
||||
aom_prob upd_p) {
|
||||
(void)upd_p;
|
||||
#if CONFIG_TILE_GROUPS
|
||||
// Just use the default maximum number of tile groups to avoid passing in the
|
||||
// actual
|
||||
// number
|
||||
av1_cond_prob_diff_update(w, cur_p, ct, DEFAULT_MAX_NUM_TG);
|
||||
#else
|
||||
av1_cond_prob_diff_update(w, cur_p, ct, 1);
|
||||
#endif
|
||||
}
|
||||
|
||||
#if !CONFIG_EC_ADAPT
|
||||
static void write_mv_update(const aom_tree_index *tree,
|
||||
aom_prob probs[/*n - 1*/],
|
||||
const unsigned int counts[/*n - 1*/], int n,
|
||||
aom_writer *w) {
|
||||
int i;
|
||||
unsigned int branch_ct[32][2];
|
||||
|
||||
// Assuming max number of probabilities <= 32
|
||||
assert(n <= 32);
|
||||
|
||||
av1_tree_probs_from_distribution(tree, branch_ct, counts);
|
||||
for (i = 0; i < n - 1; ++i)
|
||||
update_mv(w, branch_ct[i], &probs[i], MV_UPDATE_PROB);
|
||||
}
|
||||
#endif
|
||||
|
||||
void av1_write_nmv_probs(AV1_COMMON *cm, int usehp, aom_writer *w,
|
||||
nmv_context_counts *const nmv_counts) {
|
||||
int i;
|
||||
|
|
@ -164,34 +168,6 @@ void av1_write_nmv_probs(AV1_COMMON *cm, int usehp, aom_writer *w,
|
|||
for (nmv_ctx = 0; nmv_ctx < NMV_CONTEXTS; ++nmv_ctx) {
|
||||
nmv_context *const mvc = &cm->fc->nmvc[nmv_ctx];
|
||||
nmv_context_counts *const counts = &nmv_counts[nmv_ctx];
|
||||
#if !CONFIG_EC_ADAPT
|
||||
write_mv_update(av1_mv_joint_tree, mvc->joints, counts->joints, MV_JOINTS,
|
||||
w);
|
||||
|
||||
for (i = 0; i < 2; ++i) {
|
||||
int j;
|
||||
nmv_component *comp = &mvc->comps[i];
|
||||
nmv_component_counts *comp_counts = &counts->comps[i];
|
||||
|
||||
update_mv(w, comp_counts->sign, &comp->sign, MV_UPDATE_PROB);
|
||||
write_mv_update(av1_mv_class_tree, comp->classes, comp_counts->classes,
|
||||
MV_CLASSES, w);
|
||||
write_mv_update(av1_mv_class0_tree, comp->class0, comp_counts->class0,
|
||||
CLASS0_SIZE, w);
|
||||
for (j = 0; j < MV_OFFSET_BITS; ++j)
|
||||
update_mv(w, comp_counts->bits[j], &comp->bits[j], MV_UPDATE_PROB);
|
||||
}
|
||||
|
||||
for (i = 0; i < 2; ++i) {
|
||||
int j;
|
||||
for (j = 0; j < CLASS0_SIZE; ++j)
|
||||
write_mv_update(av1_mv_fp_tree, mvc->comps[i].class0_fp[j],
|
||||
counts->comps[i].class0_fp[j], MV_FP_SIZE, w);
|
||||
|
||||
write_mv_update(av1_mv_fp_tree, mvc->comps[i].fp, counts->comps[i].fp,
|
||||
MV_FP_SIZE, w);
|
||||
}
|
||||
#endif
|
||||
|
||||
if (usehp) {
|
||||
for (i = 0; i < 2; ++i) {
|
||||
|
|
@ -202,6 +178,7 @@ void av1_write_nmv_probs(AV1_COMMON *cm, int usehp, aom_writer *w,
|
|||
}
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
void av1_encode_mv(AV1_COMP *cpi, aom_writer *w, const MV *mv, const MV *ref,
|
||||
nmv_context *mvctx, int usehp) {
|
||||
|
|
@ -230,18 +207,19 @@ void av1_encode_dv(aom_writer *w, const MV *mv, const MV *ref,
|
|||
|
||||
aom_write_symbol(w, j, mvctx->joint_cdf, MV_JOINTS);
|
||||
if (mv_joint_vertical(j))
|
||||
encode_mv_component(w, diff.row, &mvctx->comps[0], 0);
|
||||
encode_mv_component(w, diff.row, &mvctx->comps[0], MV_SUBPEL_NONE);
|
||||
|
||||
if (mv_joint_horizontal(j))
|
||||
encode_mv_component(w, diff.col, &mvctx->comps[1], 0);
|
||||
encode_mv_component(w, diff.col, &mvctx->comps[1], MV_SUBPEL_NONE);
|
||||
}
|
||||
#endif // CONFIG_INTRABC
|
||||
|
||||
void av1_build_nmv_cost_table(int *mvjoint, int *mvcost[2],
|
||||
const nmv_context *ctx, int usehp) {
|
||||
const nmv_context *ctx,
|
||||
MvSubpelPrecision precision) {
|
||||
av1_cost_tokens(mvjoint, ctx->joints, av1_mv_joint_tree);
|
||||
build_nmv_component_cost_table(mvcost[0], &ctx->comps[0], usehp);
|
||||
build_nmv_component_cost_table(mvcost[1], &ctx->comps[1], usehp);
|
||||
build_nmv_component_cost_table(mvcost[0], &ctx->comps[0], precision);
|
||||
build_nmv_component_cost_table(mvcost[1], &ctx->comps[1], precision);
|
||||
}
|
||||
|
||||
#if CONFIG_EXT_INTER
|
||||
|
|
@ -284,6 +262,27 @@ static void inc_mvs(const MB_MODE_INFO *mbmi, const MB_MODE_INFO_EXT *mbmi_ext,
|
|||
mbmi_ext->ref_mv_stack[rf_type], 0, mbmi->ref_mv_idx);
|
||||
nmv_context_counts *counts = &nmv_counts[nmv_ctx];
|
||||
av1_inc_mv(&diff, counts, 1);
|
||||
#if CONFIG_COMPOUND_SINGLEREF
|
||||
} else {
|
||||
assert( // mode == SR_NEAREST_NEWMV ||
|
||||
mode == SR_NEAR_NEWMV || mode == SR_ZERO_NEWMV || mode == SR_NEW_NEWMV);
|
||||
const MV *ref = &mbmi_ext->ref_mvs[mbmi->ref_frame[0]][0].as_mv;
|
||||
int8_t rf_type = av1_ref_frame_type(mbmi->ref_frame);
|
||||
int nmv_ctx =
|
||||
av1_nmv_ctx(mbmi_ext->ref_mv_count[rf_type],
|
||||
mbmi_ext->ref_mv_stack[rf_type], 0, mbmi->ref_mv_idx);
|
||||
nmv_context_counts *counts = &nmv_counts[nmv_ctx];
|
||||
(void)pred_mvs;
|
||||
MV diff;
|
||||
if (mode == SR_NEW_NEWMV) {
|
||||
diff.row = mvs[0].as_mv.row - ref->row;
|
||||
diff.col = mvs[0].as_mv.col - ref->col;
|
||||
av1_inc_mv(&diff, counts, 1);
|
||||
}
|
||||
diff.row = mvs[1].as_mv.row - ref->row;
|
||||
diff.col = mvs[1].as_mv.col - ref->col;
|
||||
av1_inc_mv(&diff, counts, 1);
|
||||
#endif // CONFIG_COMPOUND_SINGLEREF
|
||||
}
|
||||
}
|
||||
|
||||
|
|
@ -328,7 +327,7 @@ static void inc_mvs_sub8x8(const MODE_INFO *mi, int block, const int_mv mvs[2],
|
|||
av1_inc_mv(&diff, counts, 1);
|
||||
}
|
||||
}
|
||||
#else
|
||||
#else // !CONFIG_EXT_INTER
|
||||
static void inc_mvs(const MB_MODE_INFO *mbmi, const MB_MODE_INFO_EXT *mbmi_ext,
|
||||
const int_mv mvs[2], const int_mv pred_mvs[2],
|
||||
nmv_context_counts *nmv_counts) {
|
||||
|
|
|
|||
5
third_party/aom/av1/encoder/encodemv.h
vendored
5
third_party/aom/av1/encoder/encodemv.h
vendored
|
|
@ -20,14 +20,17 @@ extern "C" {
|
|||
|
||||
void av1_entropy_mv_init(void);
|
||||
|
||||
#if !CONFIG_NEW_MULTISYMBOL
|
||||
void av1_write_nmv_probs(AV1_COMMON *cm, int usehp, aom_writer *w,
|
||||
nmv_context_counts *const counts);
|
||||
#endif
|
||||
|
||||
void av1_encode_mv(AV1_COMP *cpi, aom_writer *w, const MV *mv, const MV *ref,
|
||||
nmv_context *mvctx, int usehp);
|
||||
|
||||
void av1_build_nmv_cost_table(int *mvjoint, int *mvcost[2],
|
||||
const nmv_context *mvctx, int usehp);
|
||||
const nmv_context *mvctx,
|
||||
MvSubpelPrecision precision);
|
||||
|
||||
void av1_update_mv_count(ThreadData *td);
|
||||
|
||||
|
|
|
|||
1213
third_party/aom/av1/encoder/encoder.c
vendored
1213
third_party/aom/av1/encoder/encoder.c
vendored
File diff suppressed because it is too large
Load diff
119
third_party/aom/av1/encoder/encoder.h
vendored
119
third_party/aom/av1/encoder/encoder.h
vendored
|
|
@ -21,6 +21,7 @@
|
|||
#include "av1/common/entropymode.h"
|
||||
#include "av1/common/thread_common.h"
|
||||
#include "av1/common/onyxc_int.h"
|
||||
#include "av1/common/resize.h"
|
||||
#include "av1/encoder/aq_cyclicrefresh.h"
|
||||
#if CONFIG_ANS
|
||||
#include "aom_dsp/ans.h"
|
||||
|
|
@ -52,6 +53,10 @@
|
|||
extern "C" {
|
||||
#endif
|
||||
|
||||
#if CONFIG_SPEED_REFS
|
||||
#define MIN_SPEED_REFS_BLKSIZE BLOCK_16X16
|
||||
#endif // CONFIG_SPEED_REFS
|
||||
|
||||
typedef struct {
|
||||
int nmv_vec_cost[NMV_CONTEXTS][MV_JOINTS];
|
||||
int nmv_costs[NMV_CONTEXTS][2][MV_VALS];
|
||||
|
|
@ -128,7 +133,14 @@ typedef enum {
|
|||
RESIZE_NONE = 0, // No frame resizing allowed.
|
||||
RESIZE_FIXED = 1, // All frames are coded at the specified dimension.
|
||||
RESIZE_DYNAMIC = 2 // Coded size of each frame is determined by the codec.
|
||||
} RESIZE_TYPE;
|
||||
} RESIZE_MODE;
|
||||
#if CONFIG_FRAME_SUPERRES
|
||||
typedef enum {
|
||||
SUPERRES_NONE = 0,
|
||||
SUPERRES_FIXED = 1,
|
||||
SUPERRES_DYNAMIC = 2
|
||||
} SUPERRES_MODE;
|
||||
#endif // CONFIG_FRAME_SUPERRES
|
||||
|
||||
typedef struct AV1EncoderConfig {
|
||||
BITSTREAM_PROFILE profile;
|
||||
|
|
@ -190,22 +202,22 @@ typedef struct AV1EncoderConfig {
|
|||
int qm_minlevel;
|
||||
int qm_maxlevel;
|
||||
#endif
|
||||
#if CONFIG_TILE_GROUPS
|
||||
unsigned int num_tile_groups;
|
||||
unsigned int mtu;
|
||||
#endif
|
||||
|
||||
#if CONFIG_TEMPMV_SIGNALING
|
||||
unsigned int disable_tempmv;
|
||||
#endif
|
||||
// Internal frame size scaling.
|
||||
RESIZE_TYPE resize_mode;
|
||||
int scaled_frame_width;
|
||||
int scaled_frame_height;
|
||||
RESIZE_MODE resize_mode;
|
||||
uint8_t resize_scale_numerator;
|
||||
uint8_t resize_kf_scale_numerator;
|
||||
|
||||
#if CONFIG_FRAME_SUPERRES
|
||||
// Frame Super-Resolution size scaling
|
||||
int superres_enabled;
|
||||
// Frame Super-Resolution size scaling.
|
||||
SUPERRES_MODE superres_mode;
|
||||
uint8_t superres_scale_numerator;
|
||||
uint8_t superres_kf_scale_numerator;
|
||||
#endif // CONFIG_FRAME_SUPERRES
|
||||
|
||||
// Enable feature to reduce the frame quantization every x frames.
|
||||
|
|
@ -265,6 +277,10 @@ typedef struct AV1EncoderConfig {
|
|||
int use_highbitdepth;
|
||||
#endif
|
||||
aom_color_space_t color_space;
|
||||
#if CONFIG_COLORSPACE_HEADERS
|
||||
aom_transfer_function_t transfer_function;
|
||||
aom_chroma_sample_position_t chroma_sample_position;
|
||||
#endif
|
||||
int color_range;
|
||||
int render_width;
|
||||
int render_height;
|
||||
|
|
@ -276,7 +292,8 @@ typedef struct AV1EncoderConfig {
|
|||
int ans_window_size_log2;
|
||||
#endif // CONFIG_ANS && ANS_MAX_SYMBOLS
|
||||
#if CONFIG_EXT_TILE
|
||||
unsigned int tile_encoding_mode;
|
||||
unsigned int large_scale_tile;
|
||||
unsigned int single_tile_decoding;
|
||||
#endif // CONFIG_EXT_TILE
|
||||
|
||||
unsigned int motion_vector_unit_test;
|
||||
|
|
@ -289,8 +306,8 @@ static INLINE int is_lossless_requested(const AV1EncoderConfig *cfg) {
|
|||
// TODO(jingning) All spatially adaptive variables should go to TileDataEnc.
|
||||
typedef struct TileDataEnc {
|
||||
TileInfo tile_info;
|
||||
int thresh_freq_fact[BLOCK_SIZES][MAX_MODES];
|
||||
int mode_map[BLOCK_SIZES][MAX_MODES];
|
||||
int thresh_freq_fact[BLOCK_SIZES_ALL][MAX_MODES];
|
||||
int mode_map[BLOCK_SIZES_ALL][MAX_MODES];
|
||||
int m_search_count;
|
||||
int ex_search_count;
|
||||
#if CONFIG_PVQ
|
||||
|
|
@ -299,9 +316,7 @@ typedef struct TileDataEnc {
|
|||
#if CONFIG_CFL
|
||||
CFL_CTX cfl;
|
||||
#endif
|
||||
#if CONFIG_EC_ADAPT
|
||||
DECLARE_ALIGNED(16, FRAME_CONTEXT, tctx);
|
||||
#endif
|
||||
} TileDataEnc;
|
||||
|
||||
typedef struct RD_COUNTS {
|
||||
|
|
@ -311,6 +326,8 @@ typedef struct RD_COUNTS {
|
|||
// Stores number of 4x4 blocks using global motion per reference frame.
|
||||
int global_motion_used[TOTAL_REFS_PER_FRAME];
|
||||
#endif // CONFIG_GLOBAL_MOTION
|
||||
int single_ref_used_flag;
|
||||
int compound_ref_used_flag;
|
||||
} RD_COUNTS;
|
||||
|
||||
typedef struct ThreadData {
|
||||
|
|
@ -372,18 +389,11 @@ typedef struct AV1_COMP {
|
|||
|
||||
YV12_BUFFER_CONFIG *source;
|
||||
YV12_BUFFER_CONFIG *last_source; // NULL for first frame and alt_ref frames
|
||||
YV12_BUFFER_CONFIG *un_scaled_source;
|
||||
YV12_BUFFER_CONFIG *unscaled_source;
|
||||
YV12_BUFFER_CONFIG scaled_source;
|
||||
YV12_BUFFER_CONFIG *unscaled_last_source;
|
||||
YV12_BUFFER_CONFIG scaled_last_source;
|
||||
|
||||
// Up-sampled reference buffers
|
||||
// NOTE(zoeliu): It is needed to allocate sufficient space to the up-sampled
|
||||
// reference buffers, which should include the up-sampled version of all the
|
||||
// possibly stored references plus the currently coded frame itself.
|
||||
EncRefCntBuffer upsampled_ref_bufs[REF_FRAMES + 1];
|
||||
int upsampled_ref_idx[REF_FRAMES + 1];
|
||||
|
||||
// For a still frame, this flag is set to 1 to skip partition search.
|
||||
int partition_search_skippable_frame;
|
||||
|
||||
|
|
@ -471,7 +481,7 @@ typedef struct AV1_COMP {
|
|||
fractional_mv_step_fp *find_fractional_mv_step;
|
||||
av1_full_search_fn_t full_search_sad; // It is currently unused.
|
||||
av1_diamond_search_fn_t diamond_search_sad;
|
||||
aom_variance_fn_ptr_t fn_ptr[BLOCK_SIZES];
|
||||
aom_variance_fn_ptr_t fn_ptr[BLOCK_SIZES_ALL];
|
||||
uint64_t time_receive_data;
|
||||
uint64_t time_compress_data;
|
||||
uint64_t time_pick_lpf;
|
||||
|
|
@ -538,17 +548,24 @@ typedef struct AV1_COMP {
|
|||
#if CONFIG_EXT_INTER
|
||||
unsigned int inter_compound_mode_cost[INTER_MODE_CONTEXTS]
|
||||
[INTER_COMPOUND_MODES];
|
||||
#if CONFIG_COMPOUND_SINGLEREF
|
||||
unsigned int inter_singleref_comp_mode_cost[INTER_MODE_CONTEXTS]
|
||||
[INTER_SINGLEREF_COMP_MODES];
|
||||
#endif // CONFIG_COMPOUND_SINGLEREF
|
||||
#if CONFIG_INTERINTRA
|
||||
unsigned int interintra_mode_cost[BLOCK_SIZE_GROUPS][INTERINTRA_MODES];
|
||||
#endif // CONFIG_INTERINTRA
|
||||
#endif // CONFIG_EXT_INTER
|
||||
#if CONFIG_MOTION_VAR || CONFIG_WARPED_MOTION
|
||||
int motion_mode_cost[BLOCK_SIZES][MOTION_MODES];
|
||||
int motion_mode_cost[BLOCK_SIZES_ALL][MOTION_MODES];
|
||||
#if CONFIG_MOTION_VAR && CONFIG_WARPED_MOTION
|
||||
int motion_mode_cost1[BLOCK_SIZES][2];
|
||||
int motion_mode_cost1[BLOCK_SIZES_ALL][2];
|
||||
#endif // CONFIG_MOTION_VAR && CONFIG_WARPED_MOTION
|
||||
#if CONFIG_MOTION_VAR && CONFIG_NCOBMC_ADAPT_WEIGHT
|
||||
int ncobmc_mode_cost[ADAPT_OVERLAP_BLOCKS][MAX_NCOBMC_MODES];
|
||||
#endif // CONFIG_MOTION_VAR && CONFIG_NCOBMC_ADAPT_WEIGHT
|
||||
#endif // CONFIG_MOTION_VAR || CONFIG_WARPED_MOTION
|
||||
int intra_uv_mode_cost[INTRA_MODES][INTRA_MODES];
|
||||
int intra_uv_mode_cost[INTRA_MODES][UV_INTRA_MODES];
|
||||
int y_mode_costs[INTRA_MODES][INTRA_MODES][INTRA_MODES];
|
||||
int switchable_interp_costs[SWITCHABLE_FILTER_CONTEXTS][SWITCHABLE_FILTERS];
|
||||
#if CONFIG_EXT_PARTITION_TYPES
|
||||
|
|
@ -601,18 +618,10 @@ typedef struct AV1_COMP {
|
|||
TileBufferEnc tile_buffers[MAX_TILE_ROWS][MAX_TILE_COLS];
|
||||
|
||||
int resize_state;
|
||||
int resize_scale_num;
|
||||
int resize_scale_den;
|
||||
int resize_next_scale_num;
|
||||
int resize_next_scale_den;
|
||||
int resize_avg_qp;
|
||||
int resize_buffer_underflow;
|
||||
int resize_count;
|
||||
|
||||
#if CONFIG_FRAME_SUPERRES
|
||||
int superres_pending;
|
||||
#endif // CONFIG_FRAME_SUPERRES
|
||||
|
||||
// VARIANCE_AQ segment map refresh
|
||||
int vaq_refresh;
|
||||
|
||||
|
|
@ -640,6 +649,15 @@ typedef struct AV1_COMP {
|
|||
#if CONFIG_LV_MAP
|
||||
tran_low_t *tcoeff_buf[MAX_MB_PLANE];
|
||||
#endif
|
||||
|
||||
#if CONFIG_SPEED_REFS
|
||||
int sb_scanning_pass_idx;
|
||||
#endif // CONFIG_SPEED_REFS
|
||||
|
||||
#if CONFIG_FLEX_REFS
|
||||
int extra_arf_allowed;
|
||||
int bwd_ref_allowed;
|
||||
#endif // CONFIG_FLEX_REFS
|
||||
} AV1_COMP;
|
||||
|
||||
void av1_initialize_enc(void);
|
||||
|
|
@ -729,14 +747,6 @@ static INLINE YV12_BUFFER_CONFIG *get_ref_frame_buffer(
|
|||
: NULL;
|
||||
}
|
||||
|
||||
static INLINE const YV12_BUFFER_CONFIG *get_upsampled_ref(
|
||||
const AV1_COMP *cpi, const MV_REFERENCE_FRAME ref_frame) {
|
||||
// Use up-sampled reference frames.
|
||||
const int buf_idx =
|
||||
cpi->upsampled_ref_idx[get_ref_frame_map_idx(cpi, ref_frame)];
|
||||
return &cpi->upsampled_ref_bufs[buf_idx].buf;
|
||||
}
|
||||
|
||||
#if CONFIG_EXT_REFS || CONFIG_TEMPMV_SIGNALING
|
||||
static INLINE int enc_is_ref_frame_buf(AV1_COMP *cpi, RefCntBuffer *frame_buf) {
|
||||
MV_REFERENCE_FRAME ref_frame;
|
||||
|
|
@ -831,23 +841,22 @@ static INLINE void uref_cnt_fb(EncRefCntBuffer *ubufs, int *uidx,
|
|||
ubufs[new_uidx].ref_count++;
|
||||
}
|
||||
|
||||
// Returns 1 if a resize is pending and 0 otherwise.
|
||||
static INLINE int av1_resize_pending(const struct AV1_COMP *cpi) {
|
||||
return cpi->resize_scale_num != cpi->resize_next_scale_num ||
|
||||
cpi->resize_scale_den != cpi->resize_next_scale_den;
|
||||
}
|
||||
|
||||
// Returns 1 if a frame is unscaled and 0 otherwise.
|
||||
static INLINE int av1_resize_unscaled(const struct AV1_COMP *cpi) {
|
||||
return cpi->resize_scale_num == cpi->resize_scale_den;
|
||||
static INLINE int av1_resize_unscaled(const AV1_COMMON *cm) {
|
||||
#if CONFIG_FRAME_SUPERRES
|
||||
return cm->superres_upscaled_width == cm->render_width &&
|
||||
cm->superres_upscaled_height == cm->render_height;
|
||||
#else
|
||||
return cm->width == cm->render_width && cm->height == cm->render_height;
|
||||
#endif // CONFIG_FRAME_SUPERRES
|
||||
}
|
||||
|
||||
// Moves resizing to the next state. This is just setting the numerator and
|
||||
// denominator to the next numerator and denominator, causing
|
||||
// av1_resize_pending to subsequently return false.
|
||||
static INLINE void av1_resize_step(struct AV1_COMP *cpi) {
|
||||
cpi->resize_scale_num = cpi->resize_next_scale_num;
|
||||
cpi->resize_scale_den = cpi->resize_next_scale_den;
|
||||
static INLINE int av1_frame_unscaled(const AV1_COMMON *cm) {
|
||||
#if CONFIG_FRAME_SUPERRES
|
||||
return av1_superres_unscaled(cm) && av1_resize_unscaled(cm);
|
||||
#else
|
||||
return av1_resize_unscaled(cm);
|
||||
#endif // CONFIG_FRAME_SUPERRES
|
||||
}
|
||||
|
||||
#ifdef __cplusplus
|
||||
|
|
|
|||
347
third_party/aom/av1/encoder/encodetxb.c
vendored
347
third_party/aom/av1/encoder/encodetxb.c
vendored
|
|
@ -70,38 +70,43 @@ static void write_golomb(aom_writer *w, int level) {
|
|||
}
|
||||
|
||||
void av1_write_coeffs_txb(const AV1_COMMON *const cm, MACROBLOCKD *xd,
|
||||
aom_writer *w, int block, int plane,
|
||||
const tran_low_t *tcoeff, uint16_t eob,
|
||||
TXB_CTX *txb_ctx) {
|
||||
aom_writer *w, int blk_row, int blk_col, int block,
|
||||
int plane, TX_SIZE tx_size, const tran_low_t *tcoeff,
|
||||
uint16_t eob, TXB_CTX *txb_ctx) {
|
||||
aom_prob *nz_map;
|
||||
aom_prob *eob_flag;
|
||||
MB_MODE_INFO *mbmi = &xd->mi[0]->mbmi;
|
||||
const PLANE_TYPE plane_type = get_plane_type(plane);
|
||||
const TX_SIZE tx_size = get_tx_size(plane, xd);
|
||||
const TX_TYPE tx_type = get_tx_type(plane_type, xd, block, tx_size);
|
||||
const SCAN_ORDER *const scan_order =
|
||||
get_scan(cm, tx_size, tx_type, is_inter_block(mbmi));
|
||||
const TX_SIZE txs_ctx = get_txsize_context(tx_size);
|
||||
const TX_TYPE tx_type =
|
||||
av1_get_tx_type(plane_type, xd, blk_row, blk_col, block, tx_size);
|
||||
const SCAN_ORDER *const scan_order = get_scan(cm, tx_size, tx_type, mbmi);
|
||||
const int16_t *scan = scan_order->scan;
|
||||
const int16_t *iscan = scan_order->iscan;
|
||||
int c;
|
||||
int is_nz;
|
||||
const int bwl = b_width_log2_lookup[txsize_to_bsize[tx_size]] + 2;
|
||||
const int height = tx_size_high[tx_size];
|
||||
const int seg_eob = tx_size_2d[tx_size];
|
||||
uint8_t txb_mask[32 * 32] = { 0 };
|
||||
uint16_t update_eob = 0;
|
||||
|
||||
aom_write(w, eob == 0, cm->fc->txb_skip[tx_size][txb_ctx->txb_skip_ctx]);
|
||||
(void)blk_row;
|
||||
(void)blk_col;
|
||||
|
||||
aom_write(w, eob == 0, cm->fc->txb_skip[txs_ctx][txb_ctx->txb_skip_ctx]);
|
||||
|
||||
if (eob == 0) return;
|
||||
#if CONFIG_TXK_SEL
|
||||
av1_write_tx_type(cm, xd, block, plane, w);
|
||||
av1_write_tx_type(cm, xd, blk_row, blk_col, block, plane,
|
||||
get_min_tx_size(tx_size), w);
|
||||
#endif
|
||||
|
||||
nz_map = cm->fc->nz_map[tx_size][plane_type];
|
||||
eob_flag = cm->fc->eob_flag[tx_size][plane_type];
|
||||
nz_map = cm->fc->nz_map[txs_ctx][plane_type];
|
||||
eob_flag = cm->fc->eob_flag[txs_ctx][plane_type];
|
||||
|
||||
for (c = 0; c < eob; ++c) {
|
||||
int coeff_ctx = get_nz_map_ctx(tcoeff, txb_mask, scan[c], bwl);
|
||||
int eob_ctx = get_eob_ctx(tcoeff, scan[c], bwl);
|
||||
int coeff_ctx = get_nz_map_ctx(tcoeff, scan[c], bwl, height, iscan);
|
||||
int eob_ctx = get_eob_ctx(tcoeff, scan[c], txs_ctx);
|
||||
|
||||
tran_low_t v = tcoeff[scan[c]];
|
||||
is_nz = (v != 0);
|
||||
|
|
@ -113,12 +118,11 @@ void av1_write_coeffs_txb(const AV1_COMMON *const cm, MACROBLOCKD *xd,
|
|||
if (is_nz) {
|
||||
aom_write(w, c == (eob - 1), eob_flag[eob_ctx]);
|
||||
}
|
||||
txb_mask[scan[c]] = 1;
|
||||
}
|
||||
|
||||
int i;
|
||||
for (i = 0; i < NUM_BASE_LEVELS; ++i) {
|
||||
aom_prob *coeff_base = cm->fc->coeff_base[tx_size][plane_type][i];
|
||||
aom_prob *coeff_base = cm->fc->coeff_base[txs_ctx][plane_type][i];
|
||||
|
||||
update_eob = 0;
|
||||
for (c = eob - 1; c >= 0; --c) {
|
||||
|
|
@ -129,7 +133,7 @@ void av1_write_coeffs_txb(const AV1_COMMON *const cm, MACROBLOCKD *xd,
|
|||
|
||||
if (level <= i) continue;
|
||||
|
||||
ctx = get_base_ctx(tcoeff, scan[c], bwl, i + 1);
|
||||
ctx = get_base_ctx(tcoeff, scan[c], bwl, height, i + 1);
|
||||
|
||||
if (level == i + 1) {
|
||||
aom_write(w, 1, coeff_base[ctx]);
|
||||
|
|
@ -161,13 +165,13 @@ void av1_write_coeffs_txb(const AV1_COMMON *const cm, MACROBLOCKD *xd,
|
|||
}
|
||||
|
||||
// level is above 1.
|
||||
ctx = get_br_ctx(tcoeff, scan[c], bwl);
|
||||
ctx = get_br_ctx(tcoeff, scan[c], bwl, height);
|
||||
for (idx = 0; idx < COEFF_BASE_RANGE; ++idx) {
|
||||
if (level == (idx + 1 + NUM_BASE_LEVELS)) {
|
||||
aom_write(w, 1, cm->fc->coeff_lps[tx_size][plane_type][ctx]);
|
||||
aom_write(w, 1, cm->fc->coeff_lps[txs_ctx][plane_type][ctx]);
|
||||
break;
|
||||
}
|
||||
aom_write(w, 0, cm->fc->coeff_lps[tx_size][plane_type][ctx]);
|
||||
aom_write(w, 0, cm->fc->coeff_lps[txs_ctx][plane_type][ctx]);
|
||||
}
|
||||
if (idx < COEFF_BASE_RANGE) continue;
|
||||
|
||||
|
|
@ -183,7 +187,10 @@ void av1_write_coeffs_mb(const AV1_COMMON *const cm, MACROBLOCK *x,
|
|||
BLOCK_SIZE bsize = mbmi->sb_type;
|
||||
struct macroblockd_plane *pd = &xd->plane[plane];
|
||||
|
||||
#if CONFIG_CB4X4
|
||||
#if CONFIG_CHROMA_SUB8X8
|
||||
const BLOCK_SIZE plane_bsize =
|
||||
AOMMAX(BLOCK_4X4, get_plane_block_size(bsize, pd));
|
||||
#elif CONFIG_CB4X4
|
||||
const BLOCK_SIZE plane_bsize = get_plane_block_size(bsize, pd);
|
||||
#else
|
||||
const BLOCK_SIZE plane_bsize =
|
||||
|
|
@ -191,7 +198,7 @@ void av1_write_coeffs_mb(const AV1_COMMON *const cm, MACROBLOCK *x,
|
|||
#endif
|
||||
const int max_blocks_wide = max_block_wide(xd, plane_bsize, plane);
|
||||
const int max_blocks_high = max_block_high(xd, plane_bsize, plane);
|
||||
TX_SIZE tx_size = get_tx_size(plane, xd);
|
||||
const TX_SIZE tx_size = av1_get_tx_size(plane, xd);
|
||||
const int bkw = tx_size_wide_unit[tx_size];
|
||||
const int bkh = tx_size_high_unit[tx_size];
|
||||
const int step = tx_size_wide_unit[tx_size] * tx_size_high_unit[tx_size];
|
||||
|
|
@ -203,7 +210,8 @@ void av1_write_coeffs_mb(const AV1_COMMON *const cm, MACROBLOCK *x,
|
|||
uint16_t eob = x->mbmi_ext->eobs[plane][block];
|
||||
TXB_CTX txb_ctx = { x->mbmi_ext->txb_skip_ctx[plane][block],
|
||||
x->mbmi_ext->dc_sign_ctx[plane][block] };
|
||||
av1_write_coeffs_txb(cm, xd, w, block, plane, tcoeff, eob, &txb_ctx);
|
||||
av1_write_coeffs_txb(cm, xd, w, row, col, block, plane, tx_size, tcoeff,
|
||||
eob, &txb_ctx);
|
||||
block += step;
|
||||
}
|
||||
}
|
||||
|
|
@ -211,7 +219,7 @@ void av1_write_coeffs_mb(const AV1_COMMON *const cm, MACROBLOCK *x,
|
|||
|
||||
static INLINE void get_base_ctx_set(const tran_low_t *tcoeffs,
|
||||
int c, // raster order
|
||||
const int bwl,
|
||||
const int bwl, const int height,
|
||||
int ctx_set[NUM_BASE_LEVELS]) {
|
||||
const int row = c >> bwl;
|
||||
const int col = c - (row << bwl);
|
||||
|
|
@ -226,7 +234,7 @@ static INLINE void get_base_ctx_set(const tran_low_t *tcoeffs,
|
|||
int ref_col = col + base_ref_offset[idx][1];
|
||||
int pos = (ref_row << bwl) + ref_col;
|
||||
|
||||
if (ref_row < 0 || ref_col < 0 || ref_row >= stride || ref_col >= stride)
|
||||
if (ref_row < 0 || ref_col < 0 || ref_row >= height || ref_col >= stride)
|
||||
continue;
|
||||
|
||||
abs_coeff = abs(tcoeffs[pos]);
|
||||
|
|
@ -280,12 +288,14 @@ static INLINE int get_base_cost(tran_low_t abs_qc, int ctx,
|
|||
}
|
||||
|
||||
int av1_cost_coeffs_txb(const AV1_COMP *const cpi, MACROBLOCK *x, int plane,
|
||||
int block, TXB_CTX *txb_ctx) {
|
||||
int blk_row, int blk_col, int block, TX_SIZE tx_size,
|
||||
TXB_CTX *txb_ctx) {
|
||||
const AV1_COMMON *const cm = &cpi->common;
|
||||
MACROBLOCKD *const xd = &x->e_mbd;
|
||||
const TX_SIZE tx_size = get_tx_size(plane, xd);
|
||||
TX_SIZE txs_ctx = get_txsize_context(tx_size);
|
||||
const PLANE_TYPE plane_type = get_plane_type(plane);
|
||||
const TX_TYPE tx_type = get_tx_type(plane_type, xd, block, tx_size);
|
||||
const TX_TYPE tx_type =
|
||||
av1_get_tx_type(plane_type, xd, blk_row, blk_col, block, tx_size);
|
||||
MB_MODE_INFO *mbmi = &xd->mi[0]->mbmi;
|
||||
const struct macroblock_plane *p = &x->plane[plane];
|
||||
const int eob = p->eobs[block];
|
||||
|
|
@ -293,27 +303,26 @@ int av1_cost_coeffs_txb(const AV1_COMP *const cpi, MACROBLOCK *x, int plane,
|
|||
int c, cost;
|
||||
const int seg_eob = AOMMIN(eob, tx_size_2d[tx_size] - 1);
|
||||
int txb_skip_ctx = txb_ctx->txb_skip_ctx;
|
||||
aom_prob *nz_map = xd->fc->nz_map[tx_size][plane_type];
|
||||
aom_prob *nz_map = xd->fc->nz_map[txs_ctx][plane_type];
|
||||
|
||||
const int bwl = b_width_log2_lookup[txsize_to_bsize[tx_size]] + 2;
|
||||
// txb_mask is only initialized for once here. After that, it will be set when
|
||||
// coding zero map and then reset when coding level 1 info.
|
||||
uint8_t txb_mask[32 * 32] = { 0 };
|
||||
aom_prob(*coeff_base)[COEFF_BASE_CONTEXTS] =
|
||||
xd->fc->coeff_base[tx_size][plane_type];
|
||||
const int height = tx_size_high[tx_size];
|
||||
|
||||
const SCAN_ORDER *const scan_order =
|
||||
get_scan(cm, tx_size, tx_type, is_inter_block(mbmi));
|
||||
aom_prob(*coeff_base)[COEFF_BASE_CONTEXTS] =
|
||||
xd->fc->coeff_base[txs_ctx][plane_type];
|
||||
|
||||
const SCAN_ORDER *const scan_order = get_scan(cm, tx_size, tx_type, mbmi);
|
||||
const int16_t *scan = scan_order->scan;
|
||||
const int16_t *iscan = scan_order->iscan;
|
||||
|
||||
cost = 0;
|
||||
|
||||
if (eob == 0) {
|
||||
cost = av1_cost_bit(xd->fc->txb_skip[tx_size][txb_skip_ctx], 1);
|
||||
cost = av1_cost_bit(xd->fc->txb_skip[txs_ctx][txb_skip_ctx], 1);
|
||||
return cost;
|
||||
}
|
||||
|
||||
cost = av1_cost_bit(xd->fc->txb_skip[tx_size][txb_skip_ctx], 0);
|
||||
cost = av1_cost_bit(xd->fc->txb_skip[txs_ctx][txb_skip_ctx], 0);
|
||||
|
||||
#if CONFIG_TXK_SEL
|
||||
cost += av1_tx_type_cost(cpi, xd, mbmi->sb_type, plane, tx_size, tx_type);
|
||||
|
|
@ -325,7 +334,7 @@ int av1_cost_coeffs_txb(const AV1_COMP *const cpi, MACROBLOCK *x, int plane,
|
|||
int level = abs(v);
|
||||
|
||||
if (c < seg_eob) {
|
||||
int coeff_ctx = get_nz_map_ctx(qcoeff, txb_mask, scan[c], bwl);
|
||||
int coeff_ctx = get_nz_map_ctx(qcoeff, scan[c], bwl, height, iscan);
|
||||
cost += av1_cost_bit(nz_map[coeff_ctx], is_nz);
|
||||
}
|
||||
|
||||
|
|
@ -342,7 +351,7 @@ int av1_cost_coeffs_txb(const AV1_COMP *const cpi, MACROBLOCK *x, int plane,
|
|||
cost += av1_cost_bit(128, sign);
|
||||
}
|
||||
|
||||
get_base_ctx_set(qcoeff, scan[c], bwl, ctx_ls);
|
||||
get_base_ctx_set(qcoeff, scan[c], bwl, height, ctx_ls);
|
||||
|
||||
int i;
|
||||
for (i = 0; i < NUM_BASE_LEVELS; ++i) {
|
||||
|
|
@ -359,15 +368,15 @@ int av1_cost_coeffs_txb(const AV1_COMP *const cpi, MACROBLOCK *x, int plane,
|
|||
int idx;
|
||||
int ctx;
|
||||
|
||||
ctx = get_br_ctx(qcoeff, scan[c], bwl);
|
||||
ctx = get_br_ctx(qcoeff, scan[c], bwl, height);
|
||||
|
||||
for (idx = 0; idx < COEFF_BASE_RANGE; ++idx) {
|
||||
if (level == (idx + 1 + NUM_BASE_LEVELS)) {
|
||||
cost +=
|
||||
av1_cost_bit(xd->fc->coeff_lps[tx_size][plane_type][ctx], 1);
|
||||
av1_cost_bit(xd->fc->coeff_lps[txs_ctx][plane_type][ctx], 1);
|
||||
break;
|
||||
}
|
||||
cost += av1_cost_bit(xd->fc->coeff_lps[tx_size][plane_type][ctx], 0);
|
||||
cost += av1_cost_bit(xd->fc->coeff_lps[txs_ctx][plane_type][ctx], 0);
|
||||
}
|
||||
|
||||
if (idx >= COEFF_BASE_RANGE) {
|
||||
|
|
@ -389,13 +398,11 @@ int av1_cost_coeffs_txb(const AV1_COMP *const cpi, MACROBLOCK *x, int plane,
|
|||
}
|
||||
|
||||
if (c < seg_eob) {
|
||||
int eob_ctx = get_eob_ctx(qcoeff, scan[c], bwl);
|
||||
cost += av1_cost_bit(xd->fc->eob_flag[tx_size][plane_type][eob_ctx],
|
||||
int eob_ctx = get_eob_ctx(qcoeff, scan[c], txs_ctx);
|
||||
cost += av1_cost_bit(xd->fc->eob_flag[txs_ctx][plane_type][eob_ctx],
|
||||
c == (eob - 1));
|
||||
}
|
||||
}
|
||||
|
||||
txb_mask[scan[c]] = 1;
|
||||
}
|
||||
|
||||
return cost;
|
||||
|
|
@ -409,26 +416,26 @@ static INLINE int has_base(tran_low_t qc, int base_idx) {
|
|||
static void gen_base_count_mag_arr(int (*base_count_arr)[MAX_TX_SQUARE],
|
||||
int (*base_mag_arr)[2],
|
||||
const tran_low_t *qcoeff, int stride,
|
||||
int eob, const int16_t *scan) {
|
||||
int height, int eob, const int16_t *scan) {
|
||||
for (int c = 0; c < eob; ++c) {
|
||||
const int coeff_idx = scan[c]; // raster order
|
||||
if (!has_base(qcoeff[coeff_idx], 0)) continue;
|
||||
const int row = coeff_idx / stride;
|
||||
const int col = coeff_idx % stride;
|
||||
int *mag = base_mag_arr[coeff_idx];
|
||||
get_mag(mag, qcoeff, stride, row, col, base_ref_offset,
|
||||
get_mag(mag, qcoeff, stride, height, row, col, base_ref_offset,
|
||||
BASE_CONTEXT_POSITION_NUM);
|
||||
for (int i = 0; i < NUM_BASE_LEVELS; ++i) {
|
||||
if (!has_base(qcoeff[coeff_idx], i)) continue;
|
||||
int *count = base_count_arr[i] + coeff_idx;
|
||||
*count = get_level_count(qcoeff, stride, row, col, i, base_ref_offset,
|
||||
BASE_CONTEXT_POSITION_NUM);
|
||||
*count = get_level_count(qcoeff, stride, height, row, col, i,
|
||||
base_ref_offset, BASE_CONTEXT_POSITION_NUM);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static void gen_nz_count_arr(int(*nz_count_arr), const tran_low_t *qcoeff,
|
||||
int stride, int eob,
|
||||
int stride, int height, int eob,
|
||||
const SCAN_ORDER *scan_order) {
|
||||
const int16_t *scan = scan_order->scan;
|
||||
const int16_t *iscan = scan_order->iscan;
|
||||
|
|
@ -436,7 +443,8 @@ static void gen_nz_count_arr(int(*nz_count_arr), const tran_low_t *qcoeff,
|
|||
const int coeff_idx = scan[c]; // raster order
|
||||
const int row = coeff_idx / stride;
|
||||
const int col = coeff_idx % stride;
|
||||
nz_count_arr[coeff_idx] = get_nz_count(qcoeff, stride, row, col, iscan);
|
||||
nz_count_arr[coeff_idx] =
|
||||
get_nz_count(qcoeff, stride, height, row, col, iscan);
|
||||
}
|
||||
}
|
||||
|
||||
|
|
@ -478,8 +486,8 @@ static INLINE int has_br(tran_low_t qc) {
|
|||
}
|
||||
|
||||
static void gen_br_count_mag_arr(int *br_count_arr, int (*br_mag_arr)[2],
|
||||
const tran_low_t *qcoeff, int stride, int eob,
|
||||
const int16_t *scan) {
|
||||
const tran_low_t *qcoeff, int stride,
|
||||
int height, int eob, const int16_t *scan) {
|
||||
for (int c = 0; c < eob; ++c) {
|
||||
const int coeff_idx = scan[c]; // raster order
|
||||
if (!has_br(qcoeff[coeff_idx])) continue;
|
||||
|
|
@ -487,9 +495,9 @@ static void gen_br_count_mag_arr(int *br_count_arr, int (*br_mag_arr)[2],
|
|||
const int col = coeff_idx % stride;
|
||||
int *count = br_count_arr + coeff_idx;
|
||||
int *mag = br_mag_arr[coeff_idx];
|
||||
*count = get_level_count(qcoeff, stride, row, col, NUM_BASE_LEVELS,
|
||||
*count = get_level_count(qcoeff, stride, height, row, col, NUM_BASE_LEVELS,
|
||||
br_ref_offset, BR_CONTEXT_POSITION_NUM);
|
||||
get_mag(mag, qcoeff, stride, row, col, br_ref_offset,
|
||||
get_mag(mag, qcoeff, stride, height, row, col, br_ref_offset,
|
||||
BR_CONTEXT_POSITION_NUM);
|
||||
}
|
||||
}
|
||||
|
|
@ -543,18 +551,19 @@ static INLINE int get_golomb_cost(int abs_qc) {
|
|||
void gen_txb_cache(TxbCache *txb_cache, TxbInfo *txb_info) {
|
||||
const int16_t *scan = txb_info->scan_order->scan;
|
||||
gen_nz_count_arr(txb_cache->nz_count_arr, txb_info->qcoeff, txb_info->stride,
|
||||
txb_info->eob, txb_info->scan_order);
|
||||
txb_info->height, txb_info->eob, txb_info->scan_order);
|
||||
gen_nz_ctx_arr(txb_cache->nz_ctx_arr, txb_cache->nz_count_arr,
|
||||
txb_info->qcoeff, txb_info->bwl, txb_info->eob,
|
||||
txb_info->scan_order);
|
||||
gen_base_count_mag_arr(txb_cache->base_count_arr, txb_cache->base_mag_arr,
|
||||
txb_info->qcoeff, txb_info->stride, txb_info->eob,
|
||||
scan);
|
||||
txb_info->qcoeff, txb_info->stride, txb_info->height,
|
||||
txb_info->eob, scan);
|
||||
gen_base_ctx_arr(txb_cache->base_ctx_arr, txb_cache->base_count_arr,
|
||||
txb_cache->base_mag_arr, txb_info->qcoeff, txb_info->stride,
|
||||
txb_info->eob, scan);
|
||||
gen_br_count_mag_arr(txb_cache->br_count_arr, txb_cache->br_mag_arr,
|
||||
txb_info->qcoeff, txb_info->stride, txb_info->eob, scan);
|
||||
txb_info->qcoeff, txb_info->stride, txb_info->height,
|
||||
txb_info->eob, scan);
|
||||
gen_br_ctx_arr(txb_cache->br_ctx_arr, txb_cache->br_count_arr,
|
||||
txb_cache->br_mag_arr, txb_info->qcoeff, txb_info->stride,
|
||||
txb_info->eob, scan);
|
||||
|
|
@ -781,7 +790,7 @@ static int try_self_level_down(tran_low_t *low_coeff, int coeff_idx,
|
|||
|
||||
if (scan_idx < txb_info->seg_eob) {
|
||||
const int eob_ctx =
|
||||
get_eob_ctx(txb_info->qcoeff, coeff_idx, txb_info->bwl);
|
||||
get_eob_ctx(txb_info->qcoeff, coeff_idx, txb_info->txs_ctx);
|
||||
cost_diff -= av1_cost_bit(txb_probs->eob_flag[eob_ctx],
|
||||
scan_idx == (txb_info->eob - 1));
|
||||
}
|
||||
|
|
@ -853,9 +862,13 @@ int try_level_down(int coeff_idx, const TxbCache *txb_cache,
|
|||
const int nb_row = row - sig_ref_offset[i][0];
|
||||
const int nb_col = col - sig_ref_offset[i][1];
|
||||
const int nb_coeff_idx = nb_row * txb_info->stride + nb_col;
|
||||
|
||||
if (!(nb_row >= 0 && nb_col >= 0 && nb_row < txb_info->height &&
|
||||
nb_col < txb_info->stride))
|
||||
continue;
|
||||
|
||||
const int nb_scan_idx = iscan[nb_coeff_idx];
|
||||
if (nb_scan_idx < eob && nb_row >= 0 && nb_col >= 0 &&
|
||||
nb_row < txb_info->stride && nb_col < txb_info->stride) {
|
||||
if (nb_scan_idx < eob) {
|
||||
const int cost_diff = try_neighbor_level_down_nz(
|
||||
nb_coeff_idx, coeff_idx, txb_cache, txb_probs, txb_info);
|
||||
if (cost_map)
|
||||
|
|
@ -871,9 +884,13 @@ int try_level_down(int coeff_idx, const TxbCache *txb_cache,
|
|||
const int nb_row = row - base_ref_offset[i][0];
|
||||
const int nb_col = col - base_ref_offset[i][1];
|
||||
const int nb_coeff_idx = nb_row * txb_info->stride + nb_col;
|
||||
|
||||
if (!(nb_row >= 0 && nb_col >= 0 && nb_row < txb_info->height &&
|
||||
nb_col < txb_info->stride))
|
||||
continue;
|
||||
|
||||
const int nb_scan_idx = iscan[nb_coeff_idx];
|
||||
if (nb_scan_idx < eob && nb_row >= 0 && nb_col >= 0 &&
|
||||
nb_row < txb_info->stride && nb_col < txb_info->stride) {
|
||||
if (nb_scan_idx < eob) {
|
||||
const int cost_diff = try_neighbor_level_down_base(
|
||||
nb_coeff_idx, coeff_idx, txb_cache, txb_probs, txb_info);
|
||||
if (cost_map)
|
||||
|
|
@ -889,9 +906,13 @@ int try_level_down(int coeff_idx, const TxbCache *txb_cache,
|
|||
const int nb_row = row - br_ref_offset[i][0];
|
||||
const int nb_col = col - br_ref_offset[i][1];
|
||||
const int nb_coeff_idx = nb_row * txb_info->stride + nb_col;
|
||||
|
||||
if (!(nb_row >= 0 && nb_col >= 0 && nb_row < txb_info->height &&
|
||||
nb_col < txb_info->stride))
|
||||
continue;
|
||||
|
||||
const int nb_scan_idx = iscan[nb_coeff_idx];
|
||||
if (nb_scan_idx < eob && nb_row >= 0 && nb_col >= 0 &&
|
||||
nb_row < txb_info->stride && nb_col < txb_info->stride) {
|
||||
if (nb_scan_idx < eob) {
|
||||
const int cost_diff = try_neighbor_level_down_br(
|
||||
nb_coeff_idx, coeff_idx, txb_cache, txb_probs, txb_info);
|
||||
if (cost_map)
|
||||
|
|
@ -925,7 +946,7 @@ static int get_low_coeff_cost(int coeff_idx, const TxbCache *txb_cache,
|
|||
cost += get_base_cost(abs_qc, ctx, txb_probs->coeff_base, base_idx);
|
||||
if (scan_idx < txb_info->seg_eob) {
|
||||
const int eob_ctx =
|
||||
get_eob_ctx(txb_info->qcoeff, coeff_idx, txb_info->bwl);
|
||||
get_eob_ctx(txb_info->qcoeff, coeff_idx, txb_info->txs_ctx);
|
||||
cost += av1_cost_bit(txb_probs->eob_flag[eob_ctx],
|
||||
scan_idx == (txb_info->eob - 1));
|
||||
}
|
||||
|
|
@ -982,7 +1003,7 @@ int try_change_eob(int *new_eob, int coeff_idx, const TxbCache *txb_cache,
|
|||
// Note that get_eob_ctx does NOT actually account for qcoeff, so we don't
|
||||
// need to lower down the qcoeff here
|
||||
const int eob_ctx =
|
||||
get_eob_ctx(txb_info->qcoeff, scan[*new_eob - 1], txb_info->bwl);
|
||||
get_eob_ctx(txb_info->qcoeff, scan[*new_eob - 1], txb_info->txs_ctx);
|
||||
cost_diff -= av1_cost_bit(txb_probs->eob_flag[eob_ctx], 0);
|
||||
cost_diff += av1_cost_bit(txb_probs->eob_flag[eob_ctx], 1);
|
||||
} else {
|
||||
|
|
@ -1016,10 +1037,14 @@ void update_level_down(int coeff_idx, TxbCache *txb_cache, TxbInfo *txb_info) {
|
|||
for (int i = 0; i < SIG_REF_OFFSET_NUM; ++i) {
|
||||
const int nb_row = row - sig_ref_offset[i][0];
|
||||
const int nb_col = col - sig_ref_offset[i][1];
|
||||
|
||||
if (!(nb_row >= 0 && nb_col >= 0 && nb_row < txb_info->height &&
|
||||
nb_col < txb_info->stride))
|
||||
continue;
|
||||
|
||||
const int nb_coeff_idx = nb_row * txb_info->stride + nb_col;
|
||||
const int nb_scan_idx = iscan[nb_coeff_idx];
|
||||
if (nb_scan_idx < eob && nb_row >= 0 && nb_col >= 0 &&
|
||||
nb_row < txb_info->stride && nb_col < txb_info->stride) {
|
||||
if (nb_scan_idx < eob) {
|
||||
const int scan_idx = iscan[coeff_idx];
|
||||
if (scan_idx < nb_scan_idx) {
|
||||
const int level = 1;
|
||||
|
|
@ -1030,7 +1055,7 @@ void update_level_down(int coeff_idx, TxbCache *txb_cache, TxbInfo *txb_info) {
|
|||
const int count = txb_cache->nz_count_arr[nb_coeff_idx];
|
||||
txb_cache->nz_ctx_arr[nb_coeff_idx][0] = get_nz_map_ctx_from_count(
|
||||
count, txb_info->qcoeff, nb_coeff_idx, txb_info->bwl, iscan);
|
||||
// int ref_ctx = get_nz_map_ctx2(txb_info->qcoeff, nb_coeff_idx,
|
||||
// int ref_ctx = get_nz_map_ctx(txb_info->qcoeff, nb_coeff_idx,
|
||||
// txb_info->bwl, iscan);
|
||||
// if (ref_ctx != txb_cache->nz_ctx_arr[nb_coeff_idx][0])
|
||||
// printf("nz ctx %d ref_ctx %d\n",
|
||||
|
|
@ -1043,11 +1068,15 @@ void update_level_down(int coeff_idx, TxbCache *txb_cache, TxbInfo *txb_info) {
|
|||
const int nb_row = row - base_ref_offset[i][0];
|
||||
const int nb_col = col - base_ref_offset[i][1];
|
||||
const int nb_coeff_idx = nb_row * txb_info->stride + nb_col;
|
||||
|
||||
if (!(nb_row >= 0 && nb_col >= 0 && nb_row < txb_info->height &&
|
||||
nb_col < txb_info->stride))
|
||||
continue;
|
||||
|
||||
const tran_low_t nb_coeff = txb_info->qcoeff[nb_coeff_idx];
|
||||
if (!has_base(nb_coeff, 0)) continue;
|
||||
const int nb_scan_idx = iscan[nb_coeff_idx];
|
||||
if (nb_scan_idx < eob && nb_row >= 0 && nb_col >= 0 &&
|
||||
nb_row < txb_info->stride && nb_col < txb_info->stride) {
|
||||
if (nb_scan_idx < eob) {
|
||||
if (row >= nb_row && col >= nb_col)
|
||||
update_mag_arr(txb_cache->base_mag_arr[nb_coeff_idx], abs_qc);
|
||||
const int mag =
|
||||
|
|
@ -1076,11 +1105,15 @@ void update_level_down(int coeff_idx, TxbCache *txb_cache, TxbInfo *txb_info) {
|
|||
const int nb_row = row - br_ref_offset[i][0];
|
||||
const int nb_col = col - br_ref_offset[i][1];
|
||||
const int nb_coeff_idx = nb_row * txb_info->stride + nb_col;
|
||||
|
||||
if (!(nb_row >= 0 && nb_col >= 0 && nb_row < txb_info->height &&
|
||||
nb_col < txb_info->stride))
|
||||
continue;
|
||||
|
||||
const int nb_scan_idx = iscan[nb_coeff_idx];
|
||||
const tran_low_t nb_coeff = txb_info->qcoeff[nb_coeff_idx];
|
||||
if (!has_br(nb_coeff)) continue;
|
||||
if (nb_scan_idx < eob && nb_row >= 0 && nb_col >= 0 &&
|
||||
nb_row < txb_info->stride && nb_col < txb_info->stride) {
|
||||
if (nb_scan_idx < eob) {
|
||||
const int level = 1 + NUM_BASE_LEVELS;
|
||||
if (abs_qc == level) {
|
||||
txb_cache->br_count_arr[nb_coeff_idx] -= 1;
|
||||
|
|
@ -1112,8 +1145,8 @@ static int get_coeff_cost(tran_low_t qc, int scan_idx, TxbInfo *txb_info,
|
|||
const int16_t *iscan = txb_info->scan_order->iscan;
|
||||
|
||||
if (scan_idx < txb_info->seg_eob) {
|
||||
int coeff_ctx =
|
||||
get_nz_map_ctx2(txb_info->qcoeff, scan[scan_idx], txb_info->bwl, iscan);
|
||||
int coeff_ctx = get_nz_map_ctx(txb_info->qcoeff, scan[scan_idx],
|
||||
txb_info->bwl, txb_info->height, iscan);
|
||||
cost += av1_cost_bit(txb_probs->nz_map[coeff_ctx], is_nz);
|
||||
}
|
||||
|
||||
|
|
@ -1122,7 +1155,8 @@ static int get_coeff_cost(tran_low_t qc, int scan_idx, TxbInfo *txb_info,
|
|||
txb_ctx->dc_sign_ctx);
|
||||
|
||||
int ctx_ls[NUM_BASE_LEVELS] = { 0 };
|
||||
get_base_ctx_set(txb_info->qcoeff, scan[scan_idx], txb_info->bwl, ctx_ls);
|
||||
get_base_ctx_set(txb_info->qcoeff, scan[scan_idx], txb_info->bwl,
|
||||
txb_info->height, ctx_ls);
|
||||
|
||||
int i;
|
||||
for (i = 0; i < NUM_BASE_LEVELS; ++i) {
|
||||
|
|
@ -1130,14 +1164,15 @@ static int get_coeff_cost(tran_low_t qc, int scan_idx, TxbInfo *txb_info,
|
|||
}
|
||||
|
||||
if (abs_qc > NUM_BASE_LEVELS) {
|
||||
int ctx = get_br_ctx(txb_info->qcoeff, scan[scan_idx], txb_info->bwl);
|
||||
int ctx = get_br_ctx(txb_info->qcoeff, scan[scan_idx], txb_info->bwl,
|
||||
txb_info->height);
|
||||
cost += get_br_cost(abs_qc, ctx, txb_probs->coeff_lps);
|
||||
cost += get_golomb_cost(abs_qc);
|
||||
}
|
||||
|
||||
if (scan_idx < txb_info->seg_eob) {
|
||||
int eob_ctx =
|
||||
get_eob_ctx(txb_info->qcoeff, scan[scan_idx], txb_info->bwl);
|
||||
get_eob_ctx(txb_info->qcoeff, scan[scan_idx], txb_info->txs_ctx);
|
||||
cost += av1_cost_bit(txb_probs->eob_flag[eob_ctx],
|
||||
scan_idx == (txb_info->eob - 1));
|
||||
}
|
||||
|
|
@ -1323,8 +1358,7 @@ void try_level_down_facade(LevelDownStats *stats, int scan_idx,
|
|||
test_level_down(coeff_idx, txb_cache, txb_probs, txb_info);
|
||||
#endif
|
||||
}
|
||||
stats->rd_diff = RDCOST(txb_info->rdmult, txb_info->rddiv, stats->cost_diff,
|
||||
stats->dist_diff);
|
||||
stats->rd_diff = RDCOST(txb_info->rdmult, stats->cost_diff, stats->dist_diff);
|
||||
if (stats->rd_diff < 0) stats->update = 1;
|
||||
return;
|
||||
}
|
||||
|
|
@ -1424,18 +1458,17 @@ static int optimize_txb(TxbInfo *txb_info, const TxbProbs *txb_probs,
|
|||
|
||||
// These numbers are empirically obtained.
|
||||
static const int plane_rd_mult[REF_TYPES][PLANE_TYPES] = {
|
||||
#if CONFIG_EC_ADAPT
|
||||
{ 17, 13 }, { 16, 10 },
|
||||
#else
|
||||
{ 20, 12 }, { 16, 12 },
|
||||
#endif
|
||||
};
|
||||
|
||||
int av1_optimize_txb(const AV1_COMMON *cm, MACROBLOCK *x, int plane, int block,
|
||||
TX_SIZE tx_size, TXB_CTX *txb_ctx) {
|
||||
int av1_optimize_txb(const AV1_COMMON *cm, MACROBLOCK *x, int plane,
|
||||
int blk_row, int blk_col, int block, TX_SIZE tx_size,
|
||||
TXB_CTX *txb_ctx) {
|
||||
MACROBLOCKD *const xd = &x->e_mbd;
|
||||
const PLANE_TYPE plane_type = get_plane_type(plane);
|
||||
const TX_TYPE tx_type = get_tx_type(plane_type, xd, block, tx_size);
|
||||
const TX_SIZE txs_ctx = get_txsize_context(tx_size);
|
||||
const TX_TYPE tx_type =
|
||||
av1_get_tx_type(plane_type, xd, blk_row, blk_col, block, tx_size);
|
||||
const MB_MODE_INFO *mbmi = &xd->mi[0]->mbmi;
|
||||
const struct macroblock_plane *p = &x->plane[plane];
|
||||
struct macroblockd_plane *pd = &xd->plane[plane];
|
||||
|
|
@ -1445,34 +1478,34 @@ int av1_optimize_txb(const AV1_COMMON *cm, MACROBLOCK *x, int plane, int block,
|
|||
const tran_low_t *tcoeff = BLOCK_OFFSET(p->coeff, block);
|
||||
const int16_t *dequant = pd->dequant;
|
||||
const int seg_eob = AOMMIN(eob, tx_size_2d[tx_size] - 1);
|
||||
const aom_prob *nz_map = xd->fc->nz_map[tx_size][plane_type];
|
||||
const aom_prob *nz_map = xd->fc->nz_map[txs_ctx][plane_type];
|
||||
|
||||
const int bwl = b_width_log2_lookup[txsize_to_bsize[tx_size]] + 2;
|
||||
const int stride = 1 << bwl;
|
||||
const int height = tx_size_high[tx_size];
|
||||
aom_prob(*coeff_base)[COEFF_BASE_CONTEXTS] =
|
||||
xd->fc->coeff_base[tx_size][plane_type];
|
||||
xd->fc->coeff_base[txs_ctx][plane_type];
|
||||
|
||||
const aom_prob *coeff_lps = xd->fc->coeff_lps[tx_size][plane_type];
|
||||
const aom_prob *coeff_lps = xd->fc->coeff_lps[txs_ctx][plane_type];
|
||||
|
||||
const int is_inter = is_inter_block(mbmi);
|
||||
const SCAN_ORDER *const scan_order =
|
||||
get_scan(cm, tx_size, tx_type, is_inter_block(mbmi));
|
||||
const SCAN_ORDER *const scan_order = get_scan(cm, tx_size, tx_type, mbmi);
|
||||
|
||||
const TxbProbs txb_probs = { xd->fc->dc_sign[plane_type],
|
||||
nz_map,
|
||||
coeff_base,
|
||||
coeff_lps,
|
||||
xd->fc->eob_flag[tx_size][plane_type],
|
||||
xd->fc->txb_skip[tx_size] };
|
||||
xd->fc->eob_flag[txs_ctx][plane_type],
|
||||
xd->fc->txb_skip[txs_ctx] };
|
||||
|
||||
const int shift = av1_get_tx_scale(tx_size);
|
||||
const int64_t rdmult =
|
||||
(x->rdmult * plane_rd_mult[is_inter][plane_type] + 2) >> 2;
|
||||
const int64_t rddiv = x->rddiv;
|
||||
|
||||
TxbInfo txb_info = { qcoeff, dqcoeff, tcoeff, dequant, shift,
|
||||
tx_size, bwl, stride, eob, seg_eob,
|
||||
scan_order, txb_ctx, rdmult, rddiv };
|
||||
TxbInfo txb_info = { qcoeff, dqcoeff, tcoeff, dequant, shift,
|
||||
tx_size, txs_ctx, bwl, stride, height,
|
||||
eob, seg_eob, scan_order, txb_ctx, rdmult };
|
||||
|
||||
TxbCache txb_cache;
|
||||
gen_txb_cache(&txb_cache, &txb_info);
|
||||
|
||||
|
|
@ -1510,9 +1543,9 @@ void av1_update_txb_context_b(int plane, int block, int blk_row, int blk_col,
|
|||
const uint16_t eob = p->eobs[block];
|
||||
const tran_low_t *qcoeff = BLOCK_OFFSET(p->qcoeff, block);
|
||||
const PLANE_TYPE plane_type = pd->plane_type;
|
||||
const TX_TYPE tx_type = get_tx_type(plane_type, xd, block, tx_size);
|
||||
const SCAN_ORDER *const scan_order =
|
||||
get_scan(cm, tx_size, tx_type, is_inter_block(mbmi));
|
||||
const TX_TYPE tx_type =
|
||||
av1_get_tx_type(plane_type, xd, blk_row, blk_col, block, tx_size);
|
||||
const SCAN_ORDER *const scan_order = get_scan(cm, tx_size, tx_type, mbmi);
|
||||
(void)plane_bsize;
|
||||
|
||||
int cul_level = av1_get_txb_entropy_context(qcoeff, scan_order, eob);
|
||||
|
|
@ -1536,25 +1569,28 @@ void av1_update_and_record_txb_context(int plane, int block, int blk_row,
|
|||
const tran_low_t *qcoeff = BLOCK_OFFSET(p->qcoeff, block);
|
||||
tran_low_t *tcoeff = BLOCK_OFFSET(x->mbmi_ext->tcoeff[plane], block);
|
||||
const int segment_id = mbmi->segment_id;
|
||||
const TX_TYPE tx_type = get_tx_type(plane_type, xd, block, tx_size);
|
||||
const SCAN_ORDER *const scan_order =
|
||||
get_scan(cm, tx_size, tx_type, is_inter_block(mbmi));
|
||||
const TX_TYPE tx_type =
|
||||
av1_get_tx_type(plane_type, xd, blk_row, blk_col, block, tx_size);
|
||||
const SCAN_ORDER *const scan_order = get_scan(cm, tx_size, tx_type, mbmi);
|
||||
const int16_t *scan = scan_order->scan;
|
||||
const int16_t *iscan = scan_order->iscan;
|
||||
const int seg_eob = get_tx_eob(&cpi->common.seg, segment_id, tx_size);
|
||||
int c, i;
|
||||
TXB_CTX txb_ctx;
|
||||
get_txb_ctx(plane_bsize, tx_size, plane, pd->above_context + blk_col,
|
||||
pd->left_context + blk_row, &txb_ctx);
|
||||
const int bwl = b_width_log2_lookup[txsize_to_bsize[tx_size]] + 2;
|
||||
const int height = tx_size_high[tx_size];
|
||||
int cul_level = 0;
|
||||
unsigned int(*nz_map_count)[SIG_COEF_CONTEXTS][2];
|
||||
uint8_t txb_mask[32 * 32] = { 0 };
|
||||
|
||||
nz_map_count = &td->counts->nz_map[tx_size][plane_type];
|
||||
TX_SIZE txsize_ctx = get_txsize_context(tx_size);
|
||||
|
||||
nz_map_count = &td->counts->nz_map[txsize_ctx][plane_type];
|
||||
|
||||
memcpy(tcoeff, qcoeff, sizeof(*tcoeff) * seg_eob);
|
||||
|
||||
++td->counts->txb_skip[tx_size][txb_ctx.txb_skip_ctx][eob == 0];
|
||||
++td->counts->txb_skip[txsize_ctx][txb_ctx.txb_skip_ctx][eob == 0];
|
||||
x->mbmi_ext->txb_skip_ctx[plane][block] = txb_ctx.txb_skip_ctx;
|
||||
|
||||
x->mbmi_ext->eobs[plane][block] = eob;
|
||||
|
|
@ -1565,24 +1601,23 @@ void av1_update_and_record_txb_context(int plane, int block, int blk_row,
|
|||
}
|
||||
|
||||
#if CONFIG_TXK_SEL
|
||||
av1_update_tx_type_count(cm, xd, block, plane, mbmi->sb_type, tx_size,
|
||||
td->counts);
|
||||
av1_update_tx_type_count(cm, xd, blk_row, blk_col, block, plane,
|
||||
mbmi->sb_type, get_min_tx_size(tx_size), td->counts);
|
||||
#endif
|
||||
|
||||
for (c = 0; c < eob; ++c) {
|
||||
tran_low_t v = qcoeff[scan[c]];
|
||||
int is_nz = (v != 0);
|
||||
int coeff_ctx = get_nz_map_ctx(tcoeff, txb_mask, scan[c], bwl);
|
||||
int eob_ctx = get_eob_ctx(tcoeff, scan[c], bwl);
|
||||
int coeff_ctx = get_nz_map_ctx(tcoeff, scan[c], bwl, height, iscan);
|
||||
int eob_ctx = get_eob_ctx(tcoeff, scan[c], txsize_ctx);
|
||||
|
||||
if (c == seg_eob - 1) break;
|
||||
|
||||
++(*nz_map_count)[coeff_ctx][is_nz];
|
||||
|
||||
if (is_nz) {
|
||||
++td->counts->eob_flag[tx_size][plane_type][eob_ctx][c == (eob - 1)];
|
||||
++td->counts->eob_flag[txsize_ctx][plane_type][eob_ctx][c == (eob - 1)];
|
||||
}
|
||||
txb_mask[scan[c]] = 1;
|
||||
}
|
||||
|
||||
// Reverse process order to handle coefficient level and sign.
|
||||
|
|
@ -1595,10 +1630,10 @@ void av1_update_and_record_txb_context(int plane, int block, int blk_row,
|
|||
|
||||
if (level <= i) continue;
|
||||
|
||||
ctx = get_base_ctx(tcoeff, scan[c], bwl, i + 1);
|
||||
ctx = get_base_ctx(tcoeff, scan[c], bwl, height, i + 1);
|
||||
|
||||
if (level == i + 1) {
|
||||
++td->counts->coeff_base[tx_size][plane_type][i][ctx][1];
|
||||
++td->counts->coeff_base[txsize_ctx][plane_type][i][ctx][1];
|
||||
if (c == 0) {
|
||||
int dc_sign_ctx = txb_ctx.dc_sign_ctx;
|
||||
|
||||
|
|
@ -1608,7 +1643,7 @@ void av1_update_and_record_txb_context(int plane, int block, int blk_row,
|
|||
cul_level += level;
|
||||
continue;
|
||||
}
|
||||
++td->counts->coeff_base[tx_size][plane_type][i][ctx][0];
|
||||
++td->counts->coeff_base[txsize_ctx][plane_type][i][ctx][0];
|
||||
update_eob = AOMMAX(update_eob, c);
|
||||
}
|
||||
}
|
||||
|
|
@ -1630,13 +1665,13 @@ void av1_update_and_record_txb_context(int plane, int block, int blk_row,
|
|||
}
|
||||
|
||||
// level is above 1.
|
||||
ctx = get_br_ctx(tcoeff, scan[c], bwl);
|
||||
ctx = get_br_ctx(tcoeff, scan[c], bwl, height);
|
||||
for (idx = 0; idx < COEFF_BASE_RANGE; ++idx) {
|
||||
if (level == (idx + 1 + NUM_BASE_LEVELS)) {
|
||||
++td->counts->coeff_lps[tx_size][plane_type][ctx][1];
|
||||
++td->counts->coeff_lps[txsize_ctx][plane_type][ctx][1];
|
||||
break;
|
||||
}
|
||||
++td->counts->coeff_lps[tx_size][plane_type][ctx][0];
|
||||
++td->counts->coeff_lps[txsize_ctx][plane_type][ctx][0];
|
||||
}
|
||||
if (idx < COEFF_BASE_RANGE) continue;
|
||||
|
||||
|
|
@ -1835,46 +1870,74 @@ int64_t av1_search_txk_type(const AV1_COMP *cpi, MACROBLOCK *x, int plane,
|
|||
TX_TYPE txk_end = TX_TYPES - 1;
|
||||
TX_TYPE best_tx_type = txk_start;
|
||||
int64_t best_rd = INT64_MAX;
|
||||
uint8_t best_eob = 0;
|
||||
const int coeff_ctx = combine_entropy_contexts(*a, *l);
|
||||
RD_STATS best_rd_stats;
|
||||
TX_TYPE tx_type;
|
||||
|
||||
av1_invalid_rd_stats(&best_rd_stats);
|
||||
|
||||
for (tx_type = txk_start; tx_type <= txk_end; ++tx_type) {
|
||||
if (plane == 0) mbmi->txk_type[block] = tx_type;
|
||||
TX_TYPE ref_tx_type =
|
||||
get_tx_type(get_plane_type(plane), xd, block, tx_size);
|
||||
if (plane == 0) mbmi->txk_type[(blk_row << 4) + blk_col] = tx_type;
|
||||
TX_TYPE ref_tx_type = av1_get_tx_type(get_plane_type(plane), xd, blk_row,
|
||||
blk_col, block, tx_size);
|
||||
if (tx_type != ref_tx_type) {
|
||||
// use get_tx_type() to check if the tx_type is valid for the current mode
|
||||
// if it's not, we skip it here.
|
||||
// use av1_get_tx_type() to check if the tx_type is valid for the current
|
||||
// mode if it's not, we skip it here.
|
||||
continue;
|
||||
}
|
||||
|
||||
#if CONFIG_EXT_TX
|
||||
int is_inter = is_inter_block(mbmi);
|
||||
int ext_tx_set = get_ext_tx_set(get_min_tx_size(tx_size), mbmi->sb_type,
|
||||
is_inter, cm->reduced_tx_set_used);
|
||||
if (!(is_inter && ext_tx_used_inter[ext_tx_set][tx_type]) &&
|
||||
!(!is_inter && ext_tx_used_intra[ext_tx_set][tx_type]))
|
||||
continue;
|
||||
#endif // CONFIG_EXT_TX
|
||||
|
||||
RD_STATS this_rd_stats;
|
||||
av1_invalid_rd_stats(&this_rd_stats);
|
||||
av1_xform_quant(cm, x, plane, block, blk_row, blk_col, plane_bsize, tx_size,
|
||||
coeff_ctx, AV1_XFORM_QUANT_FP);
|
||||
av1_optimize_b(cm, x, plane, block, plane_bsize, tx_size, a, l);
|
||||
av1_optimize_b(cm, x, plane, blk_row, blk_col, block, plane_bsize, tx_size,
|
||||
a, l);
|
||||
av1_dist_block(cpi, x, plane, plane_bsize, block, blk_row, blk_col, tx_size,
|
||||
&this_rd_stats.dist, &this_rd_stats.sse,
|
||||
OUTPUT_HAS_PREDICTED_PIXELS);
|
||||
const SCAN_ORDER *scan_order =
|
||||
get_scan(cm, tx_size, tx_type, is_inter_block(mbmi));
|
||||
this_rd_stats.rate = av1_cost_coeffs(
|
||||
cpi, x, plane, block, tx_size, scan_order, a, l, use_fast_coef_costing);
|
||||
int rd =
|
||||
RDCOST(x->rdmult, x->rddiv, this_rd_stats.rate, this_rd_stats.dist);
|
||||
const SCAN_ORDER *scan_order = get_scan(cm, tx_size, tx_type, mbmi);
|
||||
this_rd_stats.rate =
|
||||
av1_cost_coeffs(cpi, x, plane, blk_row, blk_col, block, tx_size,
|
||||
scan_order, a, l, use_fast_coef_costing);
|
||||
int rd = RDCOST(x->rdmult, this_rd_stats.rate, this_rd_stats.dist);
|
||||
|
||||
if (rd < best_rd) {
|
||||
best_rd = rd;
|
||||
*rd_stats = this_rd_stats;
|
||||
best_rd_stats = this_rd_stats;
|
||||
best_tx_type = tx_type;
|
||||
best_eob = x->plane[plane].txb_entropy_ctx[block];
|
||||
}
|
||||
}
|
||||
if (plane == 0) mbmi->txk_type[block] = best_tx_type;
|
||||
// TODO(angiebird): Instead of re-call av1_xform_quant and av1_optimize_b,
|
||||
// copy the best result in the above tx_type search for loop
|
||||
av1_xform_quant(cm, x, plane, block, blk_row, blk_col, plane_bsize, tx_size,
|
||||
coeff_ctx, AV1_XFORM_QUANT_FP);
|
||||
av1_optimize_b(cm, x, plane, block, plane_bsize, tx_size, a, l);
|
||||
|
||||
av1_merge_rd_stats(rd_stats, &best_rd_stats);
|
||||
|
||||
// if (x->plane[plane].eobs[block] == 0)
|
||||
// if (best_tx_type != DCT_DCT)
|
||||
// exit(0);
|
||||
|
||||
if (best_eob == 0 && is_inter_block(mbmi)) best_tx_type = DCT_DCT;
|
||||
|
||||
if (plane == 0) mbmi->txk_type[(blk_row << 4) + blk_col] = best_tx_type;
|
||||
x->plane[plane].txb_entropy_ctx[block] = best_eob;
|
||||
|
||||
if (!is_inter_block(mbmi)) {
|
||||
// intra mode needs decoded result such that the next transform block
|
||||
// can use it for prediction.
|
||||
av1_xform_quant(cm, x, plane, block, blk_row, blk_col, plane_bsize, tx_size,
|
||||
coeff_ctx, AV1_XFORM_QUANT_FP);
|
||||
av1_optimize_b(cm, x, plane, blk_row, blk_col, block, plane_bsize, tx_size,
|
||||
a, l);
|
||||
|
||||
av1_inverse_transform_block_facade(xd, plane, block, blk_row, blk_col,
|
||||
x->plane[plane].eobs[block]);
|
||||
}
|
||||
|
|
|
|||
17
third_party/aom/av1/encoder/encodetxb.h
vendored
17
third_party/aom/av1/encoder/encodetxb.h
vendored
|
|
@ -30,14 +30,15 @@ typedef struct TxbInfo {
|
|||
const int16_t *dequant;
|
||||
int shift;
|
||||
TX_SIZE tx_size;
|
||||
TX_SIZE txs_ctx;
|
||||
int bwl;
|
||||
int stride;
|
||||
int height;
|
||||
int eob;
|
||||
int seg_eob;
|
||||
const SCAN_ORDER *scan_order;
|
||||
TXB_CTX *txb_ctx;
|
||||
int64_t rdmult;
|
||||
int64_t rddiv;
|
||||
} TxbInfo;
|
||||
|
||||
typedef struct TxbCache {
|
||||
|
|
@ -66,11 +67,12 @@ typedef struct TxbProbs {
|
|||
void av1_alloc_txb_buf(AV1_COMP *cpi);
|
||||
void av1_free_txb_buf(AV1_COMP *cpi);
|
||||
int av1_cost_coeffs_txb(const AV1_COMP *const cpi, MACROBLOCK *x, int plane,
|
||||
int block, TXB_CTX *txb_ctx);
|
||||
int blk_row, int blk_col, int block, TX_SIZE tx_size,
|
||||
TXB_CTX *txb_ctx);
|
||||
void av1_write_coeffs_txb(const AV1_COMMON *const cm, MACROBLOCKD *xd,
|
||||
aom_writer *w, int block, int plane,
|
||||
const tran_low_t *tcoeff, uint16_t eob,
|
||||
TXB_CTX *txb_ctx);
|
||||
aom_writer *w, int blk_row, int blk_col, int block,
|
||||
int plane, TX_SIZE tx_size, const tran_low_t *tcoeff,
|
||||
uint16_t eob, TXB_CTX *txb_ctx);
|
||||
void av1_write_coeffs_mb(const AV1_COMMON *const cm, MACROBLOCK *x,
|
||||
aom_writer *w, int plane);
|
||||
int av1_get_txb_entropy_context(const tran_low_t *qcoeff,
|
||||
|
|
@ -95,8 +97,9 @@ int64_t av1_search_txk_type(const AV1_COMP *cpi, MACROBLOCK *x, int plane,
|
|||
const ENTROPY_CONTEXT *a, const ENTROPY_CONTEXT *l,
|
||||
int use_fast_coef_costing, RD_STATS *rd_stats);
|
||||
#endif
|
||||
int av1_optimize_txb(const AV1_COMMON *cm, MACROBLOCK *x, int plane, int block,
|
||||
TX_SIZE tx_size, TXB_CTX *txb_ctx);
|
||||
int av1_optimize_txb(const AV1_COMMON *cm, MACROBLOCK *x, int plane,
|
||||
int blk_row, int blk_col, int block, TX_SIZE tx_size,
|
||||
TXB_CTX *txb_ctx);
|
||||
#ifdef __cplusplus
|
||||
}
|
||||
#endif
|
||||
|
|
|
|||
14
third_party/aom/av1/encoder/ethread.c
vendored
14
third_party/aom/av1/encoder/ethread.c
vendored
|
|
@ -26,6 +26,10 @@ static void accumulate_rd_opt(ThreadData *td, ThreadData *td_t) {
|
|||
td_t->rd_counts.global_motion_used[i];
|
||||
#endif // CONFIG_GLOBAL_MOTION
|
||||
|
||||
td->rd_counts.compound_ref_used_flag |=
|
||||
td_t->rd_counts.compound_ref_used_flag;
|
||||
td->rd_counts.single_ref_used_flag |= td_t->rd_counts.single_ref_used_flag;
|
||||
|
||||
for (i = 0; i < TX_SIZES; i++)
|
||||
for (j = 0; j < PLANE_TYPES; j++)
|
||||
for (k = 0; k < REF_TYPES; k++)
|
||||
|
|
@ -122,11 +126,9 @@ void av1_encode_tiles_mt(AV1_COMP *cpi) {
|
|||
|
||||
#if CONFIG_PALETTE
|
||||
// Allocate buffers used by palette coding mode.
|
||||
if (cpi->common.allow_screen_content_tools) {
|
||||
CHECK_MEM_ERROR(
|
||||
cm, thread_data->td->palette_buffer,
|
||||
aom_memalign(16, sizeof(*thread_data->td->palette_buffer)));
|
||||
}
|
||||
CHECK_MEM_ERROR(
|
||||
cm, thread_data->td->palette_buffer,
|
||||
aom_memalign(16, sizeof(*thread_data->td->palette_buffer)));
|
||||
#endif // CONFIG_PALETTE
|
||||
|
||||
// Create threads
|
||||
|
|
@ -168,7 +170,7 @@ void av1_encode_tiles_mt(AV1_COMP *cpi) {
|
|||
}
|
||||
|
||||
#if CONFIG_PALETTE
|
||||
if (cpi->common.allow_screen_content_tools && i < num_workers - 1)
|
||||
if (i < num_workers - 1)
|
||||
thread_data->td->mb.palette_buffer = thread_data->td->palette_buffer;
|
||||
#endif // CONFIG_PALETTE
|
||||
}
|
||||
|
|
|
|||
134
third_party/aom/av1/encoder/firstpass.c
vendored
134
third_party/aom/av1/encoder/firstpass.c
vendored
|
|
@ -456,6 +456,31 @@ static void set_first_pass_params(AV1_COMP *cpi) {
|
|||
cpi->rc.frames_to_key = INT_MAX;
|
||||
}
|
||||
|
||||
#if CONFIG_FLEX_REFS
|
||||
static double raw_motion_error_stdev(int *raw_motion_err_list,
|
||||
int raw_motion_err_counts) {
|
||||
int64_t sum_raw_err = 0;
|
||||
double raw_err_avg = 0;
|
||||
double raw_err_stdev = 0;
|
||||
if (raw_motion_err_counts == 0) return 0;
|
||||
|
||||
int i;
|
||||
for (i = 0; i < raw_motion_err_counts; i++) {
|
||||
sum_raw_err += raw_motion_err_list[i];
|
||||
}
|
||||
raw_err_avg = sum_raw_err / raw_motion_err_counts;
|
||||
for (i = 0; i < raw_motion_err_counts; i++) {
|
||||
raw_err_stdev += (raw_motion_err_list[i] - raw_err_avg) *
|
||||
(raw_motion_err_list[i] - raw_err_avg);
|
||||
}
|
||||
// Calculate the standard deviation for the motion error of all the inter
|
||||
// blocks of the 0,0 motion using the last source
|
||||
// frame as the reference.
|
||||
raw_err_stdev = sqrt(raw_err_stdev / raw_motion_err_counts);
|
||||
return raw_err_stdev;
|
||||
}
|
||||
#endif // CONFIG_FLEX_REFS
|
||||
|
||||
#define UL_INTRA_THRESH 50
|
||||
#define INVALID_ROW -1
|
||||
void av1_first_pass(AV1_COMP *cpi, const struct lookahead_entry *source) {
|
||||
|
|
@ -506,6 +531,13 @@ void av1_first_pass(AV1_COMP *cpi, const struct lookahead_entry *source) {
|
|||
od_adapt_ctx pvq_context;
|
||||
#endif
|
||||
|
||||
#if CONFIG_FLEX_REFS
|
||||
int *raw_motion_err_list;
|
||||
int raw_motion_err_counts = 0;
|
||||
CHECK_MEM_ERROR(
|
||||
cm, raw_motion_err_list,
|
||||
aom_calloc(cm->mb_rows * cm->mb_cols, sizeof(*raw_motion_err_list)));
|
||||
#endif // CONFIG_FLEX_REFS
|
||||
// First pass code requires valid last and new frame buffers.
|
||||
assert(new_yv12 != NULL);
|
||||
assert(frame_is_intra_only(cm) || (lst_yv12 != NULL));
|
||||
|
|
@ -968,6 +1000,9 @@ void av1_first_pass(AV1_COMP *cpi, const struct lookahead_entry *source) {
|
|||
}
|
||||
}
|
||||
}
|
||||
#if CONFIG_FLEX_REFS
|
||||
raw_motion_err_list[raw_motion_err_counts++] = raw_motion_error;
|
||||
#endif // CONFIG_FLEX_REFS
|
||||
} else {
|
||||
sr_coded_error += (int64_t)this_error;
|
||||
}
|
||||
|
|
@ -981,7 +1016,6 @@ void av1_first_pass(AV1_COMP *cpi, const struct lookahead_entry *source) {
|
|||
recon_yoffset += 16;
|
||||
recon_uvoffset += uv_mb_height;
|
||||
}
|
||||
|
||||
// Adjust to the next row of MBs.
|
||||
x->plane[0].src.buf += 16 * x->plane[0].src.stride - 16 * cm->mb_cols;
|
||||
x->plane[1].src.buf +=
|
||||
|
|
@ -991,7 +1025,10 @@ void av1_first_pass(AV1_COMP *cpi, const struct lookahead_entry *source) {
|
|||
|
||||
aom_clear_system_state();
|
||||
}
|
||||
|
||||
#if CONFIG_FLEX_REFS
|
||||
const double raw_err_stdev =
|
||||
raw_motion_error_stdev(raw_motion_err_list, raw_motion_err_counts);
|
||||
#endif // CONFIG_FLEX_REFS
|
||||
#if CONFIG_PVQ
|
||||
#if !CONFIG_ANS
|
||||
od_ec_enc_clear(&x->daala_enc.w.ec);
|
||||
|
|
@ -1045,6 +1082,9 @@ void av1_first_pass(AV1_COMP *cpi, const struct lookahead_entry *source) {
|
|||
fps.intra_skip_pct = (double)intra_skip_count / num_mbs;
|
||||
fps.inactive_zone_rows = (double)image_data_start_row;
|
||||
fps.inactive_zone_cols = (double)0; // TODO(paulwilkins): fix
|
||||
#if CONFIG_FLEX_REFS
|
||||
fps.raw_error_stdev = raw_err_stdev;
|
||||
#endif // CONFIG_FLEX_REFS
|
||||
|
||||
if (mvcount > 0) {
|
||||
fps.MVr = (double)sum_mvr / mvcount;
|
||||
|
|
@ -1231,27 +1271,6 @@ static void setup_rf_level_maxq(AV1_COMP *cpi) {
|
|||
}
|
||||
}
|
||||
|
||||
void av1_calculate_next_scaled_size(const AV1_COMP *cpi,
|
||||
int *scaled_frame_width,
|
||||
int *scaled_frame_height) {
|
||||
*scaled_frame_width =
|
||||
cpi->oxcf.width * cpi->resize_next_scale_num / cpi->resize_next_scale_den;
|
||||
*scaled_frame_height = cpi->oxcf.height * cpi->resize_next_scale_num /
|
||||
cpi->resize_next_scale_den;
|
||||
}
|
||||
|
||||
#if CONFIG_FRAME_SUPERRES
|
||||
void av1_calculate_superres_size(const AV1_COMP *cpi, int *encoded_width,
|
||||
int *encoded_height) {
|
||||
*encoded_width = cpi->oxcf.scaled_frame_width *
|
||||
cpi->common.superres_scale_numerator /
|
||||
SUPERRES_SCALE_DENOMINATOR;
|
||||
*encoded_height = cpi->oxcf.scaled_frame_height *
|
||||
cpi->common.superres_scale_numerator /
|
||||
SUPERRES_SCALE_DENOMINATOR;
|
||||
}
|
||||
#endif // CONFIG_FRAME_SUPERRES
|
||||
|
||||
void av1_init_second_pass(AV1_COMP *cpi) {
|
||||
const AV1EncoderConfig *const oxcf = &cpi->oxcf;
|
||||
TWO_PASS *const twopass = &cpi->twopass;
|
||||
|
|
@ -1673,6 +1692,9 @@ static void allocate_gf_group_bits(AV1_COMP *cpi, int64_t gf_group_bits,
|
|||
// (3) The bi-predictive group interval is strictly smaller than the
|
||||
// golden group interval.
|
||||
const int is_bipred_enabled =
|
||||
#if CONFIG_FLEX_REFS
|
||||
cpi->bwd_ref_allowed &&
|
||||
#endif
|
||||
rc->source_alt_ref_pending && rc->bipred_group_interval &&
|
||||
rc->bipred_group_interval <=
|
||||
(rc->baseline_gf_interval - rc->source_alt_ref_pending);
|
||||
|
|
@ -2046,6 +2068,11 @@ static void define_gf_group(AV1_COMP *cpi, FIRSTPASS_STATS *this_frame) {
|
|||
const int is_key_frame = frame_is_intra_only(cm);
|
||||
const int arf_active_or_kf = is_key_frame || rc->source_alt_ref_active;
|
||||
|
||||
#if CONFIG_FLEX_REFS
|
||||
cpi->extra_arf_allowed = 1;
|
||||
cpi->bwd_ref_allowed = 1;
|
||||
#endif
|
||||
|
||||
// Reset the GF group data structures unless this is a key
|
||||
// frame in which case it will already have been done.
|
||||
if (is_key_frame == 0) {
|
||||
|
|
@ -2106,6 +2133,12 @@ static void define_gf_group(AV1_COMP *cpi, FIRSTPASS_STATS *this_frame) {
|
|||
}
|
||||
}
|
||||
|
||||
#if CONFIG_FLEX_REFS
|
||||
double avg_sr_coded_error = 0;
|
||||
double avg_raw_err_stdev = 0;
|
||||
int non_zero_stdev_count = 0;
|
||||
#endif // CONFIG_FLEX_REFS
|
||||
|
||||
i = 0;
|
||||
while (i < rc->static_scene_max_gf_interval && i < rc->frames_to_key) {
|
||||
++i;
|
||||
|
|
@ -2129,6 +2162,14 @@ static void define_gf_group(AV1_COMP *cpi, FIRSTPASS_STATS *this_frame) {
|
|||
accumulate_frame_motion_stats(
|
||||
&next_frame, &this_frame_mv_in_out, &mv_in_out_accumulator,
|
||||
&abs_mv_in_out_accumulator, &mv_ratio_accumulator);
|
||||
#if CONFIG_FLEX_REFS
|
||||
// sum up the metric values of current gf group
|
||||
avg_sr_coded_error += next_frame.sr_coded_error;
|
||||
if (next_frame.raw_error_stdev) {
|
||||
non_zero_stdev_count++;
|
||||
avg_raw_err_stdev += next_frame.raw_error_stdev;
|
||||
}
|
||||
#endif // CONFIG_FLEX_REFS
|
||||
|
||||
// Accumulate the effect of prediction quality decay.
|
||||
if (!flash_detected) {
|
||||
|
|
@ -2175,7 +2216,6 @@ static void define_gf_group(AV1_COMP *cpi, FIRSTPASS_STATS *this_frame) {
|
|||
*this_frame = next_frame;
|
||||
old_boost_score = boost_score;
|
||||
}
|
||||
|
||||
twopass->gf_zeromotion_pct = (int)(zero_motion_accumulator * 1000.0);
|
||||
|
||||
// Was the group length constrained by the requirement for a new KF?
|
||||
|
|
@ -2202,11 +2242,35 @@ static void define_gf_group(AV1_COMP *cpi, FIRSTPASS_STATS *this_frame) {
|
|||
|
||||
// Set the interval until the next gf.
|
||||
rc->baseline_gf_interval = i - (is_key_frame || rc->source_alt_ref_pending);
|
||||
|
||||
#if CONFIG_EXT_REFS
|
||||
// Compute how many extra alt_refs we can have
|
||||
cpi->num_extra_arfs = get_number_of_extra_arfs(rc->baseline_gf_interval,
|
||||
rc->source_alt_ref_pending);
|
||||
#if CONFIG_FLEX_REFS
|
||||
const int num_mbs = (cpi->oxcf.resize_mode != RESIZE_NONE) ? cpi->initial_mbs
|
||||
: cpi->common.MBs;
|
||||
if (i) avg_sr_coded_error /= i;
|
||||
if (non_zero_stdev_count) avg_raw_err_stdev /= non_zero_stdev_count;
|
||||
|
||||
// Disable extra alter refs and backward ref for "still" gf group
|
||||
// zero_motion_accumulator indicates the minimum percentage of (0, 0) motion
|
||||
// in gf group
|
||||
// avg_sr_coded_error indicates the average of the sse per pixel of each frame
|
||||
// in gf group
|
||||
// avg_raw_err_stdev indicates the average of the standard deviation of (0, 0)
|
||||
// motion error per block of each frame in gf group
|
||||
assert(num_mbs > 0);
|
||||
const int disable_bwd_extarf =
|
||||
(zero_motion_accumulator > MIN_ZERO_MOTION &&
|
||||
avg_sr_coded_error / num_mbs < MAX_SR_CODED_ERROR &&
|
||||
avg_raw_err_stdev < MAX_RAW_ERR_VAR);
|
||||
|
||||
if (disable_bwd_extarf) cpi->extra_arf_allowed = cpi->bwd_ref_allowed = 0;
|
||||
|
||||
if (!cpi->extra_arf_allowed)
|
||||
cpi->num_extra_arfs = 0;
|
||||
else
|
||||
#endif // CONFIG_FLEX_REFS
|
||||
// Compute how many extra alt_refs we can have
|
||||
cpi->num_extra_arfs = get_number_of_extra_arfs(rc->baseline_gf_interval,
|
||||
rc->source_alt_ref_pending);
|
||||
// Currently at maximum two extra ARFs' are allowed
|
||||
assert(cpi->num_extra_arfs <= MAX_EXT_ARFS);
|
||||
#endif // CONFIG_EXT_REFS
|
||||
|
|
@ -2291,12 +2355,6 @@ static void define_gf_group(AV1_COMP *cpi, FIRSTPASS_STATS *this_frame) {
|
|||
twopass->section_intra_rating = calculate_section_intra_ratio(
|
||||
start_pos, twopass->stats_in_end, rc->baseline_gf_interval);
|
||||
}
|
||||
|
||||
if (oxcf->resize_mode == RESIZE_DYNAMIC) {
|
||||
// Default to starting GF groups at normal frame size.
|
||||
// TODO(afergs): Make a function for this
|
||||
cpi->resize_next_scale_num = cpi->resize_next_scale_den;
|
||||
}
|
||||
}
|
||||
|
||||
// Threshold for use of the lagging second reference frame. High second ref
|
||||
|
|
@ -2638,12 +2696,6 @@ static void find_next_key_frame(AV1_COMP *cpi, FIRSTPASS_STATS *this_frame) {
|
|||
// The count of bits left is adjusted elsewhere based on real coded frame
|
||||
// sizes.
|
||||
twopass->modified_error_left -= kf_group_err;
|
||||
|
||||
if (oxcf->resize_mode == RESIZE_DYNAMIC) {
|
||||
// Default to normal-sized frame on keyframes.
|
||||
// TODO(afergs): Make a function for this
|
||||
cpi->resize_next_scale_num = cpi->resize_next_scale_den;
|
||||
}
|
||||
}
|
||||
|
||||
// Define the reference buffers that will be updated post encode.
|
||||
|
|
@ -2741,7 +2793,7 @@ static void configure_buffer_updates(AV1_COMP *cpi) {
|
|||
break;
|
||||
|
||||
case LAST_BIPRED_UPDATE:
|
||||
cpi->refresh_last_frame = 0;
|
||||
cpi->refresh_last_frame = 1;
|
||||
cpi->refresh_golden_frame = 0;
|
||||
cpi->refresh_bwd_ref_frame = 0;
|
||||
cpi->refresh_alt_ref_frame = 0;
|
||||
|
|
|
|||
23
third_party/aom/av1/encoder/firstpass.h
vendored
23
third_party/aom/av1/encoder/firstpass.h
vendored
|
|
@ -52,6 +52,13 @@ typedef struct {
|
|||
#define MIN_EXT_ARF_INTERVAL 4
|
||||
#endif // CONFIG_EXT_REFS
|
||||
|
||||
#if CONFIG_FLEX_REFS
|
||||
#define MIN_ZERO_MOTION 0.95
|
||||
#define MAX_SR_CODED_ERROR 40
|
||||
#define MAX_RAW_ERR_VAR 2000
|
||||
#define MIN_MV_IN_OUT 0.4
|
||||
#endif // CONFIG_FLEX_REFS
|
||||
|
||||
#define VLOW_MOTION_THRESHOLD 950
|
||||
|
||||
typedef struct {
|
||||
|
|
@ -77,6 +84,10 @@ typedef struct {
|
|||
double new_mv_count;
|
||||
double duration;
|
||||
double count;
|
||||
#if CONFIG_FLEX_REFS
|
||||
// standard deviation for (0, 0) motion prediction error
|
||||
double raw_error_stdev;
|
||||
#endif // CONFIG_FLEX_REFS
|
||||
} FIRSTPASS_STATS;
|
||||
|
||||
typedef enum {
|
||||
|
|
@ -177,18 +188,6 @@ void av1_twopass_postencode_update(struct AV1_COMP *cpi);
|
|||
// Post encode update of the rate control parameters for 2-pass
|
||||
void av1_twopass_postencode_update(struct AV1_COMP *cpi);
|
||||
|
||||
void av1_calculate_next_scaled_size(const struct AV1_COMP *cpi,
|
||||
int *scaled_frame_width,
|
||||
int *scaled_frame_height);
|
||||
|
||||
#if CONFIG_FRAME_SUPERRES
|
||||
// This is the size after superress scaling, which could be 1:1.
|
||||
// Superres scaling happens after regular downscaling.
|
||||
// TODO(afergs): Limit overall reduction to 1/2 of the original size
|
||||
void av1_calculate_superres_size(const struct AV1_COMP *cpi, int *encoded_width,
|
||||
int *encoded_height);
|
||||
#endif // CONFIG_FRAME_SUPERRES
|
||||
|
||||
#if CONFIG_EXT_REFS
|
||||
static INLINE int get_number_of_extra_arfs(int interval, int arf_pending) {
|
||||
if (arf_pending && MAX_EXT_ARFS > 0)
|
||||
|
|
|
|||
26
third_party/aom/av1/encoder/global_motion.c
vendored
26
third_party/aom/av1/encoder/global_motion.c
vendored
|
|
@ -131,8 +131,8 @@ int64_t refine_integerized_param(WarpedMotionParams *wm,
|
|||
#endif // CONFIG_HIGHBITDEPTH
|
||||
uint8_t *ref, int r_width, int r_height,
|
||||
int r_stride, uint8_t *dst, int d_width,
|
||||
int d_height, int d_stride,
|
||||
int n_refinements) {
|
||||
int d_height, int d_stride, int n_refinements,
|
||||
int64_t best_frame_error) {
|
||||
static const int max_trans_model_params[TRANS_TYPES] = {
|
||||
0, 2, 4, 6, 8, 8, 8
|
||||
};
|
||||
|
|
@ -147,15 +147,16 @@ int64_t refine_integerized_param(WarpedMotionParams *wm,
|
|||
int32_t best_param;
|
||||
|
||||
force_wmtype(wm, wmtype);
|
||||
best_error = av1_warp_error(wm,
|
||||
best_error = av1_warp_error(
|
||||
wm,
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
use_hbd, bd,
|
||||
use_hbd, bd,
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
ref, r_width, r_height, r_stride,
|
||||
dst + border * d_stride + border, border, border,
|
||||
d_width - 2 * border, d_height - 2 * border,
|
||||
d_stride, 0, 0, 16, 16);
|
||||
step = 1 << (n_refinements + 1);
|
||||
ref, r_width, r_height, r_stride, dst + border * d_stride + border,
|
||||
border, border, d_width - 2 * border, d_height - 2 * border, d_stride, 0,
|
||||
0, SCALE_SUBPEL_SHIFTS, SCALE_SUBPEL_SHIFTS, best_frame_error);
|
||||
best_error = AOMMIN(best_error, best_frame_error);
|
||||
step = 1 << (n_refinements - 1);
|
||||
for (i = 0; i < n_refinements; i++, step >>= 1) {
|
||||
for (p = 0; p < n_params; ++p) {
|
||||
int step_dir = 0;
|
||||
|
|
@ -174,7 +175,7 @@ int64_t refine_integerized_param(WarpedMotionParams *wm,
|
|||
#endif // CONFIG_HIGHBITDEPTH
|
||||
ref, r_width, r_height, r_stride, dst + border * d_stride + border,
|
||||
border, border, d_width - 2 * border, d_height - 2 * border, d_stride,
|
||||
0, 0, 16, 16);
|
||||
0, 0, SCALE_SUBPEL_SHIFTS, SCALE_SUBPEL_SHIFTS, best_error);
|
||||
if (step_error < best_error) {
|
||||
best_error = step_error;
|
||||
best_param = *param;
|
||||
|
|
@ -190,7 +191,7 @@ int64_t refine_integerized_param(WarpedMotionParams *wm,
|
|||
#endif // CONFIG_HIGHBITDEPTH
|
||||
ref, r_width, r_height, r_stride, dst + border * d_stride + border,
|
||||
border, border, d_width - 2 * border, d_height - 2 * border, d_stride,
|
||||
0, 0, 16, 16);
|
||||
0, 0, SCALE_SUBPEL_SHIFTS, SCALE_SUBPEL_SHIFTS, best_error);
|
||||
if (step_error < best_error) {
|
||||
best_error = step_error;
|
||||
best_param = *param;
|
||||
|
|
@ -209,7 +210,8 @@ int64_t refine_integerized_param(WarpedMotionParams *wm,
|
|||
#endif // CONFIG_HIGHBITDEPTH
|
||||
ref, r_width, r_height, r_stride, dst + border * d_stride + border,
|
||||
border, border, d_width - 2 * border, d_height - 2 * border,
|
||||
d_stride, 0, 0, 16, 16);
|
||||
d_stride, 0, 0, SCALE_SUBPEL_SHIFTS, SCALE_SUBPEL_SHIFTS,
|
||||
best_error);
|
||||
if (step_error < best_error) {
|
||||
best_error = step_error;
|
||||
best_param = *param;
|
||||
|
|
|
|||
3
third_party/aom/av1/encoder/global_motion.h
vendored
3
third_party/aom/av1/encoder/global_motion.h
vendored
|
|
@ -36,7 +36,8 @@ int64_t refine_integerized_param(WarpedMotionParams *wm,
|
|||
#endif // CONFIG_HIGHBITDEPTH
|
||||
uint8_t *ref, int r_width, int r_height,
|
||||
int r_stride, uint8_t *dst, int d_width,
|
||||
int d_height, int d_stride, int n_refinements);
|
||||
int d_height, int d_stride, int n_refinements,
|
||||
int64_t best_frame_error);
|
||||
|
||||
/*
|
||||
Computes "num_motions" candidate global motion parameters between two frames.
|
||||
|
|
|
|||
421
third_party/aom/av1/encoder/hybrid_fwd_txfm.c
vendored
421
third_party/aom/av1/encoder/hybrid_fwd_txfm.c
vendored
|
|
@ -18,7 +18,7 @@
|
|||
|
||||
#if CONFIG_CHROMA_2X2
|
||||
static void fwd_txfm_2x2(const int16_t *src_diff, tran_low_t *coeff,
|
||||
int diff_stride, TX_TYPE tx_type, int lossless) {
|
||||
int diff_stride, TxfmParam *txfm_param) {
|
||||
tran_high_t a1 = src_diff[0];
|
||||
tran_high_t b1 = src_diff[1];
|
||||
tran_high_t c1 = src_diff[diff_stride];
|
||||
|
|
@ -39,134 +39,151 @@ static void fwd_txfm_2x2(const int16_t *src_diff, tran_low_t *coeff,
|
|||
coeff[2] = (tran_low_t)(4 * c1);
|
||||
coeff[3] = (tran_low_t)(4 * d1);
|
||||
|
||||
(void)tx_type;
|
||||
(void)lossless;
|
||||
(void)txfm_param;
|
||||
}
|
||||
#endif
|
||||
|
||||
static void fwd_txfm_4x4(const int16_t *src_diff, tran_low_t *coeff,
|
||||
int diff_stride, TX_TYPE tx_type, int lossless) {
|
||||
if (lossless) {
|
||||
assert(tx_type == DCT_DCT);
|
||||
int diff_stride, TxfmParam *txfm_param) {
|
||||
if (txfm_param->lossless) {
|
||||
assert(txfm_param->tx_type == DCT_DCT);
|
||||
av1_fwht4x4(src_diff, coeff, diff_stride);
|
||||
return;
|
||||
}
|
||||
|
||||
av1_fht4x4(src_diff, coeff, diff_stride, tx_type);
|
||||
#if CONFIG_LGT
|
||||
// only C version has LGTs
|
||||
av1_fht4x4_c(src_diff, coeff, diff_stride, txfm_param);
|
||||
#else
|
||||
av1_fht4x4(src_diff, coeff, diff_stride, txfm_param);
|
||||
#endif
|
||||
}
|
||||
|
||||
static void fwd_txfm_4x8(const int16_t *src_diff, tran_low_t *coeff,
|
||||
int diff_stride, TX_TYPE tx_type,
|
||||
FWD_TXFM_OPT fwd_txfm_opt) {
|
||||
(void)fwd_txfm_opt;
|
||||
av1_fht4x8(src_diff, coeff, diff_stride, tx_type);
|
||||
int diff_stride, TxfmParam *txfm_param) {
|
||||
#if CONFIG_LGT
|
||||
av1_fht4x8_c(src_diff, coeff, diff_stride, txfm_param);
|
||||
#else
|
||||
av1_fht4x8(src_diff, coeff, diff_stride, txfm_param);
|
||||
#endif
|
||||
}
|
||||
|
||||
static void fwd_txfm_8x4(const int16_t *src_diff, tran_low_t *coeff,
|
||||
int diff_stride, TX_TYPE tx_type,
|
||||
FWD_TXFM_OPT fwd_txfm_opt) {
|
||||
(void)fwd_txfm_opt;
|
||||
av1_fht8x4(src_diff, coeff, diff_stride, tx_type);
|
||||
int diff_stride, TxfmParam *txfm_param) {
|
||||
#if CONFIG_LGT
|
||||
av1_fht8x4_c(src_diff, coeff, diff_stride, txfm_param);
|
||||
#else
|
||||
av1_fht8x4(src_diff, coeff, diff_stride, txfm_param);
|
||||
#endif
|
||||
}
|
||||
|
||||
static void fwd_txfm_8x16(const int16_t *src_diff, tran_low_t *coeff,
|
||||
int diff_stride, TX_TYPE tx_type,
|
||||
FWD_TXFM_OPT fwd_txfm_opt) {
|
||||
(void)fwd_txfm_opt;
|
||||
av1_fht8x16(src_diff, coeff, diff_stride, tx_type);
|
||||
int diff_stride, TxfmParam *txfm_param) {
|
||||
#if CONFIG_LGT
|
||||
av1_fht8x16_c(src_diff, coeff, diff_stride, txfm_param);
|
||||
#else
|
||||
av1_fht8x16(src_diff, coeff, diff_stride, txfm_param);
|
||||
#endif
|
||||
}
|
||||
|
||||
static void fwd_txfm_16x8(const int16_t *src_diff, tran_low_t *coeff,
|
||||
int diff_stride, TX_TYPE tx_type,
|
||||
FWD_TXFM_OPT fwd_txfm_opt) {
|
||||
(void)fwd_txfm_opt;
|
||||
av1_fht16x8(src_diff, coeff, diff_stride, tx_type);
|
||||
int diff_stride, TxfmParam *txfm_param) {
|
||||
#if CONFIG_LGT
|
||||
av1_fht16x8_c(src_diff, coeff, diff_stride, txfm_param);
|
||||
#else
|
||||
av1_fht16x8(src_diff, coeff, diff_stride, txfm_param);
|
||||
#endif
|
||||
}
|
||||
|
||||
static void fwd_txfm_16x32(const int16_t *src_diff, tran_low_t *coeff,
|
||||
int diff_stride, TX_TYPE tx_type,
|
||||
FWD_TXFM_OPT fwd_txfm_opt) {
|
||||
(void)fwd_txfm_opt;
|
||||
av1_fht16x32(src_diff, coeff, diff_stride, tx_type);
|
||||
int diff_stride, TxfmParam *txfm_param) {
|
||||
av1_fht16x32(src_diff, coeff, diff_stride, txfm_param);
|
||||
}
|
||||
|
||||
static void fwd_txfm_32x16(const int16_t *src_diff, tran_low_t *coeff,
|
||||
int diff_stride, TX_TYPE tx_type,
|
||||
FWD_TXFM_OPT fwd_txfm_opt) {
|
||||
(void)fwd_txfm_opt;
|
||||
av1_fht32x16(src_diff, coeff, diff_stride, tx_type);
|
||||
int diff_stride, TxfmParam *txfm_param) {
|
||||
av1_fht32x16(src_diff, coeff, diff_stride, txfm_param);
|
||||
}
|
||||
|
||||
static void fwd_txfm_8x8(const int16_t *src_diff, tran_low_t *coeff,
|
||||
int diff_stride, TX_TYPE tx_type,
|
||||
FWD_TXFM_OPT fwd_txfm_opt) {
|
||||
(void)fwd_txfm_opt;
|
||||
av1_fht8x8(src_diff, coeff, diff_stride, tx_type);
|
||||
int diff_stride, TxfmParam *txfm_param) {
|
||||
#if CONFIG_LGT
|
||||
av1_fht8x8_c(src_diff, coeff, diff_stride, txfm_param);
|
||||
#else
|
||||
av1_fht8x8(src_diff, coeff, diff_stride, txfm_param);
|
||||
#endif
|
||||
}
|
||||
|
||||
static void fwd_txfm_16x16(const int16_t *src_diff, tran_low_t *coeff,
|
||||
int diff_stride, TX_TYPE tx_type,
|
||||
FWD_TXFM_OPT fwd_txfm_opt) {
|
||||
(void)fwd_txfm_opt;
|
||||
av1_fht16x16(src_diff, coeff, diff_stride, tx_type);
|
||||
int diff_stride, TxfmParam *txfm_param) {
|
||||
av1_fht16x16(src_diff, coeff, diff_stride, txfm_param);
|
||||
}
|
||||
|
||||
static void fwd_txfm_32x32(const int16_t *src_diff, tran_low_t *coeff,
|
||||
int diff_stride, TX_TYPE tx_type,
|
||||
FWD_TXFM_OPT fwd_txfm_opt) {
|
||||
(void)fwd_txfm_opt;
|
||||
av1_fht32x32(src_diff, coeff, diff_stride, tx_type);
|
||||
int diff_stride, TxfmParam *txfm_param) {
|
||||
#if CONFIG_MRC_TX
|
||||
// MRC_DCT currently only has a C implementation
|
||||
if (txfm_param->tx_type == MRC_DCT) {
|
||||
av1_fht32x32_c(src_diff, coeff, diff_stride, txfm_param);
|
||||
return;
|
||||
}
|
||||
#endif // CONFIG_MRC_TX
|
||||
av1_fht32x32(src_diff, coeff, diff_stride, txfm_param);
|
||||
}
|
||||
|
||||
#if CONFIG_TX64X64
|
||||
static void fwd_txfm_64x64(const int16_t *src_diff, tran_low_t *coeff,
|
||||
int diff_stride, TX_TYPE tx_type,
|
||||
FWD_TXFM_OPT fwd_txfm_opt) {
|
||||
(void)fwd_txfm_opt;
|
||||
int diff_stride, TxfmParam *txfm_param) {
|
||||
#if CONFIG_EXT_TX
|
||||
if (tx_type == IDTX)
|
||||
av1_fwd_idtx_c(src_diff, coeff, diff_stride, 64, tx_type);
|
||||
if (txfm_param->tx_type == IDTX)
|
||||
av1_fwd_idtx_c(src_diff, coeff, diff_stride, 64, txfm_param->tx_type);
|
||||
else
|
||||
#endif
|
||||
av1_fht64x64(src_diff, coeff, diff_stride, tx_type);
|
||||
av1_fht64x64(src_diff, coeff, diff_stride, txfm_param);
|
||||
}
|
||||
#endif // CONFIG_TX64X64
|
||||
|
||||
#if CONFIG_EXT_TX && CONFIG_RECT_TX && CONFIG_RECT_TX_EXT
|
||||
#if CONFIG_RECT_TX_EXT && (CONFIG_EXT_TX || CONFIG_VAR_TX)
|
||||
static void fwd_txfm_16x4(const int16_t *src_diff, tran_low_t *coeff,
|
||||
int diff_stride, TX_TYPE tx_type,
|
||||
FWD_TXFM_OPT fwd_txfm_opt) {
|
||||
(void)fwd_txfm_opt;
|
||||
av1_fht16x4(src_diff, coeff, diff_stride, tx_type);
|
||||
int diff_stride, TxfmParam *txfm_param) {
|
||||
#if CONFIG_LGT
|
||||
av1_fht16x4_c(src_diff, coeff, diff_stride, txfm_param);
|
||||
#else
|
||||
av1_fht16x4(src_diff, coeff, diff_stride, txfm_param);
|
||||
#endif
|
||||
}
|
||||
|
||||
static void fwd_txfm_4x16(const int16_t *src_diff, tran_low_t *coeff,
|
||||
int diff_stride, TX_TYPE tx_type,
|
||||
FWD_TXFM_OPT fwd_txfm_opt) {
|
||||
(void)fwd_txfm_opt;
|
||||
av1_fht4x16(src_diff, coeff, diff_stride, tx_type);
|
||||
int diff_stride, TxfmParam *txfm_param) {
|
||||
#if CONFIG_LGT
|
||||
av1_fht4x16_c(src_diff, coeff, diff_stride, txfm_param);
|
||||
#else
|
||||
av1_fht4x16(src_diff, coeff, diff_stride, txfm_param);
|
||||
#endif
|
||||
}
|
||||
|
||||
static void fwd_txfm_32x8(const int16_t *src_diff, tran_low_t *coeff,
|
||||
int diff_stride, TX_TYPE tx_type,
|
||||
FWD_TXFM_OPT fwd_txfm_opt) {
|
||||
(void)fwd_txfm_opt;
|
||||
av1_fht32x8(src_diff, coeff, diff_stride, tx_type);
|
||||
int diff_stride, TxfmParam *txfm_param) {
|
||||
#if CONFIG_LGT
|
||||
av1_fht32x8_c(src_diff, coeff, diff_stride, txfm_param);
|
||||
#else
|
||||
av1_fht32x8(src_diff, coeff, diff_stride, txfm_param);
|
||||
#endif
|
||||
}
|
||||
|
||||
static void fwd_txfm_8x32(const int16_t *src_diff, tran_low_t *coeff,
|
||||
int diff_stride, TX_TYPE tx_type,
|
||||
FWD_TXFM_OPT fwd_txfm_opt) {
|
||||
(void)fwd_txfm_opt;
|
||||
av1_fht8x32(src_diff, coeff, diff_stride, tx_type);
|
||||
int diff_stride, TxfmParam *txfm_param) {
|
||||
#if CONFIG_LGT
|
||||
av1_fht8x32_c(src_diff, coeff, diff_stride, txfm_param);
|
||||
#else
|
||||
av1_fht8x32(src_diff, coeff, diff_stride, txfm_param);
|
||||
#endif
|
||||
}
|
||||
#endif // CONFIG_EXT_TX && CONFIG_RECT_TX && CONFIG_RECT_TX_EXT
|
||||
#endif
|
||||
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
#if CONFIG_CHROMA_2X2
|
||||
static void highbd_fwd_txfm_2x2(const int16_t *src_diff, tran_low_t *coeff,
|
||||
int diff_stride, TX_TYPE tx_type, int lossless,
|
||||
const int bd) {
|
||||
int diff_stride, TxfmParam *txfm_param) {
|
||||
tran_high_t a1 = src_diff[0];
|
||||
tran_high_t b1 = src_diff[1];
|
||||
tran_high_t c1 = src_diff[diff_stride];
|
||||
|
|
@ -187,27 +204,27 @@ static void highbd_fwd_txfm_2x2(const int16_t *src_diff, tran_low_t *coeff,
|
|||
coeff[2] = (tran_low_t)(4 * c1);
|
||||
coeff[3] = (tran_low_t)(4 * d1);
|
||||
|
||||
(void)tx_type;
|
||||
(void)lossless;
|
||||
(void)bd;
|
||||
(void)txfm_param;
|
||||
}
|
||||
#endif
|
||||
|
||||
static void highbd_fwd_txfm_4x4(const int16_t *src_diff, tran_low_t *coeff,
|
||||
int diff_stride, TX_TYPE tx_type, int lossless,
|
||||
const int bd) {
|
||||
if (lossless) {
|
||||
int diff_stride, TxfmParam *txfm_param) {
|
||||
int32_t *dst_coeff = (int32_t *)coeff;
|
||||
const int tx_type = txfm_param->tx_type;
|
||||
const int bd = txfm_param->bd;
|
||||
if (txfm_param->lossless) {
|
||||
assert(tx_type == DCT_DCT);
|
||||
av1_highbd_fwht4x4(src_diff, coeff, diff_stride);
|
||||
return;
|
||||
}
|
||||
|
||||
switch (tx_type) {
|
||||
case DCT_DCT:
|
||||
case ADST_DCT:
|
||||
case DCT_ADST:
|
||||
case ADST_ADST:
|
||||
av1_fwd_txfm2d_4x4(src_diff, coeff, diff_stride, tx_type, bd);
|
||||
// fallthrough intended
|
||||
av1_fwd_txfm2d_4x4(src_diff, dst_coeff, diff_stride, tx_type, bd);
|
||||
break;
|
||||
#if CONFIG_EXT_TX
|
||||
case FLIPADST_DCT:
|
||||
|
|
@ -215,80 +232,79 @@ static void highbd_fwd_txfm_4x4(const int16_t *src_diff, tran_low_t *coeff,
|
|||
case FLIPADST_FLIPADST:
|
||||
case ADST_FLIPADST:
|
||||
case FLIPADST_ADST:
|
||||
av1_fwd_txfm2d_4x4(src_diff, coeff, diff_stride, tx_type, bd);
|
||||
// fallthrough intended
|
||||
av1_fwd_txfm2d_4x4(src_diff, dst_coeff, diff_stride, tx_type, bd);
|
||||
break;
|
||||
// use the c version for anything including identity for now
|
||||
case V_DCT:
|
||||
case H_DCT:
|
||||
case V_ADST:
|
||||
case H_ADST:
|
||||
case V_FLIPADST:
|
||||
case H_FLIPADST:
|
||||
av1_highbd_fht4x4_c(src_diff, coeff, diff_stride, tx_type);
|
||||
case IDTX:
|
||||
// fallthrough intended
|
||||
av1_fwd_txfm2d_4x4_c(src_diff, dst_coeff, diff_stride, tx_type, bd);
|
||||
break;
|
||||
case IDTX: av1_fwd_idtx_c(src_diff, coeff, diff_stride, 4, tx_type); break;
|
||||
#endif // CONFIG_EXT_TX
|
||||
default: assert(0);
|
||||
}
|
||||
}
|
||||
|
||||
static void highbd_fwd_txfm_4x8(const int16_t *src_diff, tran_low_t *coeff,
|
||||
int diff_stride, TX_TYPE tx_type,
|
||||
FWD_TXFM_OPT fwd_txfm_opt, const int bd) {
|
||||
(void)fwd_txfm_opt;
|
||||
(void)bd;
|
||||
av1_highbd_fht4x8(src_diff, coeff, diff_stride, tx_type);
|
||||
int diff_stride, TxfmParam *txfm_param) {
|
||||
int32_t *dst_coeff = (int32_t *)coeff;
|
||||
av1_fwd_txfm2d_4x8_c(src_diff, dst_coeff, diff_stride, txfm_param->tx_type,
|
||||
txfm_param->bd);
|
||||
}
|
||||
|
||||
static void highbd_fwd_txfm_8x4(const int16_t *src_diff, tran_low_t *coeff,
|
||||
int diff_stride, TX_TYPE tx_type,
|
||||
FWD_TXFM_OPT fwd_txfm_opt, const int bd) {
|
||||
(void)fwd_txfm_opt;
|
||||
(void)bd;
|
||||
av1_highbd_fht8x4(src_diff, coeff, diff_stride, tx_type);
|
||||
int diff_stride, TxfmParam *txfm_param) {
|
||||
int32_t *dst_coeff = (int32_t *)coeff;
|
||||
av1_fwd_txfm2d_8x4_c(src_diff, dst_coeff, diff_stride, txfm_param->tx_type,
|
||||
txfm_param->bd);
|
||||
}
|
||||
|
||||
static void highbd_fwd_txfm_8x16(const int16_t *src_diff, tran_low_t *coeff,
|
||||
int diff_stride, TX_TYPE tx_type,
|
||||
FWD_TXFM_OPT fwd_txfm_opt, const int bd) {
|
||||
(void)fwd_txfm_opt;
|
||||
(void)bd;
|
||||
av1_highbd_fht8x16(src_diff, coeff, diff_stride, tx_type);
|
||||
int diff_stride, TxfmParam *txfm_param) {
|
||||
int32_t *dst_coeff = (int32_t *)coeff;
|
||||
av1_fwd_txfm2d_8x16_c(src_diff, dst_coeff, diff_stride, txfm_param->tx_type,
|
||||
txfm_param->bd);
|
||||
}
|
||||
|
||||
static void highbd_fwd_txfm_16x8(const int16_t *src_diff, tran_low_t *coeff,
|
||||
int diff_stride, TX_TYPE tx_type,
|
||||
FWD_TXFM_OPT fwd_txfm_opt, const int bd) {
|
||||
(void)fwd_txfm_opt;
|
||||
(void)bd;
|
||||
av1_highbd_fht16x8(src_diff, coeff, diff_stride, tx_type);
|
||||
int diff_stride, TxfmParam *txfm_param) {
|
||||
int32_t *dst_coeff = (int32_t *)coeff;
|
||||
av1_fwd_txfm2d_16x8_c(src_diff, dst_coeff, diff_stride, txfm_param->tx_type,
|
||||
txfm_param->bd);
|
||||
}
|
||||
|
||||
static void highbd_fwd_txfm_16x32(const int16_t *src_diff, tran_low_t *coeff,
|
||||
int diff_stride, TX_TYPE tx_type,
|
||||
FWD_TXFM_OPT fwd_txfm_opt, const int bd) {
|
||||
(void)fwd_txfm_opt;
|
||||
(void)bd;
|
||||
av1_highbd_fht16x32(src_diff, coeff, diff_stride, tx_type);
|
||||
int diff_stride, TxfmParam *txfm_param) {
|
||||
int32_t *dst_coeff = (int32_t *)coeff;
|
||||
av1_fwd_txfm2d_16x32_c(src_diff, dst_coeff, diff_stride, txfm_param->tx_type,
|
||||
txfm_param->bd);
|
||||
}
|
||||
|
||||
static void highbd_fwd_txfm_32x16(const int16_t *src_diff, tran_low_t *coeff,
|
||||
int diff_stride, TX_TYPE tx_type,
|
||||
FWD_TXFM_OPT fwd_txfm_opt, const int bd) {
|
||||
(void)fwd_txfm_opt;
|
||||
(void)bd;
|
||||
av1_highbd_fht32x16(src_diff, coeff, diff_stride, tx_type);
|
||||
int diff_stride, TxfmParam *txfm_param) {
|
||||
int32_t *dst_coeff = (int32_t *)coeff;
|
||||
av1_fwd_txfm2d_32x16_c(src_diff, dst_coeff, diff_stride, txfm_param->tx_type,
|
||||
txfm_param->bd);
|
||||
}
|
||||
|
||||
static void highbd_fwd_txfm_8x8(const int16_t *src_diff, tran_low_t *coeff,
|
||||
int diff_stride, TX_TYPE tx_type,
|
||||
FWD_TXFM_OPT fwd_txfm_opt, const int bd) {
|
||||
(void)fwd_txfm_opt;
|
||||
int diff_stride, TxfmParam *txfm_param) {
|
||||
int32_t *dst_coeff = (int32_t *)coeff;
|
||||
const int tx_type = txfm_param->tx_type;
|
||||
const int bd = txfm_param->bd;
|
||||
switch (tx_type) {
|
||||
case DCT_DCT:
|
||||
case ADST_DCT:
|
||||
case DCT_ADST:
|
||||
case ADST_ADST:
|
||||
av1_fwd_txfm2d_8x8(src_diff, coeff, diff_stride, tx_type, bd);
|
||||
// fallthrough intended
|
||||
av1_fwd_txfm2d_8x8(src_diff, dst_coeff, diff_stride, tx_type, bd);
|
||||
break;
|
||||
#if CONFIG_EXT_TX
|
||||
case FLIPADST_DCT:
|
||||
|
|
@ -296,33 +312,37 @@ static void highbd_fwd_txfm_8x8(const int16_t *src_diff, tran_low_t *coeff,
|
|||
case FLIPADST_FLIPADST:
|
||||
case ADST_FLIPADST:
|
||||
case FLIPADST_ADST:
|
||||
av1_fwd_txfm2d_8x8(src_diff, coeff, diff_stride, tx_type, bd);
|
||||
// fallthrough intended
|
||||
av1_fwd_txfm2d_8x8(src_diff, dst_coeff, diff_stride, tx_type, bd);
|
||||
break;
|
||||
// use the c version for anything including identity for now
|
||||
case V_DCT:
|
||||
case H_DCT:
|
||||
case V_ADST:
|
||||
case H_ADST:
|
||||
case V_FLIPADST:
|
||||
case H_FLIPADST:
|
||||
// Use C version since DST exists only in C
|
||||
av1_highbd_fht8x8_c(src_diff, coeff, diff_stride, tx_type);
|
||||
case IDTX:
|
||||
// fallthrough intended
|
||||
av1_fwd_txfm2d_8x8_c(src_diff, dst_coeff, diff_stride, tx_type, bd);
|
||||
break;
|
||||
case IDTX: av1_fwd_idtx_c(src_diff, coeff, diff_stride, 8, tx_type); break;
|
||||
#endif // CONFIG_EXT_TX
|
||||
default: assert(0);
|
||||
}
|
||||
}
|
||||
|
||||
static void highbd_fwd_txfm_16x16(const int16_t *src_diff, tran_low_t *coeff,
|
||||
int diff_stride, TX_TYPE tx_type,
|
||||
FWD_TXFM_OPT fwd_txfm_opt, const int bd) {
|
||||
(void)fwd_txfm_opt;
|
||||
int diff_stride, TxfmParam *txfm_param) {
|
||||
int32_t *dst_coeff = (int32_t *)coeff;
|
||||
const int tx_type = txfm_param->tx_type;
|
||||
const int bd = txfm_param->bd;
|
||||
switch (tx_type) {
|
||||
case DCT_DCT:
|
||||
case ADST_DCT:
|
||||
case DCT_ADST:
|
||||
case ADST_ADST:
|
||||
av1_fwd_txfm2d_16x16(src_diff, coeff, diff_stride, tx_type, bd);
|
||||
// fallthrough intended
|
||||
av1_fwd_txfm2d_16x16(src_diff, dst_coeff, diff_stride, tx_type, bd);
|
||||
break;
|
||||
#if CONFIG_EXT_TX
|
||||
case FLIPADST_DCT:
|
||||
|
|
@ -330,63 +350,72 @@ static void highbd_fwd_txfm_16x16(const int16_t *src_diff, tran_low_t *coeff,
|
|||
case FLIPADST_FLIPADST:
|
||||
case ADST_FLIPADST:
|
||||
case FLIPADST_ADST:
|
||||
av1_fwd_txfm2d_16x16(src_diff, coeff, diff_stride, tx_type, bd);
|
||||
// fallthrough intended
|
||||
av1_fwd_txfm2d_16x16(src_diff, dst_coeff, diff_stride, tx_type, bd);
|
||||
break;
|
||||
// use the c version for anything including identity for now
|
||||
case V_DCT:
|
||||
case H_DCT:
|
||||
case V_ADST:
|
||||
case H_ADST:
|
||||
case V_FLIPADST:
|
||||
case H_FLIPADST:
|
||||
// Use C version since DST exists only in C
|
||||
av1_highbd_fht16x16_c(src_diff, coeff, diff_stride, tx_type);
|
||||
case IDTX:
|
||||
// fallthrough intended
|
||||
av1_fwd_txfm2d_16x16_c(src_diff, dst_coeff, diff_stride, tx_type, bd);
|
||||
break;
|
||||
case IDTX: av1_fwd_idtx_c(src_diff, coeff, diff_stride, 16, tx_type); break;
|
||||
#endif // CONFIG_EXT_TX
|
||||
default: assert(0);
|
||||
}
|
||||
}
|
||||
|
||||
static void highbd_fwd_txfm_32x32(const int16_t *src_diff, tran_low_t *coeff,
|
||||
int diff_stride, TX_TYPE tx_type,
|
||||
FWD_TXFM_OPT fwd_txfm_opt, const int bd) {
|
||||
(void)fwd_txfm_opt;
|
||||
int diff_stride, TxfmParam *txfm_param) {
|
||||
int32_t *dst_coeff = (int32_t *)coeff;
|
||||
const int tx_type = txfm_param->tx_type;
|
||||
const int bd = txfm_param->bd;
|
||||
switch (tx_type) {
|
||||
case DCT_DCT:
|
||||
av1_fwd_txfm2d_32x32(src_diff, coeff, diff_stride, tx_type, bd);
|
||||
break;
|
||||
#if CONFIG_EXT_TX
|
||||
case ADST_DCT:
|
||||
case DCT_ADST:
|
||||
case ADST_ADST:
|
||||
// fallthrough intended
|
||||
av1_fwd_txfm2d_32x32(src_diff, dst_coeff, diff_stride, tx_type, bd);
|
||||
break;
|
||||
#if CONFIG_EXT_TX
|
||||
case FLIPADST_DCT:
|
||||
case DCT_FLIPADST:
|
||||
case FLIPADST_FLIPADST:
|
||||
case ADST_FLIPADST:
|
||||
case FLIPADST_ADST:
|
||||
// fallthrough intended
|
||||
av1_fwd_txfm2d_32x32(src_diff, dst_coeff, diff_stride, tx_type, bd);
|
||||
break;
|
||||
// use the c version for anything including identity for now
|
||||
case V_DCT:
|
||||
case H_DCT:
|
||||
case V_ADST:
|
||||
case H_ADST:
|
||||
case V_FLIPADST:
|
||||
case H_FLIPADST:
|
||||
av1_highbd_fht32x32_c(src_diff, coeff, diff_stride, tx_type);
|
||||
case IDTX:
|
||||
// fallthrough intended
|
||||
av1_fwd_txfm2d_32x32_c(src_diff, dst_coeff, diff_stride, tx_type, bd);
|
||||
break;
|
||||
case IDTX: av1_fwd_idtx_c(src_diff, coeff, diff_stride, 32, tx_type); break;
|
||||
#endif // CONFIG_EXT_TX
|
||||
default: assert(0); break;
|
||||
default: assert(0);
|
||||
}
|
||||
}
|
||||
|
||||
#if CONFIG_TX64X64
|
||||
static void highbd_fwd_txfm_64x64(const int16_t *src_diff, tran_low_t *coeff,
|
||||
int diff_stride, TX_TYPE tx_type,
|
||||
FWD_TXFM_OPT fwd_txfm_opt, const int bd) {
|
||||
(void)fwd_txfm_opt;
|
||||
(void)bd;
|
||||
int diff_stride, TxfmParam *txfm_param) {
|
||||
int32_t *dst_coeff = (int32_t *)coeff;
|
||||
const int tx_type = txfm_param->tx_type;
|
||||
const int bd = txfm_param->bd;
|
||||
switch (tx_type) {
|
||||
case DCT_DCT:
|
||||
av1_highbd_fht64x64(src_diff, coeff, diff_stride, tx_type);
|
||||
av1_fwd_txfm2d_64x64(src_diff, dst_coeff, diff_stride, tx_type, bd);
|
||||
break;
|
||||
#if CONFIG_EXT_TX
|
||||
case ADST_DCT:
|
||||
|
|
@ -403,141 +432,119 @@ static void highbd_fwd_txfm_64x64(const int16_t *src_diff, tran_low_t *coeff,
|
|||
case H_ADST:
|
||||
case V_FLIPADST:
|
||||
case H_FLIPADST:
|
||||
av1_highbd_fht64x64(src_diff, coeff, diff_stride, tx_type);
|
||||
// TODO(sarahparker)
|
||||
// I've deleted the 64x64 implementations that existed in lieu
|
||||
// of adst, flipadst and identity for simplicity but will bring back
|
||||
// in a later change. This shouldn't impact performance since
|
||||
// DCT_DCT is the only extended type currently allowed for 64x64,
|
||||
// as dictated by get_ext_tx_set_type in blockd.h.
|
||||
av1_fwd_txfm2d_64x64_c(src_diff, dst_coeff, diff_stride, DCT_DCT, bd);
|
||||
break;
|
||||
case IDTX:
|
||||
av1_fwd_idtx_c(src_diff, dst_coeff, diff_stride, 64, tx_type);
|
||||
break;
|
||||
case IDTX: av1_fwd_idtx_c(src_diff, coeff, diff_stride, 64, tx_type); break;
|
||||
#endif // CONFIG_EXT_TX
|
||||
default: assert(0); break;
|
||||
}
|
||||
}
|
||||
#endif // CONFIG_TX64X64
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
|
||||
void av1_fwd_txfm(const int16_t *src_diff, tran_low_t *coeff, int diff_stride,
|
||||
FWD_TXFM_PARAM *fwd_txfm_param) {
|
||||
const int fwd_txfm_opt = FWD_TXFM_OPT_NORMAL;
|
||||
const TX_TYPE tx_type = fwd_txfm_param->tx_type;
|
||||
const TX_SIZE tx_size = fwd_txfm_param->tx_size;
|
||||
const int lossless = fwd_txfm_param->lossless;
|
||||
TxfmParam *txfm_param) {
|
||||
const TX_SIZE tx_size = txfm_param->tx_size;
|
||||
switch (tx_size) {
|
||||
#if CONFIG_TX64X64
|
||||
case TX_64X64:
|
||||
fwd_txfm_64x64(src_diff, coeff, diff_stride, tx_type, fwd_txfm_opt);
|
||||
fwd_txfm_64x64(src_diff, coeff, diff_stride, txfm_param);
|
||||
break;
|
||||
#endif // CONFIG_TX64X64
|
||||
case TX_32X32:
|
||||
fwd_txfm_32x32(src_diff, coeff, diff_stride, tx_type, fwd_txfm_opt);
|
||||
fwd_txfm_32x32(src_diff, coeff, diff_stride, txfm_param);
|
||||
break;
|
||||
case TX_16X16:
|
||||
fwd_txfm_16x16(src_diff, coeff, diff_stride, tx_type, fwd_txfm_opt);
|
||||
break;
|
||||
case TX_8X8:
|
||||
fwd_txfm_8x8(src_diff, coeff, diff_stride, tx_type, fwd_txfm_opt);
|
||||
break;
|
||||
case TX_4X8:
|
||||
fwd_txfm_4x8(src_diff, coeff, diff_stride, tx_type, fwd_txfm_opt);
|
||||
break;
|
||||
case TX_8X4:
|
||||
fwd_txfm_8x4(src_diff, coeff, diff_stride, tx_type, fwd_txfm_opt);
|
||||
fwd_txfm_16x16(src_diff, coeff, diff_stride, txfm_param);
|
||||
break;
|
||||
case TX_8X8: fwd_txfm_8x8(src_diff, coeff, diff_stride, txfm_param); break;
|
||||
case TX_4X8: fwd_txfm_4x8(src_diff, coeff, diff_stride, txfm_param); break;
|
||||
case TX_8X4: fwd_txfm_8x4(src_diff, coeff, diff_stride, txfm_param); break;
|
||||
case TX_8X16:
|
||||
fwd_txfm_8x16(src_diff, coeff, diff_stride, tx_type, fwd_txfm_opt);
|
||||
fwd_txfm_8x16(src_diff, coeff, diff_stride, txfm_param);
|
||||
break;
|
||||
case TX_16X8:
|
||||
fwd_txfm_16x8(src_diff, coeff, diff_stride, tx_type, fwd_txfm_opt);
|
||||
fwd_txfm_16x8(src_diff, coeff, diff_stride, txfm_param);
|
||||
break;
|
||||
case TX_16X32:
|
||||
fwd_txfm_16x32(src_diff, coeff, diff_stride, tx_type, fwd_txfm_opt);
|
||||
fwd_txfm_16x32(src_diff, coeff, diff_stride, txfm_param);
|
||||
break;
|
||||
case TX_32X16:
|
||||
fwd_txfm_32x16(src_diff, coeff, diff_stride, tx_type, fwd_txfm_opt);
|
||||
break;
|
||||
case TX_4X4:
|
||||
fwd_txfm_4x4(src_diff, coeff, diff_stride, tx_type, lossless);
|
||||
fwd_txfm_32x16(src_diff, coeff, diff_stride, txfm_param);
|
||||
break;
|
||||
case TX_4X4: fwd_txfm_4x4(src_diff, coeff, diff_stride, txfm_param); break;
|
||||
#if CONFIG_CHROMA_2X2
|
||||
case TX_2X2:
|
||||
fwd_txfm_2x2(src_diff, coeff, diff_stride, tx_type, lossless);
|
||||
break;
|
||||
case TX_2X2: fwd_txfm_2x2(src_diff, coeff, diff_stride, txfm_param); break;
|
||||
#endif
|
||||
#if CONFIG_EXT_TX && CONFIG_RECT_TX && CONFIG_RECT_TX_EXT
|
||||
#if CONFIG_RECT_TX_EXT && (CONFIG_EXT_TX || CONFIG_VAR_TX)
|
||||
case TX_4X16:
|
||||
fwd_txfm_4x16(src_diff, coeff, diff_stride, tx_type, fwd_txfm_opt);
|
||||
fwd_txfm_4x16(src_diff, coeff, diff_stride, txfm_param);
|
||||
break;
|
||||
case TX_16X4:
|
||||
fwd_txfm_16x4(src_diff, coeff, diff_stride, tx_type, fwd_txfm_opt);
|
||||
fwd_txfm_16x4(src_diff, coeff, diff_stride, txfm_param);
|
||||
break;
|
||||
case TX_8X32:
|
||||
fwd_txfm_8x32(src_diff, coeff, diff_stride, tx_type, fwd_txfm_opt);
|
||||
fwd_txfm_8x32(src_diff, coeff, diff_stride, txfm_param);
|
||||
break;
|
||||
case TX_32X8:
|
||||
fwd_txfm_32x8(src_diff, coeff, diff_stride, tx_type, fwd_txfm_opt);
|
||||
fwd_txfm_32x8(src_diff, coeff, diff_stride, txfm_param);
|
||||
break;
|
||||
#endif // CONFIG_EXT_TX && CONFIG_RECT_TX && CONFIG_RECT_TX_EXT
|
||||
#endif
|
||||
default: assert(0); break;
|
||||
}
|
||||
}
|
||||
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
void av1_highbd_fwd_txfm(const int16_t *src_diff, tran_low_t *coeff,
|
||||
int diff_stride, FWD_TXFM_PARAM *fwd_txfm_param) {
|
||||
const int fwd_txfm_opt = FWD_TXFM_OPT_NORMAL;
|
||||
const TX_TYPE tx_type = fwd_txfm_param->tx_type;
|
||||
const TX_SIZE tx_size = fwd_txfm_param->tx_size;
|
||||
const int lossless = fwd_txfm_param->lossless;
|
||||
const int bd = fwd_txfm_param->bd;
|
||||
int diff_stride, TxfmParam *txfm_param) {
|
||||
const TX_SIZE tx_size = txfm_param->tx_size;
|
||||
switch (tx_size) {
|
||||
#if CONFIG_TX64X64
|
||||
case TX_64X64:
|
||||
highbd_fwd_txfm_64x64(src_diff, coeff, diff_stride, tx_type, fwd_txfm_opt,
|
||||
bd);
|
||||
highbd_fwd_txfm_64x64(src_diff, coeff, diff_stride, txfm_param);
|
||||
break;
|
||||
#endif // CONFIG_TX64X64
|
||||
case TX_32X32:
|
||||
highbd_fwd_txfm_32x32(src_diff, coeff, diff_stride, tx_type, fwd_txfm_opt,
|
||||
bd);
|
||||
highbd_fwd_txfm_32x32(src_diff, coeff, diff_stride, txfm_param);
|
||||
break;
|
||||
case TX_16X16:
|
||||
highbd_fwd_txfm_16x16(src_diff, coeff, diff_stride, tx_type, fwd_txfm_opt,
|
||||
bd);
|
||||
highbd_fwd_txfm_16x16(src_diff, coeff, diff_stride, txfm_param);
|
||||
break;
|
||||
case TX_8X8:
|
||||
highbd_fwd_txfm_8x8(src_diff, coeff, diff_stride, tx_type, fwd_txfm_opt,
|
||||
bd);
|
||||
highbd_fwd_txfm_8x8(src_diff, coeff, diff_stride, txfm_param);
|
||||
break;
|
||||
case TX_4X8:
|
||||
highbd_fwd_txfm_4x8(src_diff, coeff, diff_stride, tx_type, fwd_txfm_opt,
|
||||
bd);
|
||||
highbd_fwd_txfm_4x8(src_diff, coeff, diff_stride, txfm_param);
|
||||
break;
|
||||
case TX_8X4:
|
||||
highbd_fwd_txfm_8x4(src_diff, coeff, diff_stride, tx_type, fwd_txfm_opt,
|
||||
bd);
|
||||
highbd_fwd_txfm_8x4(src_diff, coeff, diff_stride, txfm_param);
|
||||
break;
|
||||
case TX_8X16:
|
||||
highbd_fwd_txfm_8x16(src_diff, coeff, diff_stride, tx_type, fwd_txfm_opt,
|
||||
bd);
|
||||
highbd_fwd_txfm_8x16(src_diff, coeff, diff_stride, txfm_param);
|
||||
break;
|
||||
case TX_16X8:
|
||||
highbd_fwd_txfm_16x8(src_diff, coeff, diff_stride, tx_type, fwd_txfm_opt,
|
||||
bd);
|
||||
highbd_fwd_txfm_16x8(src_diff, coeff, diff_stride, txfm_param);
|
||||
break;
|
||||
case TX_16X32:
|
||||
highbd_fwd_txfm_16x32(src_diff, coeff, diff_stride, tx_type, fwd_txfm_opt,
|
||||
bd);
|
||||
highbd_fwd_txfm_16x32(src_diff, coeff, diff_stride, txfm_param);
|
||||
break;
|
||||
case TX_32X16:
|
||||
highbd_fwd_txfm_32x16(src_diff, coeff, diff_stride, tx_type, fwd_txfm_opt,
|
||||
bd);
|
||||
highbd_fwd_txfm_32x16(src_diff, coeff, diff_stride, txfm_param);
|
||||
break;
|
||||
case TX_4X4:
|
||||
highbd_fwd_txfm_4x4(src_diff, coeff, diff_stride, tx_type, lossless, bd);
|
||||
highbd_fwd_txfm_4x4(src_diff, coeff, diff_stride, txfm_param);
|
||||
break;
|
||||
#if CONFIG_CHROMA_2X2
|
||||
case TX_2X2:
|
||||
highbd_fwd_txfm_2x2(src_diff, coeff, diff_stride, tx_type, lossless, bd);
|
||||
highbd_fwd_txfm_2x2(src_diff, coeff, diff_stride, txfm_param);
|
||||
break;
|
||||
#endif
|
||||
default: assert(0); break;
|
||||
}
|
||||
}
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
|
|
|
|||
17
third_party/aom/av1/encoder/hybrid_fwd_txfm.h
vendored
17
third_party/aom/av1/encoder/hybrid_fwd_txfm.h
vendored
|
|
@ -14,28 +14,15 @@
|
|||
|
||||
#include "./aom_config.h"
|
||||
|
||||
typedef enum FWD_TXFM_OPT { FWD_TXFM_OPT_NORMAL } FWD_TXFM_OPT;
|
||||
|
||||
typedef struct FWD_TXFM_PARAM {
|
||||
TX_TYPE tx_type;
|
||||
TX_SIZE tx_size;
|
||||
int lossless;
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
int bd;
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
} FWD_TXFM_PARAM;
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
|
||||
void av1_fwd_txfm(const int16_t *src_diff, tran_low_t *coeff, int diff_stride,
|
||||
FWD_TXFM_PARAM *fwd_txfm_param);
|
||||
TxfmParam *txfm_param);
|
||||
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
void av1_highbd_fwd_txfm(const int16_t *src_diff, tran_low_t *coeff,
|
||||
int diff_stride, FWD_TXFM_PARAM *fwd_txfm_param);
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
int diff_stride, TxfmParam *txfm_param);
|
||||
|
||||
#ifdef __cplusplus
|
||||
} // extern "C"
|
||||
|
|
|
|||
245
third_party/aom/av1/encoder/mcomp.c
vendored
245
third_party/aom/av1/encoder/mcomp.c
vendored
|
|
@ -228,49 +228,45 @@ static INLINE const uint8_t *pre(const uint8_t *buf, int stride, int r, int c) {
|
|||
|
||||
#define CHECK_BETTER0(v, r, c) CHECK_BETTER(v, r, c)
|
||||
|
||||
static INLINE const uint8_t *upre(const uint8_t *buf, int stride, int r,
|
||||
int c) {
|
||||
return &buf[(r)*stride + (c)];
|
||||
}
|
||||
|
||||
/* checks if (r, c) has better score than previous best */
|
||||
#if CONFIG_EXT_INTER
|
||||
#define CHECK_BETTER1(v, r, c) \
|
||||
if (c >= minc && c <= maxc && r >= minr && r <= maxr) { \
|
||||
MV this_mv = { r, c }; \
|
||||
thismse = upsampled_pref_error( \
|
||||
xd, vfp, src_address, src_stride, upre(y, y_stride, r, c), y_stride, \
|
||||
second_pred, mask, mask_stride, invert_mask, w, h, &sse); \
|
||||
v = mv_err_cost(&this_mv, ref_mv, mvjcost, mvcost, error_per_bit); \
|
||||
v += thismse; \
|
||||
if (v < besterr) { \
|
||||
besterr = v; \
|
||||
br = r; \
|
||||
bc = c; \
|
||||
*distortion = thismse; \
|
||||
*sse1 = sse; \
|
||||
} \
|
||||
} else { \
|
||||
v = INT_MAX; \
|
||||
#define CHECK_BETTER1(v, r, c) \
|
||||
if (c >= minc && c <= maxc && r >= minr && r <= maxr) { \
|
||||
MV this_mv = { r, c }; \
|
||||
thismse = upsampled_pref_error(xd, vfp, src_address, src_stride, \
|
||||
pre(y, y_stride, r, c), y_stride, sp(c), \
|
||||
sp(r), second_pred, mask, mask_stride, \
|
||||
invert_mask, w, h, &sse); \
|
||||
v = mv_err_cost(&this_mv, ref_mv, mvjcost, mvcost, error_per_bit); \
|
||||
v += thismse; \
|
||||
if (v < besterr) { \
|
||||
besterr = v; \
|
||||
br = r; \
|
||||
bc = c; \
|
||||
*distortion = thismse; \
|
||||
*sse1 = sse; \
|
||||
} \
|
||||
} else { \
|
||||
v = INT_MAX; \
|
||||
}
|
||||
#else
|
||||
#define CHECK_BETTER1(v, r, c) \
|
||||
if (c >= minc && c <= maxc && r >= minr && r <= maxr) { \
|
||||
MV this_mv = { r, c }; \
|
||||
thismse = upsampled_pref_error(xd, vfp, src_address, src_stride, \
|
||||
upre(y, y_stride, r, c), y_stride, \
|
||||
second_pred, w, h, &sse); \
|
||||
v = mv_err_cost(&this_mv, ref_mv, mvjcost, mvcost, error_per_bit); \
|
||||
v += thismse; \
|
||||
if (v < besterr) { \
|
||||
besterr = v; \
|
||||
br = r; \
|
||||
bc = c; \
|
||||
*distortion = thismse; \
|
||||
*sse1 = sse; \
|
||||
} \
|
||||
} else { \
|
||||
v = INT_MAX; \
|
||||
#define CHECK_BETTER1(v, r, c) \
|
||||
if (c >= minc && c <= maxc && r >= minr && r <= maxr) { \
|
||||
MV this_mv = { r, c }; \
|
||||
thismse = upsampled_pref_error(xd, vfp, src_address, src_stride, \
|
||||
pre(y, y_stride, r, c), y_stride, sp(c), \
|
||||
sp(r), second_pred, w, h, &sse); \
|
||||
v = mv_err_cost(&this_mv, ref_mv, mvjcost, mvcost, error_per_bit); \
|
||||
v += thismse; \
|
||||
if (v < besterr) { \
|
||||
besterr = v; \
|
||||
br = r; \
|
||||
bc = c; \
|
||||
*distortion = thismse; \
|
||||
*sse1 = sse; \
|
||||
} \
|
||||
} else { \
|
||||
v = INT_MAX; \
|
||||
}
|
||||
#endif // CONFIG_EXT_INTER
|
||||
|
||||
|
|
@ -700,16 +696,14 @@ static const MV search_step_table[12] = {
|
|||
};
|
||||
/* clang-format on */
|
||||
|
||||
static int upsampled_pref_error(const MACROBLOCKD *xd,
|
||||
const aom_variance_fn_ptr_t *vfp,
|
||||
const uint8_t *const src, const int src_stride,
|
||||
const uint8_t *const y, int y_stride,
|
||||
const uint8_t *second_pred,
|
||||
static int upsampled_pref_error(
|
||||
const MACROBLOCKD *xd, const aom_variance_fn_ptr_t *vfp,
|
||||
const uint8_t *const src, const int src_stride, const uint8_t *const y,
|
||||
int y_stride, int subpel_x_q3, int subpel_y_q3, const uint8_t *second_pred,
|
||||
#if CONFIG_EXT_INTER
|
||||
const uint8_t *mask, int mask_stride,
|
||||
int invert_mask,
|
||||
const uint8_t *mask, int mask_stride, int invert_mask,
|
||||
#endif
|
||||
int w, int h, unsigned int *sse) {
|
||||
int w, int h, unsigned int *sse) {
|
||||
unsigned int besterr;
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
if (xd->cur_buf->flags & YV12_FLAG_HIGHBITDEPTH) {
|
||||
|
|
@ -717,15 +711,17 @@ static int upsampled_pref_error(const MACROBLOCKD *xd,
|
|||
if (second_pred != NULL) {
|
||||
#if CONFIG_EXT_INTER
|
||||
if (mask)
|
||||
aom_highbd_comp_mask_upsampled_pred(pred16, second_pred, w, h, y,
|
||||
y_stride, mask, mask_stride,
|
||||
invert_mask);
|
||||
aom_highbd_comp_mask_upsampled_pred(
|
||||
pred16, second_pred, w, h, subpel_x_q3, subpel_y_q3, y, y_stride,
|
||||
mask, mask_stride, invert_mask, xd->bd);
|
||||
else
|
||||
#endif
|
||||
aom_highbd_comp_avg_upsampled_pred(pred16, second_pred, w, h, y,
|
||||
y_stride);
|
||||
aom_highbd_comp_avg_upsampled_pred(pred16, second_pred, w, h,
|
||||
subpel_x_q3, subpel_y_q3, y,
|
||||
y_stride, xd->bd);
|
||||
} else {
|
||||
aom_highbd_upsampled_pred(pred16, w, h, y, y_stride);
|
||||
aom_highbd_upsampled_pred(pred16, w, h, subpel_x_q3, subpel_y_q3, y,
|
||||
y_stride, xd->bd);
|
||||
}
|
||||
|
||||
besterr = vfp->vf(CONVERT_TO_BYTEPTR(pred16), w, src, src_stride, sse);
|
||||
|
|
@ -738,13 +734,15 @@ static int upsampled_pref_error(const MACROBLOCKD *xd,
|
|||
if (second_pred != NULL) {
|
||||
#if CONFIG_EXT_INTER
|
||||
if (mask)
|
||||
aom_comp_mask_upsampled_pred(pred, second_pred, w, h, y, y_stride, mask,
|
||||
aom_comp_mask_upsampled_pred(pred, second_pred, w, h, subpel_x_q3,
|
||||
subpel_y_q3, y, y_stride, mask,
|
||||
mask_stride, invert_mask);
|
||||
else
|
||||
#endif
|
||||
aom_comp_avg_upsampled_pred(pred, second_pred, w, h, y, y_stride);
|
||||
aom_comp_avg_upsampled_pred(pred, second_pred, w, h, subpel_x_q3,
|
||||
subpel_y_q3, y, y_stride);
|
||||
} else {
|
||||
aom_upsampled_pred(pred, w, h, y, y_stride);
|
||||
aom_upsampled_pred(pred, w, h, subpel_x_q3, subpel_y_q3, y, y_stride);
|
||||
}
|
||||
|
||||
besterr = vfp->vf(pred, w, src, src_stride, sse);
|
||||
|
|
@ -764,12 +762,12 @@ static unsigned int upsampled_setup_center_error(
|
|||
#endif
|
||||
int w, int h, int offset, int *mvjcost, int *mvcost[2], unsigned int *sse1,
|
||||
int *distortion) {
|
||||
unsigned int besterr = upsampled_pref_error(xd, vfp, src, src_stride,
|
||||
y + offset, y_stride, second_pred,
|
||||
unsigned int besterr = upsampled_pref_error(
|
||||
xd, vfp, src, src_stride, y + offset, y_stride, 0, 0, second_pred,
|
||||
#if CONFIG_EXT_INTER
|
||||
mask, mask_stride, invert_mask,
|
||||
mask, mask_stride, invert_mask,
|
||||
#endif
|
||||
w, h, sse1);
|
||||
w, h, sse1);
|
||||
*distortion = besterr;
|
||||
besterr += mv_err_cost(bestmv, ref_mv, mvjcost, mvcost, error_per_bit);
|
||||
return besterr;
|
||||
|
|
@ -824,7 +822,7 @@ int av1_find_best_sub_pixel_tree(
|
|||
#if CONFIG_EXT_INTER
|
||||
mask, mask_stride, invert_mask,
|
||||
#endif
|
||||
w, h, (offset * 8), mvjcost, mvcost, sse1, distortion);
|
||||
w, h, offset, mvjcost, mvcost, sse1, distortion);
|
||||
else
|
||||
besterr =
|
||||
setup_center_error(xd, bestmv, ref_mv, error_per_bit, vfp, src_address,
|
||||
|
|
@ -845,17 +843,15 @@ int av1_find_best_sub_pixel_tree(
|
|||
MV this_mv = { tr, tc };
|
||||
|
||||
if (use_upsampled_ref) {
|
||||
const uint8_t *const pre_address = y + tr * y_stride + tc;
|
||||
|
||||
thismse = upsampled_pref_error(xd, vfp, src_address, src_stride,
|
||||
pre_address, y_stride, second_pred,
|
||||
pre(y, y_stride, tr, tc), y_stride,
|
||||
sp(tc), sp(tr), second_pred,
|
||||
#if CONFIG_EXT_INTER
|
||||
mask, mask_stride, invert_mask,
|
||||
#endif
|
||||
w, h, &sse);
|
||||
} else {
|
||||
const uint8_t *const pre_address =
|
||||
y + (tr >> 3) * y_stride + (tc >> 3);
|
||||
const uint8_t *const pre_address = pre(y, y_stride, tr, tc);
|
||||
if (second_pred == NULL)
|
||||
thismse = vfp->svf(pre_address, y_stride, sp(tc), sp(tr),
|
||||
src_address, src_stride, &sse);
|
||||
|
|
@ -894,16 +890,15 @@ int av1_find_best_sub_pixel_tree(
|
|||
MV this_mv = { tr, tc };
|
||||
|
||||
if (use_upsampled_ref) {
|
||||
const uint8_t *const pre_address = y + tr * y_stride + tc;
|
||||
|
||||
thismse = upsampled_pref_error(xd, vfp, src_address, src_stride,
|
||||
pre_address, y_stride, second_pred,
|
||||
pre(y, y_stride, tr, tc), y_stride,
|
||||
sp(tc), sp(tr), second_pred,
|
||||
#if CONFIG_EXT_INTER
|
||||
mask, mask_stride, invert_mask,
|
||||
#endif
|
||||
w, h, &sse);
|
||||
} else {
|
||||
const uint8_t *const pre_address = y + (tr >> 3) * y_stride + (tc >> 3);
|
||||
const uint8_t *const pre_address = pre(y, y_stride, tr, tc);
|
||||
|
||||
if (second_pred == NULL)
|
||||
thismse = vfp->svf(pre_address, y_stride, sp(tc), sp(tr), src_address,
|
||||
|
|
@ -992,9 +987,16 @@ unsigned int av1_compute_motion_cost(const AV1_COMP *cpi, MACROBLOCK *const x,
|
|||
}
|
||||
|
||||
// Refine MV in a small range
|
||||
#if WARPED_MOTION_SORT_SAMPLES
|
||||
unsigned int av1_refine_warped_mv(const AV1_COMP *cpi, MACROBLOCK *const x,
|
||||
BLOCK_SIZE bsize, int mi_row, int mi_col,
|
||||
int *pts0, int *pts_inref0, int *pts_mv0,
|
||||
int total_samples) {
|
||||
#else
|
||||
unsigned int av1_refine_warped_mv(const AV1_COMP *cpi, MACROBLOCK *const x,
|
||||
BLOCK_SIZE bsize, int mi_row, int mi_col,
|
||||
int *pts, int *pts_inref) {
|
||||
#endif // WARPED_MOTION_SORT_SAMPLES
|
||||
const AV1_COMMON *const cm = &cpi->common;
|
||||
MACROBLOCKD *xd = &x->e_mbd;
|
||||
MODE_INFO *mi = xd->mi[0];
|
||||
|
|
@ -1007,6 +1009,9 @@ unsigned int av1_refine_warped_mv(const AV1_COMP *cpi, MACROBLOCK *const x,
|
|||
int16_t *tr = &mbmi->mv[0].as_mv.row;
|
||||
int16_t *tc = &mbmi->mv[0].as_mv.col;
|
||||
WarpedMotionParams best_wm_params = mbmi->wm_params[0];
|
||||
#if WARPED_MOTION_SORT_SAMPLES
|
||||
int best_num_proj_ref = mbmi->num_proj_ref[0];
|
||||
#endif // WARPED_MOTION_SORT_SAMPLES
|
||||
unsigned int bestmse;
|
||||
int minc, maxc, minr, maxr;
|
||||
const int start = cm->allow_high_precision_mv ? 0 : 4;
|
||||
|
|
@ -1033,6 +1038,16 @@ unsigned int av1_refine_warped_mv(const AV1_COMP *cpi, MACROBLOCK *const x,
|
|||
|
||||
if (*tc >= minc && *tc <= maxc && *tr >= minr && *tr <= maxr) {
|
||||
MV this_mv = { *tr, *tc };
|
||||
#if WARPED_MOTION_SORT_SAMPLES
|
||||
int pts[SAMPLES_ARRAY_SIZE], pts_inref[SAMPLES_ARRAY_SIZE];
|
||||
|
||||
memcpy(pts, pts0, total_samples * 2 * sizeof(*pts0));
|
||||
memcpy(pts_inref, pts_inref0, total_samples * 2 * sizeof(*pts_inref0));
|
||||
if (total_samples > 1)
|
||||
mbmi->num_proj_ref[0] =
|
||||
sortSamples(pts_mv0, &this_mv, pts, pts_inref, total_samples);
|
||||
#endif // WARPED_MOTION_SORT_SAMPLES
|
||||
|
||||
if (!find_projection(mbmi->num_proj_ref[0], pts, pts_inref, bsize, *tr,
|
||||
*tc, &mbmi->wm_params[0], mi_row, mi_col)) {
|
||||
thismse =
|
||||
|
|
@ -1041,6 +1056,9 @@ unsigned int av1_refine_warped_mv(const AV1_COMP *cpi, MACROBLOCK *const x,
|
|||
if (thismse < bestmse) {
|
||||
best_idx = idx;
|
||||
best_wm_params = mbmi->wm_params[0];
|
||||
#if WARPED_MOTION_SORT_SAMPLES
|
||||
best_num_proj_ref = mbmi->num_proj_ref[0];
|
||||
#endif // WARPED_MOTION_SORT_SAMPLES
|
||||
bestmse = thismse;
|
||||
}
|
||||
}
|
||||
|
|
@ -1058,7 +1076,9 @@ unsigned int av1_refine_warped_mv(const AV1_COMP *cpi, MACROBLOCK *const x,
|
|||
*tr = br;
|
||||
*tc = bc;
|
||||
mbmi->wm_params[0] = best_wm_params;
|
||||
|
||||
#if WARPED_MOTION_SORT_SAMPLES
|
||||
mbmi->num_proj_ref[0] = best_num_proj_ref;
|
||||
#endif // WARPED_MOTION_SORT_SAMPLES
|
||||
return bestmse;
|
||||
}
|
||||
#endif // CONFIG_WARPED_MOTION
|
||||
|
|
@ -2653,19 +2673,20 @@ int av1_full_pixel_search(const AV1_COMP *cpi, MACROBLOCK *x, BLOCK_SIZE bsize,
|
|||
#define CHECK_BETTER0(v, r, c) CHECK_BETTER(v, r, c)
|
||||
|
||||
#undef CHECK_BETTER1
|
||||
#define CHECK_BETTER1(v, r, c) \
|
||||
if (c >= minc && c <= maxc && r >= minr && r <= maxr) { \
|
||||
thismse = upsampled_obmc_pref_error( \
|
||||
xd, mask, vfp, z, upre(y, y_stride, r, c), y_stride, w, h, &sse); \
|
||||
if ((v = MVC(r, c) + thismse) < besterr) { \
|
||||
besterr = v; \
|
||||
br = r; \
|
||||
bc = c; \
|
||||
*distortion = thismse; \
|
||||
*sse1 = sse; \
|
||||
} \
|
||||
} else { \
|
||||
v = INT_MAX; \
|
||||
#define CHECK_BETTER1(v, r, c) \
|
||||
if (c >= minc && c <= maxc && r >= minr && r <= maxr) { \
|
||||
thismse = \
|
||||
upsampled_obmc_pref_error(xd, mask, vfp, z, pre(y, y_stride, r, c), \
|
||||
y_stride, sp(c), sp(r), w, h, &sse); \
|
||||
if ((v = MVC(r, c) + thismse) < besterr) { \
|
||||
besterr = v; \
|
||||
br = r; \
|
||||
bc = c; \
|
||||
*distortion = thismse; \
|
||||
*sse1 = sse; \
|
||||
} \
|
||||
} else { \
|
||||
v = INT_MAX; \
|
||||
}
|
||||
|
||||
static unsigned int setup_obmc_center_error(
|
||||
|
|
@ -2684,12 +2705,14 @@ static int upsampled_obmc_pref_error(const MACROBLOCKD *xd, const int32_t *mask,
|
|||
const aom_variance_fn_ptr_t *vfp,
|
||||
const int32_t *const wsrc,
|
||||
const uint8_t *const y, int y_stride,
|
||||
int w, int h, unsigned int *sse) {
|
||||
int subpel_x_q3, int subpel_y_q3, int w,
|
||||
int h, unsigned int *sse) {
|
||||
unsigned int besterr;
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
if (xd->cur_buf->flags & YV12_FLAG_HIGHBITDEPTH) {
|
||||
DECLARE_ALIGNED(16, uint16_t, pred16[MAX_SB_SQUARE]);
|
||||
aom_highbd_upsampled_pred(pred16, w, h, y, y_stride);
|
||||
aom_highbd_upsampled_pred(pred16, w, h, subpel_x_q3, subpel_y_q3, y,
|
||||
y_stride, xd->bd);
|
||||
|
||||
besterr = vfp->ovf(CONVERT_TO_BYTEPTR(pred16), w, wsrc, mask, sse);
|
||||
} else {
|
||||
|
|
@ -2698,7 +2721,7 @@ static int upsampled_obmc_pref_error(const MACROBLOCKD *xd, const int32_t *mask,
|
|||
DECLARE_ALIGNED(16, uint8_t, pred[MAX_SB_SQUARE]);
|
||||
(void)xd;
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
aom_upsampled_pred(pred, w, h, y, y_stride);
|
||||
aom_upsampled_pred(pred, w, h, subpel_x_q3, subpel_y_q3, y, y_stride);
|
||||
|
||||
besterr = vfp->ovf(pred, w, wsrc, mask, sse);
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
|
|
@ -2714,18 +2737,17 @@ static unsigned int upsampled_setup_obmc_center_error(
|
|||
int h, int offset, int *mvjcost, int *mvcost[2], unsigned int *sse1,
|
||||
int *distortion) {
|
||||
unsigned int besterr = upsampled_obmc_pref_error(
|
||||
xd, mask, vfp, wsrc, y + offset, y_stride, w, h, sse1);
|
||||
xd, mask, vfp, wsrc, y + offset, y_stride, 0, 0, w, h, sse1);
|
||||
*distortion = besterr;
|
||||
besterr += mv_err_cost(bestmv, ref_mv, mvjcost, mvcost, error_per_bit);
|
||||
return besterr;
|
||||
}
|
||||
|
||||
int av1_find_best_obmc_sub_pixel_tree_up(
|
||||
const AV1_COMP *cpi, MACROBLOCK *x, int mi_row, int mi_col, MV *bestmv,
|
||||
const MV *ref_mv, int allow_hp, int error_per_bit,
|
||||
const aom_variance_fn_ptr_t *vfp, int forced_stop, int iters_per_step,
|
||||
int *mvjcost, int *mvcost[2], int *distortion, unsigned int *sse1,
|
||||
int is_second, int use_upsampled_ref) {
|
||||
MACROBLOCK *x, MV *bestmv, const MV *ref_mv, int allow_hp,
|
||||
int error_per_bit, const aom_variance_fn_ptr_t *vfp, int forced_stop,
|
||||
int iters_per_step, int *mvjcost, int *mvcost[2], int *distortion,
|
||||
unsigned int *sse1, int is_second, int use_upsampled_ref) {
|
||||
const int32_t *wsrc = x->wsrc_buf;
|
||||
const int32_t *mask = x->mask_buf;
|
||||
const int *const z = wsrc;
|
||||
|
|
@ -2756,21 +2778,11 @@ int av1_find_best_obmc_sub_pixel_tree_up(
|
|||
int y_stride;
|
||||
const uint8_t *y;
|
||||
|
||||
const struct buf_2d backup_pred = pd->pre[is_second];
|
||||
int minc, maxc, minr, maxr;
|
||||
|
||||
av1_set_subpel_mv_search_range(&x->mv_limits, &minc, &maxc, &minr, &maxr,
|
||||
ref_mv);
|
||||
|
||||
if (use_upsampled_ref) {
|
||||
int ref = xd->mi[0]->mbmi.ref_frame[is_second];
|
||||
const YV12_BUFFER_CONFIG *upsampled_ref = get_upsampled_ref(cpi, ref);
|
||||
setup_pred_plane(&pd->pre[is_second], mbmi->sb_type,
|
||||
upsampled_ref->y_buffer, upsampled_ref->y_crop_width,
|
||||
upsampled_ref->y_crop_height, upsampled_ref->y_stride,
|
||||
(mi_row << 3), (mi_col << 3), NULL, pd->subsampling_x,
|
||||
pd->subsampling_y);
|
||||
}
|
||||
y = pd->pre[is_second].buf;
|
||||
y_stride = pd->pre[is_second].stride;
|
||||
offset = bestmv->row * y_stride + bestmv->col;
|
||||
|
|
@ -2784,7 +2796,7 @@ int av1_find_best_obmc_sub_pixel_tree_up(
|
|||
if (use_upsampled_ref)
|
||||
besterr = upsampled_setup_obmc_center_error(
|
||||
xd, mask, bestmv, ref_mv, error_per_bit, vfp, z, y, y_stride, w, h,
|
||||
(offset * 8), mvjcost, mvcost, sse1, distortion);
|
||||
offset, mvjcost, mvcost, sse1, distortion);
|
||||
else
|
||||
besterr = setup_obmc_center_error(mask, bestmv, ref_mv, error_per_bit, vfp,
|
||||
z, y, y_stride, offset, mvjcost, mvcost,
|
||||
|
|
@ -2797,15 +2809,13 @@ int av1_find_best_obmc_sub_pixel_tree_up(
|
|||
tc = bc + search_step[idx].col;
|
||||
if (tc >= minc && tc <= maxc && tr >= minr && tr <= maxr) {
|
||||
MV this_mv = { tr, tc };
|
||||
const uint8_t *const pre_address = pre(y, y_stride, tr, tc);
|
||||
|
||||
if (use_upsampled_ref) {
|
||||
const uint8_t *const pre_address = y + tr * y_stride + tc;
|
||||
|
||||
thismse = upsampled_obmc_pref_error(
|
||||
xd, mask, vfp, src_address, pre_address, y_stride, w, h, &sse);
|
||||
thismse =
|
||||
upsampled_obmc_pref_error(xd, mask, vfp, src_address, pre_address,
|
||||
y_stride, sp(tc), sp(tr), w, h, &sse);
|
||||
} else {
|
||||
const uint8_t *const pre_address =
|
||||
y + (tr >> 3) * y_stride + (tc >> 3);
|
||||
thismse = vfp->osvf(pre_address, y_stride, sp(tc), sp(tr),
|
||||
src_address, mask, &sse);
|
||||
}
|
||||
|
|
@ -2833,15 +2843,12 @@ int av1_find_best_obmc_sub_pixel_tree_up(
|
|||
MV this_mv = { tr, tc };
|
||||
|
||||
if (use_upsampled_ref) {
|
||||
const uint8_t *const pre_address = y + tr * y_stride + tc;
|
||||
|
||||
thismse = upsampled_obmc_pref_error(xd, mask, vfp, src_address,
|
||||
pre_address, y_stride, w, h, &sse);
|
||||
pre(y, y_stride, tr, tc), y_stride,
|
||||
sp(tc), sp(tr), w, h, &sse);
|
||||
} else {
|
||||
const uint8_t *const pre_address = y + (tr >> 3) * y_stride + (tc >> 3);
|
||||
|
||||
thismse = vfp->osvf(pre_address, y_stride, sp(tc), sp(tr), src_address,
|
||||
mask, &sse);
|
||||
thismse = vfp->osvf(pre(y, y_stride, tr, tc), y_stride, sp(tc), sp(tr),
|
||||
src_address, mask, &sse);
|
||||
}
|
||||
|
||||
cost_array[4] = thismse + mv_err_cost(&this_mv, ref_mv, mvjcost, mvcost,
|
||||
|
|
@ -2889,10 +2896,6 @@ int av1_find_best_obmc_sub_pixel_tree_up(
|
|||
bestmv->row = br;
|
||||
bestmv->col = bc;
|
||||
|
||||
if (use_upsampled_ref) {
|
||||
pd->pre[is_second] = backup_pred;
|
||||
}
|
||||
|
||||
return besterr;
|
||||
}
|
||||
|
||||
|
|
|
|||
17
third_party/aom/av1/encoder/mcomp.h
vendored
17
third_party/aom/av1/encoder/mcomp.h
vendored
|
|
@ -143,11 +143,10 @@ int av1_obmc_full_pixel_diamond(const struct AV1_COMP *cpi, MACROBLOCK *x,
|
|||
const aom_variance_fn_ptr_t *fn_ptr,
|
||||
const MV *ref_mv, MV *dst_mv, int is_second);
|
||||
int av1_find_best_obmc_sub_pixel_tree_up(
|
||||
const struct AV1_COMP *cpi, MACROBLOCK *x, int mi_row, int mi_col,
|
||||
MV *bestmv, const MV *ref_mv, int allow_hp, int error_per_bit,
|
||||
const aom_variance_fn_ptr_t *vfp, int forced_stop, int iters_per_step,
|
||||
int *mvjcost, int *mvcost[2], int *distortion, unsigned int *sse1,
|
||||
int is_second, int use_upsampled_ref);
|
||||
MACROBLOCK *x, MV *bestmv, const MV *ref_mv, int allow_hp,
|
||||
int error_per_bit, const aom_variance_fn_ptr_t *vfp, int forced_stop,
|
||||
int iters_per_step, int *mvjcost, int *mvcost[2], int *distortion,
|
||||
unsigned int *sse1, int is_second, int use_upsampled_ref);
|
||||
#endif // CONFIG_MOTION_VAR
|
||||
#ifdef __cplusplus
|
||||
} // extern "C"
|
||||
|
|
@ -157,10 +156,18 @@ int av1_find_best_obmc_sub_pixel_tree_up(
|
|||
unsigned int av1_compute_motion_cost(const struct AV1_COMP *cpi,
|
||||
MACROBLOCK *const x, BLOCK_SIZE bsize,
|
||||
int mi_row, int mi_col, const MV *this_mv);
|
||||
#if WARPED_MOTION_SORT_SAMPLES
|
||||
unsigned int av1_refine_warped_mv(const struct AV1_COMP *cpi,
|
||||
MACROBLOCK *const x, BLOCK_SIZE bsize,
|
||||
int mi_row, int mi_col, int *pts0,
|
||||
int *pts_inref0, int *pts_mv0,
|
||||
int total_samples);
|
||||
#else
|
||||
unsigned int av1_refine_warped_mv(const struct AV1_COMP *cpi,
|
||||
MACROBLOCK *const x, BLOCK_SIZE bsize,
|
||||
int mi_row, int mi_col, int *pts,
|
||||
int *pts_inref);
|
||||
#endif // WARPED_MOTION_SORT_SAMPLES
|
||||
#endif // CONFIG_WARPED_MOTION
|
||||
|
||||
#endif // AV1_ENCODER_MCOMP_H_
|
||||
|
|
|
|||
48
third_party/aom/av1/encoder/palette.c
vendored
48
third_party/aom/av1/encoder/palette.c
vendored
|
|
@ -145,27 +145,6 @@ int av1_remove_duplicates(float *centroids, int num_centroids) {
|
|||
return num_unique;
|
||||
}
|
||||
|
||||
int av1_count_colors(const uint8_t *src, int stride, int rows, int cols) {
|
||||
int n = 0, r, c, i, val_count[256];
|
||||
uint8_t val;
|
||||
memset(val_count, 0, sizeof(val_count));
|
||||
|
||||
for (r = 0; r < rows; ++r) {
|
||||
for (c = 0; c < cols; ++c) {
|
||||
val = src[r * stride + c];
|
||||
++val_count[val];
|
||||
}
|
||||
}
|
||||
|
||||
for (i = 0; i < 256; ++i) {
|
||||
if (val_count[i]) {
|
||||
++n;
|
||||
}
|
||||
}
|
||||
|
||||
return n;
|
||||
}
|
||||
|
||||
#if CONFIG_PALETTE_DELTA_ENCODING
|
||||
static int delta_encode_cost(const int *colors, int num, int bit_depth,
|
||||
int min_val) {
|
||||
|
|
@ -291,30 +270,3 @@ int av1_palette_color_cost_uv(const PALETTE_MODE_INFO *const pmi,
|
|||
return 2 * bit_depth * n * av1_cost_bit(128, 0);
|
||||
#endif // CONFIG_PALETTE_DELTA_ENCODING
|
||||
}
|
||||
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
int av1_count_colors_highbd(const uint8_t *src8, int stride, int rows, int cols,
|
||||
int bit_depth) {
|
||||
int n = 0, r, c, i;
|
||||
uint16_t val;
|
||||
uint16_t *src = CONVERT_TO_SHORTPTR(src8);
|
||||
int val_count[1 << 12];
|
||||
|
||||
assert(bit_depth <= 12);
|
||||
memset(val_count, 0, (1 << 12) * sizeof(val_count[0]));
|
||||
for (r = 0; r < rows; ++r) {
|
||||
for (c = 0; c < cols; ++c) {
|
||||
val = src[r * stride + c];
|
||||
++val_count[val];
|
||||
}
|
||||
}
|
||||
|
||||
for (i = 0; i < (1 << bit_depth); ++i) {
|
||||
if (val_count[i]) {
|
||||
++n;
|
||||
}
|
||||
}
|
||||
|
||||
return n;
|
||||
}
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
|
|
|
|||
8
third_party/aom/av1/encoder/palette.h
vendored
8
third_party/aom/av1/encoder/palette.h
vendored
|
|
@ -36,14 +36,6 @@ void av1_k_means(const float *data, float *centroids, uint8_t *indices, int n,
|
|||
// method.
|
||||
int av1_remove_duplicates(float *centroids, int num_centroids);
|
||||
|
||||
// Returns the number of colors in 'src'.
|
||||
int av1_count_colors(const uint8_t *src, int stride, int rows, int cols);
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
// Same as av1_count_colors(), but for high-bitdepth mode.
|
||||
int av1_count_colors_highbd(const uint8_t *src8, int stride, int rows, int cols,
|
||||
int bit_depth);
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
|
||||
#if CONFIG_PALETTE_DELTA_ENCODING
|
||||
// Given a color cache and a set of base colors, find if each cache color is
|
||||
// present in the base colors, record the binary results in "cache_color_found".
|
||||
|
|
|
|||
102
third_party/aom/av1/encoder/pickcdef.c
vendored
102
third_party/aom/av1/encoder/pickcdef.c
vendored
|
|
@ -19,13 +19,19 @@
|
|||
#include "av1/common/reconinter.h"
|
||||
#include "av1/encoder/encoder.h"
|
||||
|
||||
#define REDUCED_STRENGTHS 8
|
||||
#define REDUCED_TOTAL_STRENGTHS (REDUCED_STRENGTHS * CLPF_STRENGTHS)
|
||||
#define TOTAL_STRENGTHS (DERING_STRENGTHS * CLPF_STRENGTHS)
|
||||
|
||||
static int priconv[REDUCED_STRENGTHS] = { 0, 1, 2, 3, 4, 7, 12, 25 };
|
||||
|
||||
/* Search for the best strength to add as an option, knowing we
|
||||
already selected nb_strengths options. */
|
||||
static uint64_t search_one(int *lev, int nb_strengths,
|
||||
uint64_t mse[][TOTAL_STRENGTHS], int sb_count) {
|
||||
uint64_t mse[][TOTAL_STRENGTHS], int sb_count,
|
||||
int fast) {
|
||||
uint64_t tot_mse[TOTAL_STRENGTHS];
|
||||
const int total_strengths = fast ? REDUCED_TOTAL_STRENGTHS : TOTAL_STRENGTHS;
|
||||
int i, j;
|
||||
uint64_t best_tot_mse = (uint64_t)1 << 63;
|
||||
int best_id = 0;
|
||||
|
|
@ -40,13 +46,13 @@ static uint64_t search_one(int *lev, int nb_strengths,
|
|||
}
|
||||
}
|
||||
/* Find best mse when adding each possible new option. */
|
||||
for (j = 0; j < TOTAL_STRENGTHS; j++) {
|
||||
for (j = 0; j < total_strengths; j++) {
|
||||
uint64_t best = best_mse;
|
||||
if (mse[i][j] < best) best = mse[i][j];
|
||||
tot_mse[j] += best;
|
||||
}
|
||||
}
|
||||
for (j = 0; j < TOTAL_STRENGTHS; j++) {
|
||||
for (j = 0; j < total_strengths; j++) {
|
||||
if (tot_mse[j] < best_tot_mse) {
|
||||
best_tot_mse = tot_mse[j];
|
||||
best_id = j;
|
||||
|
|
@ -59,9 +65,10 @@ static uint64_t search_one(int *lev, int nb_strengths,
|
|||
/* Search for the best luma+chroma strength to add as an option, knowing we
|
||||
already selected nb_strengths options. */
|
||||
static uint64_t search_one_dual(int *lev0, int *lev1, int nb_strengths,
|
||||
uint64_t (**mse)[TOTAL_STRENGTHS],
|
||||
int sb_count) {
|
||||
uint64_t (**mse)[TOTAL_STRENGTHS], int sb_count,
|
||||
int fast) {
|
||||
uint64_t tot_mse[TOTAL_STRENGTHS][TOTAL_STRENGTHS];
|
||||
const int total_strengths = fast ? REDUCED_TOTAL_STRENGTHS : TOTAL_STRENGTHS;
|
||||
int i, j;
|
||||
uint64_t best_tot_mse = (uint64_t)1 << 63;
|
||||
int best_id0 = 0;
|
||||
|
|
@ -79,9 +86,9 @@ static uint64_t search_one_dual(int *lev0, int *lev1, int nb_strengths,
|
|||
}
|
||||
}
|
||||
/* Find best mse when adding each possible new option. */
|
||||
for (j = 0; j < TOTAL_STRENGTHS; j++) {
|
||||
for (j = 0; j < total_strengths; j++) {
|
||||
int k;
|
||||
for (k = 0; k < TOTAL_STRENGTHS; k++) {
|
||||
for (k = 0; k < total_strengths; k++) {
|
||||
uint64_t best = best_mse;
|
||||
uint64_t curr = mse[0][i][j];
|
||||
curr += mse[1][i][k];
|
||||
|
|
@ -90,9 +97,9 @@ static uint64_t search_one_dual(int *lev0, int *lev1, int nb_strengths,
|
|||
}
|
||||
}
|
||||
}
|
||||
for (j = 0; j < TOTAL_STRENGTHS; j++) {
|
||||
for (j = 0; j < total_strengths; j++) {
|
||||
int k;
|
||||
for (k = 0; k < TOTAL_STRENGTHS; k++) {
|
||||
for (k = 0; k < total_strengths; k++) {
|
||||
if (tot_mse[j][k] < best_tot_mse) {
|
||||
best_tot_mse = tot_mse[j][k];
|
||||
best_id0 = j;
|
||||
|
|
@ -108,20 +115,23 @@ static uint64_t search_one_dual(int *lev0, int *lev1, int nb_strengths,
|
|||
/* Search for the set of strengths that minimizes mse. */
|
||||
static uint64_t joint_strength_search(int *best_lev, int nb_strengths,
|
||||
uint64_t mse[][TOTAL_STRENGTHS],
|
||||
int sb_count) {
|
||||
int sb_count, int fast) {
|
||||
uint64_t best_tot_mse;
|
||||
int i;
|
||||
best_tot_mse = (uint64_t)1 << 63;
|
||||
/* Greedy search: add one strength options at a time. */
|
||||
for (i = 0; i < nb_strengths; i++) {
|
||||
best_tot_mse = search_one(best_lev, i, mse, sb_count);
|
||||
best_tot_mse = search_one(best_lev, i, mse, sb_count, fast);
|
||||
}
|
||||
/* Trying to refine the greedy search by reconsidering each
|
||||
already-selected option. */
|
||||
for (i = 0; i < 4 * nb_strengths; i++) {
|
||||
int j;
|
||||
for (j = 0; j < nb_strengths - 1; j++) best_lev[j] = best_lev[j + 1];
|
||||
best_tot_mse = search_one(best_lev, nb_strengths - 1, mse, sb_count);
|
||||
if (!fast) {
|
||||
for (i = 0; i < 4 * nb_strengths; i++) {
|
||||
int j;
|
||||
for (j = 0; j < nb_strengths - 1; j++) best_lev[j] = best_lev[j + 1];
|
||||
best_tot_mse =
|
||||
search_one(best_lev, nb_strengths - 1, mse, sb_count, fast);
|
||||
}
|
||||
}
|
||||
return best_tot_mse;
|
||||
}
|
||||
|
|
@ -130,13 +140,14 @@ static uint64_t joint_strength_search(int *best_lev, int nb_strengths,
|
|||
static uint64_t joint_strength_search_dual(int *best_lev0, int *best_lev1,
|
||||
int nb_strengths,
|
||||
uint64_t (**mse)[TOTAL_STRENGTHS],
|
||||
int sb_count) {
|
||||
int sb_count, int fast) {
|
||||
uint64_t best_tot_mse;
|
||||
int i;
|
||||
best_tot_mse = (uint64_t)1 << 63;
|
||||
/* Greedy search: add one strength options at a time. */
|
||||
for (i = 0; i < nb_strengths; i++) {
|
||||
best_tot_mse = search_one_dual(best_lev0, best_lev1, i, mse, sb_count);
|
||||
best_tot_mse =
|
||||
search_one_dual(best_lev0, best_lev1, i, mse, sb_count, fast);
|
||||
}
|
||||
/* Trying to refine the greedy search by reconsidering each
|
||||
already-selected option. */
|
||||
|
|
@ -146,8 +157,8 @@ static uint64_t joint_strength_search_dual(int *best_lev0, int *best_lev1,
|
|||
best_lev0[j] = best_lev0[j + 1];
|
||||
best_lev1[j] = best_lev1[j + 1];
|
||||
}
|
||||
best_tot_mse =
|
||||
search_one_dual(best_lev0, best_lev1, nb_strengths - 1, mse, sb_count);
|
||||
best_tot_mse = search_one_dual(best_lev0, best_lev1, nb_strengths - 1, mse,
|
||||
sb_count, fast);
|
||||
}
|
||||
return best_tot_mse;
|
||||
}
|
||||
|
|
@ -269,12 +280,12 @@ uint64_t compute_dering_dist(uint16_t *dst, int dstride, uint16_t *src,
|
|||
}
|
||||
|
||||
void av1_cdef_search(YV12_BUFFER_CONFIG *frame, const YV12_BUFFER_CONFIG *ref,
|
||||
AV1_COMMON *cm, MACROBLOCKD *xd) {
|
||||
AV1_COMMON *cm, MACROBLOCKD *xd, int fast) {
|
||||
int r, c;
|
||||
int sbr, sbc;
|
||||
uint16_t *src[3];
|
||||
uint16_t *ref_coeff[3];
|
||||
dering_list dlist[MAX_MIB_SIZE * MAX_MIB_SIZE];
|
||||
dering_list dlist[MI_SIZE_64X64 * MI_SIZE_64X64];
|
||||
int dir[OD_DERING_NBLOCKS][OD_DERING_NBLOCKS] = { { 0 } };
|
||||
int var[OD_DERING_NBLOCKS][OD_DERING_NBLOCKS] = { { 0 } };
|
||||
int stride[3];
|
||||
|
|
@ -289,8 +300,8 @@ void av1_cdef_search(YV12_BUFFER_CONFIG *frame, const YV12_BUFFER_CONFIG *ref,
|
|||
uint64_t best_tot_mse = (uint64_t)1 << 63;
|
||||
uint64_t tot_mse;
|
||||
int sb_count;
|
||||
int nvsb = (cm->mi_rows + MAX_MIB_SIZE - 1) / MAX_MIB_SIZE;
|
||||
int nhsb = (cm->mi_cols + MAX_MIB_SIZE - 1) / MAX_MIB_SIZE;
|
||||
int nvsb = (cm->mi_rows + MI_SIZE_64X64 - 1) / MI_SIZE_64X64;
|
||||
int nhsb = (cm->mi_cols + MI_SIZE_64X64 - 1) / MI_SIZE_64X64;
|
||||
int *sb_index = aom_malloc(nvsb * nhsb * sizeof(*sb_index));
|
||||
int *selected_strength = aom_malloc(nvsb * nhsb * sizeof(*sb_index));
|
||||
uint64_t(*mse[2])[TOTAL_STRENGTHS];
|
||||
|
|
@ -302,6 +313,7 @@ void av1_cdef_search(YV12_BUFFER_CONFIG *frame, const YV12_BUFFER_CONFIG *ref,
|
|||
int quantizer;
|
||||
double lambda;
|
||||
int nplanes = 3;
|
||||
const int total_strengths = fast ? REDUCED_TOTAL_STRENGTHS : TOTAL_STRENGTHS;
|
||||
DECLARE_ALIGNED(32, uint16_t, inbuf[OD_DERING_INBUF_SIZE]);
|
||||
uint16_t *in;
|
||||
DECLARE_ALIGNED(32, uint16_t, tmp_dst[MAX_SB_SQUARE]);
|
||||
|
|
@ -375,22 +387,23 @@ void av1_cdef_search(YV12_BUFFER_CONFIG *frame, const YV12_BUFFER_CONFIG *ref,
|
|||
int nvb, nhb;
|
||||
int gi;
|
||||
int dirinit = 0;
|
||||
nhb = AOMMIN(MAX_MIB_SIZE, cm->mi_cols - MAX_MIB_SIZE * sbc);
|
||||
nvb = AOMMIN(MAX_MIB_SIZE, cm->mi_rows - MAX_MIB_SIZE * sbr);
|
||||
cm->mi_grid_visible[MAX_MIB_SIZE * sbr * cm->mi_stride +
|
||||
MAX_MIB_SIZE * sbc]
|
||||
nhb = AOMMIN(MI_SIZE_64X64, cm->mi_cols - MI_SIZE_64X64 * sbc);
|
||||
nvb = AOMMIN(MI_SIZE_64X64, cm->mi_rows - MI_SIZE_64X64 * sbr);
|
||||
cm->mi_grid_visible[MI_SIZE_64X64 * sbr * cm->mi_stride +
|
||||
MI_SIZE_64X64 * sbc]
|
||||
->mbmi.cdef_strength = -1;
|
||||
if (sb_all_skip(cm, sbr * MAX_MIB_SIZE, sbc * MAX_MIB_SIZE)) continue;
|
||||
dering_count = sb_compute_dering_list(cm, sbr * MAX_MIB_SIZE,
|
||||
sbc * MAX_MIB_SIZE, dlist, 1);
|
||||
if (sb_all_skip(cm, sbr * MI_SIZE_64X64, sbc * MI_SIZE_64X64)) continue;
|
||||
dering_count = sb_compute_dering_list(cm, sbr * MI_SIZE_64X64,
|
||||
sbc * MI_SIZE_64X64, dlist, 1);
|
||||
for (pli = 0; pli < nplanes; pli++) {
|
||||
for (i = 0; i < OD_DERING_INBUF_SIZE; i++)
|
||||
inbuf[i] = OD_DERING_VERY_LARGE;
|
||||
for (gi = 0; gi < TOTAL_STRENGTHS; gi++) {
|
||||
for (gi = 0; gi < total_strengths; gi++) {
|
||||
int threshold;
|
||||
uint64_t curr_mse;
|
||||
int clpf_strength;
|
||||
threshold = gi / CLPF_STRENGTHS;
|
||||
if (fast) threshold = priconv[threshold];
|
||||
if (pli > 0 && !chroma_dering) threshold = 0;
|
||||
/* We avoid filtering the pixels for which some of the pixels to
|
||||
average
|
||||
|
|
@ -406,8 +419,8 @@ void av1_cdef_search(YV12_BUFFER_CONFIG *frame, const YV12_BUFFER_CONFIG *ref,
|
|||
if (clpf_strength == 0)
|
||||
copy_sb16_16(&in[(-yoff * OD_FILT_BSTRIDE - xoff)], OD_FILT_BSTRIDE,
|
||||
src[pli],
|
||||
(sbr * MAX_MIB_SIZE << mi_high_l2[pli]) - yoff,
|
||||
(sbc * MAX_MIB_SIZE << mi_wide_l2[pli]) - xoff,
|
||||
(sbr * MI_SIZE_64X64 << mi_high_l2[pli]) - yoff,
|
||||
(sbc * MI_SIZE_64X64 << mi_wide_l2[pli]) - xoff,
|
||||
stride[pli], ysize, xsize);
|
||||
od_dering(clpf_strength ? NULL : (uint8_t *)in, OD_FILT_BSTRIDE,
|
||||
tmp_dst, in, xdec[pli], ydec[pli], dir, &dirinit, var, pli,
|
||||
|
|
@ -416,8 +429,8 @@ void av1_cdef_search(YV12_BUFFER_CONFIG *frame, const YV12_BUFFER_CONFIG *ref,
|
|||
dering_damping, coeff_shift, clpf_strength != 0, 1);
|
||||
curr_mse = compute_dering_dist(
|
||||
ref_coeff[pli] +
|
||||
(sbr * MAX_MIB_SIZE << mi_high_l2[pli]) * stride[pli] +
|
||||
(sbc * MAX_MIB_SIZE << mi_wide_l2[pli]),
|
||||
(sbr * MI_SIZE_64X64 << mi_high_l2[pli]) * stride[pli] +
|
||||
(sbc * MI_SIZE_64X64 << mi_wide_l2[pli]),
|
||||
stride[pli], tmp_dst, dlist, dering_count, bsize[pli],
|
||||
coeff_shift, pli);
|
||||
if (pli < 2)
|
||||
|
|
@ -425,7 +438,7 @@ void av1_cdef_search(YV12_BUFFER_CONFIG *frame, const YV12_BUFFER_CONFIG *ref,
|
|||
else
|
||||
mse[1][sb_count][gi] += curr_mse;
|
||||
sb_index[sb_count] =
|
||||
MAX_MIB_SIZE * sbr * cm->mi_stride + MAX_MIB_SIZE * sbc;
|
||||
MI_SIZE_64X64 * sbr * cm->mi_stride + MI_SIZE_64X64 * sbc;
|
||||
}
|
||||
}
|
||||
sb_count++;
|
||||
|
|
@ -440,10 +453,10 @@ void av1_cdef_search(YV12_BUFFER_CONFIG *frame, const YV12_BUFFER_CONFIG *ref,
|
|||
nb_strengths = 1 << i;
|
||||
if (nplanes >= 3)
|
||||
tot_mse = joint_strength_search_dual(best_lev0, best_lev1, nb_strengths,
|
||||
mse, sb_count);
|
||||
mse, sb_count, fast);
|
||||
else
|
||||
tot_mse =
|
||||
joint_strength_search(best_lev0, nb_strengths, mse[0], sb_count);
|
||||
tot_mse = joint_strength_search(best_lev0, nb_strengths, mse[0], sb_count,
|
||||
fast);
|
||||
/* Count superblock signalling cost. */
|
||||
tot_mse += (uint64_t)(sb_count * lambda * i);
|
||||
/* Count header signalling cost. */
|
||||
|
|
@ -477,6 +490,17 @@ void av1_cdef_search(YV12_BUFFER_CONFIG *frame, const YV12_BUFFER_CONFIG *ref,
|
|||
selected_strength[i] = best_gi;
|
||||
cm->mi_grid_visible[sb_index[i]]->mbmi.cdef_strength = best_gi;
|
||||
}
|
||||
|
||||
if (fast) {
|
||||
for (int j = 0; j < nb_strengths; j++) {
|
||||
cm->cdef_strengths[j] =
|
||||
priconv[cm->cdef_strengths[j] / CLPF_STRENGTHS] * CLPF_STRENGTHS +
|
||||
(cm->cdef_strengths[j] % CLPF_STRENGTHS);
|
||||
cm->cdef_uv_strengths[j] =
|
||||
priconv[cm->cdef_uv_strengths[j] / CLPF_STRENGTHS] * CLPF_STRENGTHS +
|
||||
(cm->cdef_uv_strengths[j] % CLPF_STRENGTHS);
|
||||
}
|
||||
}
|
||||
cm->cdef_dering_damping = dering_damping;
|
||||
cm->cdef_clpf_damping = clpf_damping;
|
||||
aom_free(mse[0]);
|
||||
|
|
|
|||
111
third_party/aom/av1/encoder/picklpf.c
vendored
111
third_party/aom/av1/encoder/picklpf.c
vendored
|
|
@ -38,13 +38,23 @@ int av1_get_max_filter_level(const AV1_COMP *cpi) {
|
|||
|
||||
static int64_t try_filter_frame(const YV12_BUFFER_CONFIG *sd,
|
||||
AV1_COMP *const cpi, int filt_level,
|
||||
int partial_frame) {
|
||||
int partial_frame
|
||||
#if CONFIG_UV_LVL
|
||||
,
|
||||
int plane
|
||||
#endif
|
||||
) {
|
||||
AV1_COMMON *const cm = &cpi->common;
|
||||
int64_t filt_err;
|
||||
|
||||
#if CONFIG_VAR_TX || CONFIG_EXT_PARTITION || CONFIG_CB4X4
|
||||
#if CONFIG_UV_LVL
|
||||
av1_loop_filter_frame(cm->frame_to_show, cm, &cpi->td.mb.e_mbd, filt_level,
|
||||
plane, partial_frame);
|
||||
#else
|
||||
av1_loop_filter_frame(cm->frame_to_show, cm, &cpi->td.mb.e_mbd, filt_level, 1,
|
||||
partial_frame);
|
||||
#endif // CONFIG_UV_LVL
|
||||
#else
|
||||
if (cpi->num_workers > 1)
|
||||
av1_loop_filter_frame_mt(cm->frame_to_show, cm, cpi->td.mb.e_mbd.plane,
|
||||
|
|
@ -55,6 +65,40 @@ static int64_t try_filter_frame(const YV12_BUFFER_CONFIG *sd,
|
|||
1, partial_frame);
|
||||
#endif
|
||||
|
||||
#if CONFIG_UV_LVL
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
if (cm->use_highbitdepth) {
|
||||
if (plane == 0)
|
||||
filt_err = aom_highbd_get_y_sse(sd, cm->frame_to_show);
|
||||
else if (plane == 1)
|
||||
filt_err = aom_highbd_get_u_sse(sd, cm->frame_to_show);
|
||||
else
|
||||
filt_err = aom_highbd_get_v_sse(sd, cm->frame_to_show);
|
||||
} else {
|
||||
if (plane == 0)
|
||||
filt_err = aom_get_y_sse(sd, cm->frame_to_show);
|
||||
else if (plane == 1)
|
||||
filt_err = aom_get_u_sse(sd, cm->frame_to_show);
|
||||
else
|
||||
filt_err = aom_get_v_sse(sd, cm->frame_to_show);
|
||||
}
|
||||
#else
|
||||
if (plane == 0)
|
||||
filt_err = aom_get_y_sse(sd, cm->frame_to_show);
|
||||
else if (plane == 1)
|
||||
filt_err = aom_get_u_sse(sd, cm->frame_to_show);
|
||||
else
|
||||
filt_err = aom_get_v_sse(sd, cm->frame_to_show);
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
|
||||
// Re-instate the unfiltered frame
|
||||
if (plane == 0)
|
||||
aom_yv12_copy_y(&cpi->last_frame_uf, cm->frame_to_show);
|
||||
else if (plane == 1)
|
||||
aom_yv12_copy_u(&cpi->last_frame_uf, cm->frame_to_show);
|
||||
else
|
||||
aom_yv12_copy_v(&cpi->last_frame_uf, cm->frame_to_show);
|
||||
#else
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
if (cm->use_highbitdepth) {
|
||||
filt_err = aom_highbd_get_y_sse(sd, cm->frame_to_show);
|
||||
|
|
@ -67,12 +111,18 @@ static int64_t try_filter_frame(const YV12_BUFFER_CONFIG *sd,
|
|||
|
||||
// Re-instate the unfiltered frame
|
||||
aom_yv12_copy_y(&cpi->last_frame_uf, cm->frame_to_show);
|
||||
#endif // CONFIG_UV_LVL
|
||||
|
||||
return filt_err;
|
||||
}
|
||||
|
||||
int av1_search_filter_level(const YV12_BUFFER_CONFIG *sd, AV1_COMP *cpi,
|
||||
int partial_frame, double *best_cost_ret) {
|
||||
int partial_frame, double *best_cost_ret
|
||||
#if CONFIG_UV_LVL
|
||||
,
|
||||
int plane
|
||||
#endif
|
||||
) {
|
||||
const AV1_COMMON *const cm = &cpi->common;
|
||||
const struct loopfilter *const lf = &cm->lf;
|
||||
const int min_filter_level = 0;
|
||||
|
|
@ -82,9 +132,20 @@ int av1_search_filter_level(const YV12_BUFFER_CONFIG *sd, AV1_COMP *cpi,
|
|||
int filt_best;
|
||||
MACROBLOCK *x = &cpi->td.mb;
|
||||
|
||||
// Start the search at the previous frame filter level unless it is now out of
|
||||
// range.
|
||||
// Start the search at the previous frame filter level unless it is now out of
|
||||
// range.
|
||||
#if CONFIG_UV_LVL
|
||||
int lvl;
|
||||
switch (plane) {
|
||||
case 0: lvl = lf->filter_level; break;
|
||||
case 1: lvl = lf->filter_level_u; break;
|
||||
case 2: lvl = lf->filter_level_v; break;
|
||||
default: lvl = lf->filter_level; break;
|
||||
}
|
||||
int filt_mid = clamp(lvl, min_filter_level, max_filter_level);
|
||||
#else
|
||||
int filt_mid = clamp(lf->filter_level, min_filter_level, max_filter_level);
|
||||
#endif // CONFIG_UV_LVL
|
||||
int filter_step = filt_mid < 16 ? 4 : filt_mid / 4;
|
||||
// Sum squared error at each filter level
|
||||
int64_t ss_err[MAX_LOOP_FILTER + 1];
|
||||
|
|
@ -92,10 +153,23 @@ int av1_search_filter_level(const YV12_BUFFER_CONFIG *sd, AV1_COMP *cpi,
|
|||
// Set each entry to -1
|
||||
memset(ss_err, 0xFF, sizeof(ss_err));
|
||||
|
||||
#if CONFIG_UV_LVL
|
||||
if (plane == 0)
|
||||
aom_yv12_copy_y(cm->frame_to_show, &cpi->last_frame_uf);
|
||||
else if (plane == 1)
|
||||
aom_yv12_copy_u(cm->frame_to_show, &cpi->last_frame_uf);
|
||||
else if (plane == 2)
|
||||
aom_yv12_copy_v(cm->frame_to_show, &cpi->last_frame_uf);
|
||||
#else
|
||||
// Make a copy of the unfiltered / processed recon buffer
|
||||
aom_yv12_copy_y(cm->frame_to_show, &cpi->last_frame_uf);
|
||||
#endif // CONFIG_UV_LVL
|
||||
|
||||
#if CONFIG_UV_LVL
|
||||
best_err = try_filter_frame(sd, cpi, filt_mid, partial_frame, plane);
|
||||
#else
|
||||
best_err = try_filter_frame(sd, cpi, filt_mid, partial_frame);
|
||||
#endif // CONFIG_UV_LVL
|
||||
filt_best = filt_mid;
|
||||
ss_err[filt_mid] = best_err;
|
||||
|
||||
|
|
@ -115,7 +189,12 @@ int av1_search_filter_level(const YV12_BUFFER_CONFIG *sd, AV1_COMP *cpi,
|
|||
if (filt_direction <= 0 && filt_low != filt_mid) {
|
||||
// Get Low filter error score
|
||||
if (ss_err[filt_low] < 0) {
|
||||
#if CONFIG_UV_LVL
|
||||
ss_err[filt_low] =
|
||||
try_filter_frame(sd, cpi, filt_low, partial_frame, plane);
|
||||
#else
|
||||
ss_err[filt_low] = try_filter_frame(sd, cpi, filt_low, partial_frame);
|
||||
#endif // CONFIG_UV_LVL
|
||||
}
|
||||
// If value is close to the best so far then bias towards a lower loop
|
||||
// filter value.
|
||||
|
|
@ -131,7 +210,12 @@ int av1_search_filter_level(const YV12_BUFFER_CONFIG *sd, AV1_COMP *cpi,
|
|||
// Now look at filt_high
|
||||
if (filt_direction >= 0 && filt_high != filt_mid) {
|
||||
if (ss_err[filt_high] < 0) {
|
||||
#if CONFIG_UV_LVL
|
||||
ss_err[filt_high] =
|
||||
try_filter_frame(sd, cpi, filt_high, partial_frame, plane);
|
||||
#else
|
||||
ss_err[filt_high] = try_filter_frame(sd, cpi, filt_high, partial_frame);
|
||||
#endif // CONFIG_UV_LVL
|
||||
}
|
||||
// If value is significantly better than previous best, bias added against
|
||||
// raising filter value
|
||||
|
|
@ -154,8 +238,7 @@ int av1_search_filter_level(const YV12_BUFFER_CONFIG *sd, AV1_COMP *cpi,
|
|||
// Update best error
|
||||
best_err = ss_err[filt_best];
|
||||
|
||||
if (best_cost_ret)
|
||||
*best_cost_ret = RDCOST_DBL(x->rdmult, x->rddiv, 0, best_err);
|
||||
if (best_cost_ret) *best_cost_ret = RDCOST_DBL(x->rdmult, 0, best_err);
|
||||
return filt_best;
|
||||
}
|
||||
|
||||
|
|
@ -198,14 +281,16 @@ void av1_pick_filter_level(const YV12_BUFFER_CONFIG *sd, AV1_COMP *cpi,
|
|||
if (cm->frame_type == KEY_FRAME) filt_guess -= 4;
|
||||
lf->filter_level = clamp(filt_guess, min_filter_level, max_filter_level);
|
||||
} else {
|
||||
#if CONFIG_UV_LVL
|
||||
lf->filter_level = av1_search_filter_level(
|
||||
sd, cpi, method == LPF_PICK_FROM_SUBIMAGE, NULL, 0);
|
||||
lf->filter_level_u = av1_search_filter_level(
|
||||
sd, cpi, method == LPF_PICK_FROM_SUBIMAGE, NULL, 1);
|
||||
lf->filter_level_v = av1_search_filter_level(
|
||||
sd, cpi, method == LPF_PICK_FROM_SUBIMAGE, NULL, 2);
|
||||
#else
|
||||
lf->filter_level = av1_search_filter_level(
|
||||
sd, cpi, method == LPF_PICK_FROM_SUBIMAGE, NULL);
|
||||
#endif // CONFIG_UV_LVL
|
||||
}
|
||||
|
||||
#if CONFIG_EXT_TILE
|
||||
// TODO(any): 0 loopfilter level is only necessary if individual tile
|
||||
// decoding is required. We need to communicate this requirement to this
|
||||
// code and force loop filter level 0 only if required.
|
||||
if (cm->tile_encoding_mode) lf->filter_level = 0;
|
||||
#endif // CONFIG_EXT_TILE
|
||||
}
|
||||
|
|
|
|||
5
third_party/aom/av1/encoder/picklpf.h
vendored
5
third_party/aom/av1/encoder/picklpf.h
vendored
|
|
@ -21,8 +21,13 @@ extern "C" {
|
|||
struct yv12_buffer_config;
|
||||
struct AV1_COMP;
|
||||
int av1_get_max_filter_level(const AV1_COMP *cpi);
|
||||
#if CONFIG_UV_LVL
|
||||
int av1_search_filter_level(const YV12_BUFFER_CONFIG *sd, AV1_COMP *cpi,
|
||||
int partial_frame, double *err, int plane);
|
||||
#else
|
||||
int av1_search_filter_level(const YV12_BUFFER_CONFIG *sd, AV1_COMP *cpi,
|
||||
int partial_frame, double *err);
|
||||
#endif
|
||||
void av1_pick_filter_level(const struct yv12_buffer_config *sd,
|
||||
struct AV1_COMP *cpi, LPF_PICK_METHOD method);
|
||||
#ifdef __cplusplus
|
||||
|
|
|
|||
73
third_party/aom/av1/encoder/pickrst.c
vendored
73
third_party/aom/av1/encoder/pickrst.c
vendored
|
|
@ -437,8 +437,8 @@ static double search_sgrproj(const YV12_BUFFER_CONFIG *src, AV1_COMP *cpi,
|
|||
int width, height, src_stride, dgd_stride;
|
||||
uint8_t *dgd_buffer, *src_buffer;
|
||||
if (plane == AOM_PLANE_Y) {
|
||||
width = cm->width;
|
||||
height = cm->height;
|
||||
width = src->y_crop_width;
|
||||
height = src->y_crop_height;
|
||||
src_buffer = src->y_buffer;
|
||||
src_stride = src->y_stride;
|
||||
dgd_buffer = dgd->y_buffer;
|
||||
|
|
@ -478,7 +478,7 @@ static double search_sgrproj(const YV12_BUFFER_CONFIG *src, AV1_COMP *cpi,
|
|||
(1 << plane));
|
||||
// #bits when a tile is not restored
|
||||
bits = av1_cost_bit(RESTORE_NONE_SGRPROJ_PROB, 0);
|
||||
cost_norestore = RDCOST_DBL(x->rdmult, x->rddiv, (bits >> 4), err);
|
||||
cost_norestore = RDCOST_DBL(x->rdmult, (bits >> 4), err);
|
||||
best_tile_cost[tile_idx] = DBL_MAX;
|
||||
search_selfguided_restoration(
|
||||
dgd_buffer + v_start * dgd_stride + h_start, h_end - h_start,
|
||||
|
|
@ -498,7 +498,7 @@ static double search_sgrproj(const YV12_BUFFER_CONFIG *src, AV1_COMP *cpi,
|
|||
&ref_sgrproj_info)
|
||||
<< AV1_PROB_COST_SHIFT;
|
||||
bits += av1_cost_bit(RESTORE_NONE_SGRPROJ_PROB, 1);
|
||||
cost_sgrproj = RDCOST_DBL(x->rdmult, x->rddiv, (bits >> 4), err);
|
||||
cost_sgrproj = RDCOST_DBL(x->rdmult, (bits >> 4), err);
|
||||
if (cost_sgrproj >= cost_norestore) {
|
||||
type[tile_idx] = RESTORE_NONE;
|
||||
} else {
|
||||
|
|
@ -531,7 +531,7 @@ static double search_sgrproj(const YV12_BUFFER_CONFIG *src, AV1_COMP *cpi,
|
|||
}
|
||||
err = try_restoration_frame(src, cpi, rsi, (1 << plane), partial_frame,
|
||||
dst_frame);
|
||||
cost_sgrproj = RDCOST_DBL(x->rdmult, x->rddiv, (bits >> 4), err);
|
||||
cost_sgrproj = RDCOST_DBL(x->rdmult, (bits >> 4), err);
|
||||
|
||||
return cost_sgrproj;
|
||||
}
|
||||
|
|
@ -985,8 +985,8 @@ static double search_wiener(const YV12_BUFFER_CONFIG *src, AV1_COMP *cpi,
|
|||
int width, height, src_stride, dgd_stride;
|
||||
uint8_t *dgd_buffer, *src_buffer;
|
||||
if (plane == AOM_PLANE_Y) {
|
||||
width = cm->width;
|
||||
height = cm->height;
|
||||
width = src->y_crop_width;
|
||||
height = src->y_crop_height;
|
||||
src_buffer = src->y_buffer;
|
||||
src_stride = src->y_stride;
|
||||
dgd_buffer = dgd->y_buffer;
|
||||
|
|
@ -1039,7 +1039,7 @@ static double search_wiener(const YV12_BUFFER_CONFIG *src, AV1_COMP *cpi,
|
|||
(1 << plane));
|
||||
// #bits when a tile is not restored
|
||||
bits = av1_cost_bit(RESTORE_NONE_WIENER_PROB, 0);
|
||||
cost_norestore = RDCOST_DBL(x->rdmult, x->rddiv, (bits >> 4), err);
|
||||
cost_norestore = RDCOST_DBL(x->rdmult, (bits >> 4), err);
|
||||
best_tile_cost[tile_idx] = DBL_MAX;
|
||||
|
||||
av1_get_rest_tile_limits(tile_idx, 0, 0, nhtiles, nvtiles, tile_width,
|
||||
|
|
@ -1081,7 +1081,7 @@ static double search_wiener(const YV12_BUFFER_CONFIG *src, AV1_COMP *cpi,
|
|||
count_wiener_bits(&rsi[plane].wiener_info[tile_idx], &ref_wiener_info)
|
||||
<< AV1_PROB_COST_SHIFT;
|
||||
bits += av1_cost_bit(RESTORE_NONE_WIENER_PROB, 1);
|
||||
cost_wiener = RDCOST_DBL(x->rdmult, x->rddiv, (bits >> 4), err);
|
||||
cost_wiener = RDCOST_DBL(x->rdmult, (bits >> 4), err);
|
||||
if (cost_wiener >= cost_norestore) {
|
||||
type[tile_idx] = RESTORE_NONE;
|
||||
} else {
|
||||
|
|
@ -1114,7 +1114,7 @@ static double search_wiener(const YV12_BUFFER_CONFIG *src, AV1_COMP *cpi,
|
|||
}
|
||||
err = try_restoration_frame(src, cpi, rsi, 1 << plane, partial_frame,
|
||||
dst_frame);
|
||||
cost_wiener = RDCOST_DBL(x->rdmult, x->rddiv, (bits >> 4), err);
|
||||
cost_wiener = RDCOST_DBL(x->rdmult, (bits >> 4), err);
|
||||
|
||||
return cost_wiener;
|
||||
}
|
||||
|
|
@ -1133,8 +1133,8 @@ static double search_norestore(const YV12_BUFFER_CONFIG *src, AV1_COMP *cpi,
|
|||
int h_start, h_end, v_start, v_end;
|
||||
int width, height;
|
||||
if (plane == AOM_PLANE_Y) {
|
||||
width = cm->width;
|
||||
height = cm->height;
|
||||
width = src->y_crop_width;
|
||||
height = src->y_crop_height;
|
||||
} else {
|
||||
width = src->uv_crop_width;
|
||||
height = src->uv_crop_height;
|
||||
|
|
@ -1160,13 +1160,14 @@ static double search_norestore(const YV12_BUFFER_CONFIG *src, AV1_COMP *cpi,
|
|||
// RD cost associated with no restoration
|
||||
err = sse_restoration_frame(cm, src, cm->frame_to_show, (1 << plane));
|
||||
bits = frame_level_restore_bits[RESTORE_NONE] << AV1_PROB_COST_SHIFT;
|
||||
cost_norestore = RDCOST_DBL(x->rdmult, x->rddiv, (bits >> 4), err);
|
||||
cost_norestore = RDCOST_DBL(x->rdmult, (bits >> 4), err);
|
||||
return cost_norestore;
|
||||
}
|
||||
|
||||
static double search_switchable_restoration(
|
||||
AV1_COMP *cpi, int partial_frame, int plane, RestorationInfo *rsi,
|
||||
double *tile_cost[RESTORE_SWITCHABLE_TYPES]) {
|
||||
const YV12_BUFFER_CONFIG *src, AV1_COMP *cpi, int partial_frame, int plane,
|
||||
RestorationType *const restore_types[RESTORE_SWITCHABLE_TYPES],
|
||||
double *const tile_cost[RESTORE_SWITCHABLE_TYPES], RestorationInfo *rsi) {
|
||||
AV1_COMMON *const cm = &cpi->common;
|
||||
MACROBLOCK *x = &cpi->td.mb;
|
||||
double cost_switchable = 0;
|
||||
|
|
@ -1174,11 +1175,11 @@ static double search_switchable_restoration(
|
|||
RestorationType r;
|
||||
int width, height;
|
||||
if (plane == AOM_PLANE_Y) {
|
||||
width = cm->width;
|
||||
height = cm->height;
|
||||
width = src->y_crop_width;
|
||||
height = src->y_crop_height;
|
||||
} else {
|
||||
width = ROUND_POWER_OF_TWO(cm->width, cm->subsampling_x);
|
||||
height = ROUND_POWER_OF_TWO(cm->height, cm->subsampling_y);
|
||||
width = src->uv_crop_width;
|
||||
height = src->uv_crop_height;
|
||||
}
|
||||
const int ntiles = av1_get_rest_ntiles(
|
||||
width, height, cm->rst_info[plane].restoration_tilesize, NULL, NULL, NULL,
|
||||
|
|
@ -1192,16 +1193,17 @@ static double search_switchable_restoration(
|
|||
rsi->frame_restoration_type = RESTORE_SWITCHABLE;
|
||||
bits = frame_level_restore_bits[rsi->frame_restoration_type]
|
||||
<< AV1_PROB_COST_SHIFT;
|
||||
cost_switchable = RDCOST_DBL(x->rdmult, x->rddiv, bits >> 4, 0);
|
||||
cost_switchable = RDCOST_DBL(x->rdmult, bits >> 4, 0);
|
||||
for (tile_idx = 0; tile_idx < ntiles; ++tile_idx) {
|
||||
double best_cost = RDCOST_DBL(
|
||||
x->rdmult, x->rddiv, (cpi->switchable_restore_cost[RESTORE_NONE] >> 4),
|
||||
tile_cost[RESTORE_NONE][tile_idx]);
|
||||
double best_cost =
|
||||
RDCOST_DBL(x->rdmult, (cpi->switchable_restore_cost[RESTORE_NONE] >> 4),
|
||||
tile_cost[RESTORE_NONE][tile_idx]);
|
||||
rsi->restoration_type[tile_idx] = RESTORE_NONE;
|
||||
for (r = 1; r < RESTORE_SWITCHABLE_TYPES; r++) {
|
||||
if (force_restore_type != 0)
|
||||
if (r != force_restore_type) continue;
|
||||
int tilebits = 0;
|
||||
if (restore_types[r][tile_idx] != r) continue;
|
||||
if (r == RESTORE_WIENER)
|
||||
tilebits +=
|
||||
count_wiener_bits(&rsi->wiener_info[tile_idx], &ref_wiener_info);
|
||||
|
|
@ -1210,8 +1212,8 @@ static double search_switchable_restoration(
|
|||
count_sgrproj_bits(&rsi->sgrproj_info[tile_idx], &ref_sgrproj_info);
|
||||
tilebits <<= AV1_PROB_COST_SHIFT;
|
||||
tilebits += cpi->switchable_restore_cost[r];
|
||||
double cost = RDCOST_DBL(x->rdmult, x->rddiv, tilebits >> 4,
|
||||
tile_cost[r][tile_idx]);
|
||||
double cost =
|
||||
RDCOST_DBL(x->rdmult, tilebits >> 4, tile_cost[r][tile_idx]);
|
||||
|
||||
if (cost < best_cost) {
|
||||
rsi->restoration_type[tile_idx] = r;
|
||||
|
|
@ -1243,14 +1245,17 @@ void av1_pick_filter_restoration(const YV12_BUFFER_CONFIG *src, AV1_COMP *cpi,
|
|||
RestorationType *restore_types[RESTORE_SWITCHABLE_TYPES];
|
||||
double best_cost_restore;
|
||||
RestorationType r, best_restore;
|
||||
const int ywidth = src->y_crop_width;
|
||||
const int yheight = src->y_crop_height;
|
||||
const int uvwidth = src->uv_crop_width;
|
||||
const int uvheight = src->uv_crop_height;
|
||||
|
||||
const int ntiles_y = av1_get_rest_ntiles(cm->width, cm->height,
|
||||
cm->rst_info[0].restoration_tilesize,
|
||||
NULL, NULL, NULL, NULL);
|
||||
const int ntiles_y =
|
||||
av1_get_rest_ntiles(ywidth, yheight, cm->rst_info[0].restoration_tilesize,
|
||||
NULL, NULL, NULL, NULL);
|
||||
const int ntiles_uv = av1_get_rest_ntiles(
|
||||
ROUND_POWER_OF_TWO(cm->width, cm->subsampling_x),
|
||||
ROUND_POWER_OF_TWO(cm->height, cm->subsampling_y),
|
||||
cm->rst_info[1].restoration_tilesize, NULL, NULL, NULL, NULL);
|
||||
uvwidth, uvheight, cm->rst_info[1].restoration_tilesize, NULL, NULL, NULL,
|
||||
NULL);
|
||||
|
||||
// Assume ntiles_uv is never larger that ntiles_y and so the same arrays work.
|
||||
for (r = 0; r < RESTORE_SWITCHABLE_TYPES; r++) {
|
||||
|
|
@ -1270,9 +1275,9 @@ void av1_pick_filter_restoration(const YV12_BUFFER_CONFIG *src, AV1_COMP *cpi,
|
|||
tile_cost[r], &cpi->trial_frame_rst);
|
||||
}
|
||||
if (plane == AOM_PLANE_Y)
|
||||
cost_restore[RESTORE_SWITCHABLE] =
|
||||
search_switchable_restoration(cpi, method == LPF_PICK_FROM_SUBIMAGE,
|
||||
plane, &cm->rst_info[plane], tile_cost);
|
||||
cost_restore[RESTORE_SWITCHABLE] = search_switchable_restoration(
|
||||
src, cpi, method == LPF_PICK_FROM_SUBIMAGE, plane, restore_types,
|
||||
tile_cost, &cm->rst_info[plane]);
|
||||
else
|
||||
cost_restore[RESTORE_SWITCHABLE] = DBL_MAX;
|
||||
best_cost_restore = DBL_MAX;
|
||||
|
|
|
|||
16
third_party/aom/av1/encoder/ransac.c
vendored
16
third_party/aom/av1/encoder/ransac.c
vendored
|
|
@ -139,6 +139,8 @@ static void normalize_homography(double *pts, int n, double *T) {
|
|||
double msqe = 0;
|
||||
double scale;
|
||||
int i;
|
||||
|
||||
assert(n > 0);
|
||||
for (i = 0; i < n; ++i, p += 2) {
|
||||
mean[0] += p[0];
|
||||
mean[1] += p[1];
|
||||
|
|
@ -821,13 +823,15 @@ static int ransac(const int *matched_points, int npoints,
|
|||
|
||||
// Recompute the motions using only the inliers.
|
||||
for (i = 0; i < num_desired_motions; ++i) {
|
||||
copy_points_at_indices(points1, corners1, motions[i].inlier_indices,
|
||||
motions[i].num_inliers);
|
||||
copy_points_at_indices(points2, corners2, motions[i].inlier_indices,
|
||||
motions[i].num_inliers);
|
||||
if (motions[i].num_inliers >= minpts) {
|
||||
copy_points_at_indices(points1, corners1, motions[i].inlier_indices,
|
||||
motions[i].num_inliers);
|
||||
copy_points_at_indices(points2, corners2, motions[i].inlier_indices,
|
||||
motions[i].num_inliers);
|
||||
|
||||
find_transformation(motions[i].num_inliers, points1, points2,
|
||||
params_by_motion + (MAX_PARAMDIM - 1) * i);
|
||||
find_transformation(motions[i].num_inliers, points1, points2,
|
||||
params_by_motion + (MAX_PARAMDIM - 1) * i);
|
||||
}
|
||||
num_inliers_by_motion[i] = motions[i].num_inliers;
|
||||
}
|
||||
|
||||
|
|
|
|||
69
third_party/aom/av1/encoder/ratectrl.c
vendored
69
third_party/aom/av1/encoder/ratectrl.c
vendored
|
|
@ -94,8 +94,8 @@ static int kf_high = 5000;
|
|||
static int kf_low = 400;
|
||||
|
||||
double av1_resize_rate_factor(const AV1_COMP *cpi) {
|
||||
return (double)(cpi->resize_scale_den * cpi->resize_scale_den) /
|
||||
(cpi->resize_scale_num * cpi->resize_scale_num);
|
||||
return (double)(cpi->oxcf.width * cpi->oxcf.height) /
|
||||
(cpi->common.width * cpi->common.height);
|
||||
}
|
||||
|
||||
// Functions to compute the active minq lookup table entries based on a
|
||||
|
|
@ -1081,7 +1081,7 @@ static int rc_pick_q_and_bounds_two_pass(const AV1_COMP *cpi, int *bottom_index,
|
|||
}
|
||||
|
||||
// Modify active_best_quality for downscaled normal frames.
|
||||
if (!av1_resize_unscaled(cpi) && !frame_is_kf_gf_arf(cpi)) {
|
||||
if (!av1_frame_unscaled(cm) && !frame_is_kf_gf_arf(cpi)) {
|
||||
int qdelta = av1_compute_qdelta_by_rate(
|
||||
rc, cm->frame_type, active_best_quality, 2.0, cm->bit_depth);
|
||||
active_best_quality =
|
||||
|
|
@ -1164,7 +1164,7 @@ void av1_rc_set_frame_target(AV1_COMP *cpi, int target) {
|
|||
rc->this_frame_target = target;
|
||||
|
||||
// Modify frame size target when down-scaled.
|
||||
if (cpi->oxcf.resize_mode == RESIZE_DYNAMIC && !av1_resize_unscaled(cpi))
|
||||
if (!av1_frame_unscaled(cm))
|
||||
rc->this_frame_target =
|
||||
(int)(rc->this_frame_target * av1_resize_rate_factor(cpi));
|
||||
|
||||
|
|
@ -1663,3 +1663,64 @@ void av1_set_target_rate(AV1_COMP *cpi) {
|
|||
vbr_rate_correction(cpi, &target_rate);
|
||||
av1_rc_set_frame_target(cpi, target_rate);
|
||||
}
|
||||
|
||||
static unsigned int lcg_rand16(unsigned int *state) {
|
||||
*state = (unsigned int)(*state * 1103515245ULL + 12345);
|
||||
return *state / 65536 % 32768;
|
||||
}
|
||||
|
||||
uint8_t av1_calculate_next_resize_scale(const AV1_COMP *cpi) {
|
||||
static unsigned int seed = 56789;
|
||||
const AV1EncoderConfig *oxcf = &cpi->oxcf;
|
||||
if (oxcf->pass == 1) return SCALE_DENOMINATOR;
|
||||
uint8_t new_num = SCALE_DENOMINATOR;
|
||||
|
||||
switch (oxcf->resize_mode) {
|
||||
case RESIZE_NONE: new_num = SCALE_DENOMINATOR; break;
|
||||
case RESIZE_FIXED:
|
||||
if (cpi->common.frame_type == KEY_FRAME)
|
||||
new_num = oxcf->resize_kf_scale_numerator;
|
||||
else
|
||||
new_num = oxcf->resize_scale_numerator;
|
||||
break;
|
||||
case RESIZE_DYNAMIC:
|
||||
// RESIZE_DYNAMIC: Just random for now.
|
||||
new_num = lcg_rand16(&seed) % 4 + 13;
|
||||
break;
|
||||
default: assert(0);
|
||||
}
|
||||
return new_num;
|
||||
}
|
||||
|
||||
#if CONFIG_FRAME_SUPERRES
|
||||
// TODO(afergs): Rename av1_rc_update_superres_scale(...)?
|
||||
uint8_t av1_calculate_next_superres_scale(const AV1_COMP *cpi, int width,
|
||||
int height) {
|
||||
static unsigned int seed = 34567;
|
||||
const AV1EncoderConfig *oxcf = &cpi->oxcf;
|
||||
if (oxcf->pass == 1) return SCALE_DENOMINATOR;
|
||||
uint8_t new_num = SCALE_DENOMINATOR;
|
||||
|
||||
switch (oxcf->superres_mode) {
|
||||
case SUPERRES_NONE: new_num = SCALE_DENOMINATOR; break;
|
||||
case SUPERRES_FIXED:
|
||||
if (cpi->common.frame_type == KEY_FRAME)
|
||||
new_num = oxcf->superres_kf_scale_numerator;
|
||||
else
|
||||
new_num = oxcf->superres_scale_numerator;
|
||||
break;
|
||||
case SUPERRES_DYNAMIC:
|
||||
// SUPERRES_DYNAMIC: Just random for now.
|
||||
new_num = lcg_rand16(&seed) % 9 + 8;
|
||||
break;
|
||||
default: assert(0);
|
||||
}
|
||||
|
||||
// Make sure overall reduction is no more than 1/2 of the source size.
|
||||
av1_calculate_scaled_size(&width, &height, new_num);
|
||||
if (width * 2 < oxcf->width || height * 2 < oxcf->height)
|
||||
new_num = SCALE_DENOMINATOR;
|
||||
|
||||
return new_num;
|
||||
}
|
||||
#endif // CONFIG_FRAME_SUPERRES
|
||||
|
|
|
|||
5
third_party/aom/av1/encoder/ratectrl.h
vendored
5
third_party/aom/av1/encoder/ratectrl.h
vendored
|
|
@ -256,6 +256,11 @@ void av1_set_target_rate(struct AV1_COMP *cpi);
|
|||
|
||||
int av1_resize_one_pass_cbr(struct AV1_COMP *cpi);
|
||||
|
||||
uint8_t av1_calculate_next_resize_scale(const struct AV1_COMP *cpi);
|
||||
#if CONFIG_FRAME_SUPERRES
|
||||
uint8_t av1_calculate_next_superres_scale(const struct AV1_COMP *cpi, int width,
|
||||
int height);
|
||||
#endif // CONFIG_FRAME_SUPERRES
|
||||
#ifdef __cplusplus
|
||||
} // extern "C"
|
||||
#endif
|
||||
|
|
|
|||
196
third_party/aom/av1/encoder/rd.c
vendored
196
third_party/aom/av1/encoder/rd.c
vendored
|
|
@ -50,14 +50,15 @@
|
|||
// certain modes are assumed to be based on 8x8 blocks.
|
||||
// This table is used to correct for block size.
|
||||
// The factors here are << 2 (2 = x0.5, 32 = x8 etc).
|
||||
static const uint8_t rd_thresh_block_size_factor[BLOCK_SIZES] = {
|
||||
#if CONFIG_CB4X4
|
||||
static const uint8_t rd_thresh_block_size_factor[BLOCK_SIZES_ALL] = {
|
||||
#if CONFIG_CHROMA_2X2 || CONFIG_CHROMA_SUB8X8
|
||||
2, 2, 2,
|
||||
#endif
|
||||
2, 3, 3, 4, 6, 6, 8, 12, 12, 16, 24, 24, 32,
|
||||
2, 3, 3, 4, 6, 6, 8, 12, 12, 16, 24, 24, 32,
|
||||
#if CONFIG_EXT_PARTITION
|
||||
48, 48, 64
|
||||
48, 48, 64,
|
||||
#endif // CONFIG_EXT_PARTITION
|
||||
4, 4, 8, 8
|
||||
};
|
||||
|
||||
static void fill_mode_costs(AV1_COMP *cpi) {
|
||||
|
|
@ -66,16 +67,16 @@ static void fill_mode_costs(AV1_COMP *cpi) {
|
|||
|
||||
for (i = 0; i < INTRA_MODES; ++i)
|
||||
for (j = 0; j < INTRA_MODES; ++j)
|
||||
av1_cost_tokens(cpi->y_mode_costs[i][j], av1_kf_y_mode_prob[i][j],
|
||||
av1_intra_mode_tree);
|
||||
av1_cost_tokens_from_cdf(cpi->y_mode_costs[i][j], av1_kf_y_mode_cdf[i][j],
|
||||
av1_intra_mode_inv);
|
||||
|
||||
for (i = 0; i < BLOCK_SIZE_GROUPS; ++i)
|
||||
av1_cost_tokens(cpi->mbmode_cost[i], fc->y_mode_prob[i],
|
||||
av1_intra_mode_tree);
|
||||
av1_cost_tokens_from_cdf(cpi->mbmode_cost[i], fc->y_mode_cdf[i],
|
||||
av1_intra_mode_inv);
|
||||
|
||||
for (i = 0; i < INTRA_MODES; ++i)
|
||||
av1_cost_tokens(cpi->intra_uv_mode_cost[i], fc->uv_mode_prob[i],
|
||||
av1_intra_mode_tree);
|
||||
av1_cost_tokens_from_cdf(cpi->intra_uv_mode_cost[i], fc->uv_mode_cdf[i],
|
||||
av1_intra_mode_inv);
|
||||
|
||||
for (i = 0; i < SWITCHABLE_FILTER_CONTEXTS; ++i)
|
||||
av1_cost_tokens(cpi->switchable_interp_costs[i],
|
||||
|
|
@ -83,20 +84,18 @@ static void fill_mode_costs(AV1_COMP *cpi) {
|
|||
|
||||
#if CONFIG_PALETTE
|
||||
for (i = 0; i < PALETTE_BLOCK_SIZES; ++i) {
|
||||
av1_cost_tokens(cpi->palette_y_size_cost[i],
|
||||
av1_default_palette_y_size_prob[i], av1_palette_size_tree);
|
||||
av1_cost_tokens(cpi->palette_uv_size_cost[i],
|
||||
av1_default_palette_uv_size_prob[i], av1_palette_size_tree);
|
||||
av1_cost_tokens_from_cdf(cpi->palette_y_size_cost[i],
|
||||
fc->palette_y_size_cdf[i], NULL);
|
||||
av1_cost_tokens_from_cdf(cpi->palette_uv_size_cost[i],
|
||||
fc->palette_uv_size_cdf[i], NULL);
|
||||
}
|
||||
|
||||
for (i = 0; i < PALETTE_SIZES; ++i) {
|
||||
for (j = 0; j < PALETTE_COLOR_INDEX_CONTEXTS; ++j) {
|
||||
av1_cost_tokens(cpi->palette_y_color_cost[i][j],
|
||||
av1_default_palette_y_color_index_prob[i][j],
|
||||
av1_palette_color_index_tree[i]);
|
||||
av1_cost_tokens(cpi->palette_uv_color_cost[i][j],
|
||||
av1_default_palette_uv_color_index_prob[i][j],
|
||||
av1_palette_color_index_tree[i]);
|
||||
av1_cost_tokens_from_cdf(cpi->palette_y_color_cost[i][j],
|
||||
fc->palette_y_color_index_cdf[i][j], NULL);
|
||||
av1_cost_tokens_from_cdf(cpi->palette_uv_color_cost[i][j],
|
||||
fc->palette_uv_color_index_cdf[i][j], NULL);
|
||||
}
|
||||
}
|
||||
#endif // CONFIG_PALETTE
|
||||
|
|
@ -147,8 +146,9 @@ static void fill_mode_costs(AV1_COMP *cpi) {
|
|||
av1_switchable_restore_tree);
|
||||
#endif // CONFIG_LOOP_RESTORATION
|
||||
#if CONFIG_GLOBAL_MOTION
|
||||
av1_cost_tokens(cpi->gmtype_cost, fc->global_motion_types_prob,
|
||||
av1_global_motion_types_tree);
|
||||
for (i = 0; i < TRANS_TYPES; ++i)
|
||||
cpi->gmtype_cost[i] = (1 + (i > 0 ? GLOBAL_TYPE_BITS : 0))
|
||||
<< AV1_PROB_COST_SHIFT;
|
||||
#endif // CONFIG_GLOBAL_MOTION
|
||||
}
|
||||
|
||||
|
|
@ -301,7 +301,7 @@ static void set_block_thresholds(const AV1_COMMON *cm, RD_OPT *rd) {
|
|||
0, MAXQ);
|
||||
const int q = compute_rd_thresh_factor(qindex, cm->bit_depth);
|
||||
|
||||
for (bsize = 0; bsize < BLOCK_SIZES; ++bsize) {
|
||||
for (bsize = 0; bsize < BLOCK_SIZES_ALL; ++bsize) {
|
||||
// Threshold here seems unnecessarily harsh but fine given actual
|
||||
// range of values used for cpi->sf.thresh_mult[].
|
||||
const int t = q * rd_thresh_block_size_factor[bsize];
|
||||
|
|
@ -350,7 +350,6 @@ void av1_initialize_rd_consts(AV1_COMP *cpi) {
|
|||
|
||||
aom_clear_system_state();
|
||||
|
||||
rd->RDDIV = RDDIV_BITS; // In bits (to multiply D by 128).
|
||||
rd->RDMULT = av1_compute_rd_mult(cpi, cm->base_qindex + cm->y_dc_delta_q);
|
||||
|
||||
set_error_per_bit(x, rd->RDMULT);
|
||||
|
|
@ -367,6 +366,16 @@ void av1_initialize_rd_consts(AV1_COMP *cpi) {
|
|||
x->mvcost = x->mv_cost_stack[0];
|
||||
x->nmvjointcost = x->nmv_vec_cost[0];
|
||||
|
||||
#if CONFIG_INTRABC
|
||||
if (frame_is_intra_only(cm) && cm->allow_screen_content_tools &&
|
||||
cpi->oxcf.pass != 1) {
|
||||
av1_build_nmv_cost_table(
|
||||
x->nmv_vec_cost[0],
|
||||
cm->allow_high_precision_mv ? x->nmvcost_hp[0] : x->nmvcost[0],
|
||||
&cm->fc->ndvc, MV_SUBPEL_NONE);
|
||||
}
|
||||
#endif
|
||||
|
||||
if (cpi->oxcf.pass != 1) {
|
||||
av1_fill_token_costs(x->token_costs, cm->fc->coef_probs);
|
||||
|
||||
|
|
@ -434,6 +443,12 @@ void av1_initialize_rd_consts(AV1_COMP *cpi) {
|
|||
av1_cost_tokens((int *)cpi->inter_compound_mode_cost[i],
|
||||
cm->fc->inter_compound_mode_probs[i],
|
||||
av1_inter_compound_mode_tree);
|
||||
#if CONFIG_COMPOUND_SINGLEREF
|
||||
for (i = 0; i < INTER_MODE_CONTEXTS; ++i)
|
||||
av1_cost_tokens((int *)cpi->inter_singleref_comp_mode_cost[i],
|
||||
cm->fc->inter_singleref_comp_mode_probs[i],
|
||||
av1_inter_singleref_comp_mode_tree);
|
||||
#endif // CONFIG_COMPOUND_SINGLEREF
|
||||
#if CONFIG_INTERINTRA
|
||||
for (i = 0; i < BLOCK_SIZE_GROUPS; ++i)
|
||||
av1_cost_tokens((int *)cpi->interintra_mode_cost[i],
|
||||
|
|
@ -442,16 +457,22 @@ void av1_initialize_rd_consts(AV1_COMP *cpi) {
|
|||
#endif // CONFIG_INTERINTRA
|
||||
#endif // CONFIG_EXT_INTER
|
||||
#if CONFIG_MOTION_VAR || CONFIG_WARPED_MOTION
|
||||
for (i = BLOCK_8X8; i < BLOCK_SIZES; i++) {
|
||||
for (i = BLOCK_8X8; i < BLOCK_SIZES_ALL; i++) {
|
||||
av1_cost_tokens((int *)cpi->motion_mode_cost[i],
|
||||
cm->fc->motion_mode_prob[i], av1_motion_mode_tree);
|
||||
}
|
||||
#if CONFIG_MOTION_VAR && CONFIG_WARPED_MOTION
|
||||
for (i = BLOCK_8X8; i < BLOCK_SIZES; i++) {
|
||||
for (i = BLOCK_8X8; i < BLOCK_SIZES_ALL; i++) {
|
||||
cpi->motion_mode_cost1[i][0] = av1_cost_bit(cm->fc->obmc_prob[i], 0);
|
||||
cpi->motion_mode_cost1[i][1] = av1_cost_bit(cm->fc->obmc_prob[i], 1);
|
||||
}
|
||||
#endif // CONFIG_MOTION_VAR && CONFIG_WARPED_MOTION
|
||||
#if CONFIG_MOTION_VAR && CONFIG_NCOBMC_ADAPT_WEIGHT
|
||||
for (i = ADAPT_OVERLAP_BLOCK_8X8; i < ADAPT_OVERLAP_BLOCKS; ++i) {
|
||||
av1_cost_tokens((int *)cpi->ncobmc_mode_cost[i],
|
||||
cm->fc->ncobmc_mode_prob[i], av1_ncobmc_mode_tree);
|
||||
}
|
||||
#endif
|
||||
#endif // CONFIG_MOTION_VAR || CONFIG_WARPED_MOTION
|
||||
}
|
||||
}
|
||||
|
|
@ -648,7 +669,7 @@ static void get_entropy_contexts_plane(
|
|||
for (i = 0; i < num_4x4_h; i += 8)
|
||||
t_left[i] = !!*(const uint64_t *)&left[i];
|
||||
break;
|
||||
#if CONFIG_EXT_TX && CONFIG_RECT_TX && CONFIG_RECT_TX_EXT
|
||||
#if CONFIG_RECT_TX_EXT && (CONFIG_EXT_TX || CONFIG_VAR_TX)
|
||||
case TX_4X16:
|
||||
for (i = 0; i < num_4x4_w; i += 2)
|
||||
t_above[i] = !!*(const uint16_t *)&above[i];
|
||||
|
|
@ -675,7 +696,7 @@ static void get_entropy_contexts_plane(
|
|||
for (i = 0; i < num_4x4_h; i += 4)
|
||||
t_left[i] = !!*(const uint32_t *)&left[i];
|
||||
break;
|
||||
#endif // CONFIG_EXT_TX && CONFIG_RECT_TX && CONFIG_RECT_TX_EXT
|
||||
#endif
|
||||
|
||||
default: assert(0 && "Invalid transform size."); break;
|
||||
}
|
||||
|
|
@ -749,7 +770,7 @@ static void get_entropy_contexts_plane(
|
|||
for (i = 0; i < num_4x4_h; i += 4)
|
||||
t_left[i] = !!*(const uint32_t *)&left[i];
|
||||
break;
|
||||
#if CONFIG_EXT_TX && CONFIG_RECT_TX && CONFIG_RECT_TX_EXT
|
||||
#if CONFIG_RECT_TX_EXT && (CONFIG_EXT_TX || CONFIG_VAR_TX)
|
||||
case TX_4X16:
|
||||
memcpy(t_above, above, sizeof(ENTROPY_CONTEXT) * num_4x4_w);
|
||||
for (i = 0; i < num_4x4_h; i += 4)
|
||||
|
|
@ -772,7 +793,7 @@ static void get_entropy_contexts_plane(
|
|||
for (i = 0; i < num_4x4_h; i += 2)
|
||||
t_left[i] = !!*(const uint16_t *)&left[i];
|
||||
break;
|
||||
#endif // CONFIG_EXT_TX && CONFIG_RECT_TX && CONFIG_RECT_TX_EXT
|
||||
#endif
|
||||
default: assert(0 && "Invalid transform size."); break;
|
||||
}
|
||||
}
|
||||
|
|
@ -781,7 +802,7 @@ void av1_get_entropy_contexts(BLOCK_SIZE bsize, TX_SIZE tx_size,
|
|||
const struct macroblockd_plane *pd,
|
||||
ENTROPY_CONTEXT t_above[2 * MAX_MIB_SIZE],
|
||||
ENTROPY_CONTEXT t_left[2 * MAX_MIB_SIZE]) {
|
||||
#if CONFIG_CB4X4 && !CONFIG_CHROMA_2X2
|
||||
#if CONFIG_CHROMA_SUB8X8
|
||||
const BLOCK_SIZE plane_bsize =
|
||||
AOMMAX(BLOCK_4X4, get_plane_block_size(bsize, pd));
|
||||
#else
|
||||
|
|
@ -983,6 +1004,54 @@ void av1_set_rd_speed_thresholds(AV1_COMP *cpi) {
|
|||
|
||||
#if CONFIG_EXT_INTER
|
||||
|
||||
#if CONFIG_COMPOUND_SINGLEREF
|
||||
rd->thresh_mult[THR_SR_NEAREST_NEARMV] += 1200;
|
||||
#if CONFIG_EXT_REFS
|
||||
rd->thresh_mult[THR_SR_NEAREST_NEARL2] += 1200;
|
||||
rd->thresh_mult[THR_SR_NEAREST_NEARL3] += 1200;
|
||||
rd->thresh_mult[THR_SR_NEAREST_NEARB] += 1200;
|
||||
#endif // CONFIG_EXT_REFS
|
||||
rd->thresh_mult[THR_SR_NEAREST_NEARA] += 1200;
|
||||
rd->thresh_mult[THR_SR_NEAREST_NEARG] += 1200;
|
||||
|
||||
/*
|
||||
rd->thresh_mult[THR_SR_NEAREST_NEWMV] += 1200;
|
||||
#if CONFIG_EXT_REFS
|
||||
rd->thresh_mult[THR_SR_NEAREST_NEWL2] += 1200;
|
||||
rd->thresh_mult[THR_SR_NEAREST_NEWL3] += 1200;
|
||||
rd->thresh_mult[THR_SR_NEAREST_NEWB] += 1200;
|
||||
#endif // CONFIG_EXT_REFS
|
||||
rd->thresh_mult[THR_SR_NEAREST_NEWA] += 1200;
|
||||
rd->thresh_mult[THR_SR_NEAREST_NEWG] += 1200;*/
|
||||
|
||||
rd->thresh_mult[THR_SR_NEAR_NEWMV] += 1500;
|
||||
#if CONFIG_EXT_REFS
|
||||
rd->thresh_mult[THR_SR_NEAR_NEWL2] += 1500;
|
||||
rd->thresh_mult[THR_SR_NEAR_NEWL3] += 1500;
|
||||
rd->thresh_mult[THR_SR_NEAR_NEWB] += 1500;
|
||||
#endif // CONFIG_EXT_REFS
|
||||
rd->thresh_mult[THR_SR_NEAR_NEWA] += 1500;
|
||||
rd->thresh_mult[THR_SR_NEAR_NEWG] += 1500;
|
||||
|
||||
rd->thresh_mult[THR_SR_ZERO_NEWMV] += 2000;
|
||||
#if CONFIG_EXT_REFS
|
||||
rd->thresh_mult[THR_SR_ZERO_NEWL2] += 2000;
|
||||
rd->thresh_mult[THR_SR_ZERO_NEWL3] += 2000;
|
||||
rd->thresh_mult[THR_SR_ZERO_NEWB] += 2000;
|
||||
#endif // CONFIG_EXT_REFS
|
||||
rd->thresh_mult[THR_SR_ZERO_NEWA] += 2000;
|
||||
rd->thresh_mult[THR_SR_ZERO_NEWG] += 2000;
|
||||
|
||||
rd->thresh_mult[THR_SR_NEW_NEWMV] += 1700;
|
||||
#if CONFIG_EXT_REFS
|
||||
rd->thresh_mult[THR_SR_NEW_NEWL2] += 1700;
|
||||
rd->thresh_mult[THR_SR_NEW_NEWL3] += 1700;
|
||||
rd->thresh_mult[THR_SR_NEW_NEWB] += 1700;
|
||||
#endif // CONFIG_EXT_REFS
|
||||
rd->thresh_mult[THR_SR_NEW_NEWA] += 1700;
|
||||
rd->thresh_mult[THR_SR_NEW_NEWG] += 1700;
|
||||
#endif // CONFIG_COMPOUND_SINGLEREF
|
||||
|
||||
rd->thresh_mult[THR_COMP_NEAREST_NEARESTLA] += 1000;
|
||||
#if CONFIG_EXT_REFS
|
||||
rd->thresh_mult[THR_COMP_NEAREST_NEARESTL2A] += 1000;
|
||||
|
|
@ -994,6 +1063,13 @@ void av1_set_rd_speed_thresholds(AV1_COMP *cpi) {
|
|||
rd->thresh_mult[THR_COMP_NEAREST_NEARESTL2B] += 1000;
|
||||
rd->thresh_mult[THR_COMP_NEAREST_NEARESTL3B] += 1000;
|
||||
rd->thresh_mult[THR_COMP_NEAREST_NEARESTGB] += 1000;
|
||||
|
||||
#if CONFIG_EXT_COMP_REFS
|
||||
rd->thresh_mult[THR_COMP_NEAREST_NEARESTLL2] += 1000;
|
||||
rd->thresh_mult[THR_COMP_NEAREST_NEARESTLL3] += 1000;
|
||||
rd->thresh_mult[THR_COMP_NEAREST_NEARESTLG] += 1000;
|
||||
rd->thresh_mult[THR_COMP_NEAREST_NEARESTBA] += 1000;
|
||||
#endif // CONFIG_EXT_COMP_REFS
|
||||
#endif // CONFIG_EXT_REFS
|
||||
|
||||
#else // CONFIG_EXT_INTER
|
||||
|
|
@ -1009,6 +1085,12 @@ void av1_set_rd_speed_thresholds(AV1_COMP *cpi) {
|
|||
rd->thresh_mult[THR_COMP_NEARESTL2B] += 1000;
|
||||
rd->thresh_mult[THR_COMP_NEARESTL3B] += 1000;
|
||||
rd->thresh_mult[THR_COMP_NEARESTGB] += 1000;
|
||||
#if CONFIG_EXT_COMP_REFS
|
||||
rd->thresh_mult[THR_COMP_NEARESTLL2] += 1000;
|
||||
rd->thresh_mult[THR_COMP_NEARESTLL3] += 1000;
|
||||
rd->thresh_mult[THR_COMP_NEARESTLG] += 1000;
|
||||
rd->thresh_mult[THR_COMP_NEARESTBA] += 1000;
|
||||
#endif // CONFIG_EXT_COMP_REFS
|
||||
#endif // CONFIG_EXT_REFS
|
||||
|
||||
#endif // CONFIG_EXT_INTER
|
||||
|
|
@ -1081,6 +1163,40 @@ void av1_set_rd_speed_thresholds(AV1_COMP *cpi) {
|
|||
rd->thresh_mult[THR_COMP_NEW_NEARGB] += 1700;
|
||||
rd->thresh_mult[THR_COMP_NEW_NEWGB] += 2000;
|
||||
rd->thresh_mult[THR_COMP_ZERO_ZEROGB] += 2500;
|
||||
|
||||
#if CONFIG_EXT_COMP_REFS
|
||||
rd->thresh_mult[THR_COMP_NEAR_NEARLL2] += 1200;
|
||||
rd->thresh_mult[THR_COMP_NEAREST_NEWLL2] += 1500;
|
||||
rd->thresh_mult[THR_COMP_NEW_NEARESTLL2] += 1500;
|
||||
rd->thresh_mult[THR_COMP_NEAR_NEWLL2] += 1700;
|
||||
rd->thresh_mult[THR_COMP_NEW_NEARLL2] += 1700;
|
||||
rd->thresh_mult[THR_COMP_NEW_NEWLL2] += 2000;
|
||||
rd->thresh_mult[THR_COMP_ZERO_ZEROLL2] += 2500;
|
||||
|
||||
rd->thresh_mult[THR_COMP_NEAR_NEARLL3] += 1200;
|
||||
rd->thresh_mult[THR_COMP_NEAREST_NEWLL3] += 1500;
|
||||
rd->thresh_mult[THR_COMP_NEW_NEARESTLL3] += 1500;
|
||||
rd->thresh_mult[THR_COMP_NEAR_NEWLL3] += 1700;
|
||||
rd->thresh_mult[THR_COMP_NEW_NEARLL3] += 1700;
|
||||
rd->thresh_mult[THR_COMP_NEW_NEWLL3] += 2000;
|
||||
rd->thresh_mult[THR_COMP_ZERO_ZEROLL3] += 2500;
|
||||
|
||||
rd->thresh_mult[THR_COMP_NEAR_NEARLG] += 1200;
|
||||
rd->thresh_mult[THR_COMP_NEAREST_NEWLG] += 1500;
|
||||
rd->thresh_mult[THR_COMP_NEW_NEARESTLG] += 1500;
|
||||
rd->thresh_mult[THR_COMP_NEAR_NEWLG] += 1700;
|
||||
rd->thresh_mult[THR_COMP_NEW_NEARLG] += 1700;
|
||||
rd->thresh_mult[THR_COMP_NEW_NEWLG] += 2000;
|
||||
rd->thresh_mult[THR_COMP_ZERO_ZEROLG] += 2500;
|
||||
|
||||
rd->thresh_mult[THR_COMP_NEAR_NEARBA] += 1200;
|
||||
rd->thresh_mult[THR_COMP_NEAREST_NEWBA] += 1500;
|
||||
rd->thresh_mult[THR_COMP_NEW_NEARESTBA] += 1500;
|
||||
rd->thresh_mult[THR_COMP_NEAR_NEWBA] += 1700;
|
||||
rd->thresh_mult[THR_COMP_NEW_NEARBA] += 1700;
|
||||
rd->thresh_mult[THR_COMP_NEW_NEWBA] += 2000;
|
||||
rd->thresh_mult[THR_COMP_ZERO_ZEROBA] += 2500;
|
||||
#endif // CONFIG_EXT_COMP_REFS
|
||||
#endif // CONFIG_EXT_REFS
|
||||
|
||||
#else // CONFIG_EXT_INTER
|
||||
|
|
@ -1105,6 +1221,17 @@ void av1_set_rd_speed_thresholds(AV1_COMP *cpi) {
|
|||
rd->thresh_mult[THR_COMP_NEWL3B] += 2000;
|
||||
rd->thresh_mult[THR_COMP_NEARGB] += 1500;
|
||||
rd->thresh_mult[THR_COMP_NEWGB] += 2000;
|
||||
|
||||
#if CONFIG_EXT_COMP_REFS
|
||||
rd->thresh_mult[THR_COMP_NEARLL2] += 1500;
|
||||
rd->thresh_mult[THR_COMP_NEWLL2] += 2000;
|
||||
rd->thresh_mult[THR_COMP_NEARLL3] += 1500;
|
||||
rd->thresh_mult[THR_COMP_NEWLL3] += 2000;
|
||||
rd->thresh_mult[THR_COMP_NEARLG] += 1500;
|
||||
rd->thresh_mult[THR_COMP_NEWLG] += 2000;
|
||||
rd->thresh_mult[THR_COMP_NEARBA] += 1500;
|
||||
rd->thresh_mult[THR_COMP_NEWBA] += 2000;
|
||||
#endif // CONFIG_EXT_COMP_REFS
|
||||
#endif // CONFIG_EXT_REFS
|
||||
|
||||
rd->thresh_mult[THR_COMP_ZEROLA] += 2500;
|
||||
|
|
@ -1119,6 +1246,13 @@ void av1_set_rd_speed_thresholds(AV1_COMP *cpi) {
|
|||
rd->thresh_mult[THR_COMP_ZEROL2B] += 2500;
|
||||
rd->thresh_mult[THR_COMP_ZEROL3B] += 2500;
|
||||
rd->thresh_mult[THR_COMP_ZEROGB] += 2500;
|
||||
|
||||
#if CONFIG_EXT_COMP_REFS
|
||||
rd->thresh_mult[THR_COMP_ZEROLL2] += 2500;
|
||||
rd->thresh_mult[THR_COMP_ZEROLL3] += 2500;
|
||||
rd->thresh_mult[THR_COMP_ZEROLG] += 2500;
|
||||
rd->thresh_mult[THR_COMP_ZEROBA] += 2500;
|
||||
#endif // CONFIG_EXT_COMP_REFS
|
||||
#endif // CONFIG_EXT_REFS
|
||||
|
||||
#endif // CONFIG_EXT_INTER
|
||||
|
|
|
|||
150
third_party/aom/av1/encoder/rd.h
vendored
150
third_party/aom/av1/encoder/rd.h
vendored
|
|
@ -30,12 +30,13 @@ extern "C" {
|
|||
#define RDDIV_BITS 7
|
||||
#define RD_EPB_SHIFT 6
|
||||
|
||||
#define RDCOST(RM, DM, R, D) \
|
||||
(ROUND_POWER_OF_TWO(((int64_t)R) * (RM), AV1_PROB_COST_SHIFT) + (D << DM))
|
||||
#define RDCOST(RM, R, D) \
|
||||
(ROUND_POWER_OF_TWO(((int64_t)R) * (RM), AV1_PROB_COST_SHIFT) + \
|
||||
(D << RDDIV_BITS))
|
||||
|
||||
#define RDCOST_DBL(RM, DM, R, D) \
|
||||
#define RDCOST_DBL(RM, R, D) \
|
||||
(((((double)(R)) * (RM)) / (double)(1 << AV1_PROB_COST_SHIFT)) + \
|
||||
((double)(D) * (1 << (DM))))
|
||||
((double)(D) * (1 << RDDIV_BITS)))
|
||||
|
||||
#define QIDX_SKIP_THRESH 115
|
||||
|
||||
|
|
@ -96,6 +97,54 @@ typedef enum {
|
|||
|
||||
#if CONFIG_EXT_INTER
|
||||
|
||||
#if CONFIG_COMPOUND_SINGLEREF
|
||||
THR_SR_NEAREST_NEARMV,
|
||||
#if CONFIG_EXT_REFS
|
||||
THR_SR_NEAREST_NEARL2,
|
||||
THR_SR_NEAREST_NEARL3,
|
||||
THR_SR_NEAREST_NEARB,
|
||||
#endif // CONFIG_EXT_REFS
|
||||
THR_SR_NEAREST_NEARG,
|
||||
THR_SR_NEAREST_NEARA,
|
||||
|
||||
/*
|
||||
THR_SR_NEAREST_NEWMV,
|
||||
#if CONFIG_EXT_REFS
|
||||
THR_SR_NEAREST_NEWL2,
|
||||
THR_SR_NEAREST_NEWL3,
|
||||
THR_SR_NEAREST_NEWB,
|
||||
#endif // CONFIG_EXT_REFS
|
||||
THR_SR_NEAREST_NEWG,
|
||||
THR_SR_NEAREST_NEWA,*/
|
||||
|
||||
THR_SR_NEAR_NEWMV,
|
||||
#if CONFIG_EXT_REFS
|
||||
THR_SR_NEAR_NEWL2,
|
||||
THR_SR_NEAR_NEWL3,
|
||||
THR_SR_NEAR_NEWB,
|
||||
#endif // CONFIG_EXT_REFS
|
||||
THR_SR_NEAR_NEWG,
|
||||
THR_SR_NEAR_NEWA,
|
||||
|
||||
THR_SR_ZERO_NEWMV,
|
||||
#if CONFIG_EXT_REFS
|
||||
THR_SR_ZERO_NEWL2,
|
||||
THR_SR_ZERO_NEWL3,
|
||||
THR_SR_ZERO_NEWB,
|
||||
#endif // CONFIG_EXT_REFS
|
||||
THR_SR_ZERO_NEWG,
|
||||
THR_SR_ZERO_NEWA,
|
||||
|
||||
THR_SR_NEW_NEWMV,
|
||||
#if CONFIG_EXT_REFS
|
||||
THR_SR_NEW_NEWL2,
|
||||
THR_SR_NEW_NEWL3,
|
||||
THR_SR_NEW_NEWB,
|
||||
#endif // CONFIG_EXT_REFS
|
||||
THR_SR_NEW_NEWG,
|
||||
THR_SR_NEW_NEWA,
|
||||
#endif // CONFIG_COMPOUND_SINGLEREF
|
||||
|
||||
THR_COMP_NEAREST_NEARESTLA,
|
||||
#if CONFIG_EXT_REFS
|
||||
THR_COMP_NEAREST_NEARESTL2A,
|
||||
|
|
@ -107,6 +156,12 @@ typedef enum {
|
|||
THR_COMP_NEAREST_NEARESTL2B,
|
||||
THR_COMP_NEAREST_NEARESTL3B,
|
||||
THR_COMP_NEAREST_NEARESTGB,
|
||||
#if CONFIG_EXT_COMP_REFS
|
||||
THR_COMP_NEAREST_NEARESTLL2,
|
||||
THR_COMP_NEAREST_NEARESTLL3,
|
||||
THR_COMP_NEAREST_NEARESTLG,
|
||||
THR_COMP_NEAREST_NEARESTBA,
|
||||
#endif // CONFIG_EXT_COMP_REFS
|
||||
#endif // CONFIG_EXT_REFS
|
||||
|
||||
#else // CONFIG_EXT_INTER
|
||||
|
|
@ -122,6 +177,12 @@ typedef enum {
|
|||
THR_COMP_NEARESTL2B,
|
||||
THR_COMP_NEARESTL3B,
|
||||
THR_COMP_NEARESTGB,
|
||||
#if CONFIG_EXT_COMP_REFS
|
||||
THR_COMP_NEARESTLL2,
|
||||
THR_COMP_NEARESTLL3,
|
||||
THR_COMP_NEARESTLG,
|
||||
THR_COMP_NEARESTBA,
|
||||
#endif // CONFIG_EXT_COMP_REFS
|
||||
#endif // CONFIG_EXT_REFS
|
||||
|
||||
#endif // CONFIG_EXT_INTER
|
||||
|
|
@ -138,8 +199,6 @@ typedef enum {
|
|||
|
||||
#if CONFIG_EXT_INTER
|
||||
|
||||
THR_COMP_NEAR_NEARESTLA,
|
||||
THR_COMP_NEAREST_NEARLA,
|
||||
THR_COMP_NEAR_NEARLA,
|
||||
THR_COMP_NEW_NEARESTLA,
|
||||
THR_COMP_NEAREST_NEWLA,
|
||||
|
|
@ -149,8 +208,6 @@ typedef enum {
|
|||
THR_COMP_ZERO_ZEROLA,
|
||||
|
||||
#if CONFIG_EXT_REFS
|
||||
THR_COMP_NEAR_NEARESTL2A,
|
||||
THR_COMP_NEAREST_NEARL2A,
|
||||
THR_COMP_NEAR_NEARL2A,
|
||||
THR_COMP_NEW_NEARESTL2A,
|
||||
THR_COMP_NEAREST_NEWL2A,
|
||||
|
|
@ -159,8 +216,6 @@ typedef enum {
|
|||
THR_COMP_NEW_NEWL2A,
|
||||
THR_COMP_ZERO_ZEROL2A,
|
||||
|
||||
THR_COMP_NEAR_NEARESTL3A,
|
||||
THR_COMP_NEAREST_NEARL3A,
|
||||
THR_COMP_NEAR_NEARL3A,
|
||||
THR_COMP_NEW_NEARESTL3A,
|
||||
THR_COMP_NEAREST_NEWL3A,
|
||||
|
|
@ -170,8 +225,6 @@ typedef enum {
|
|||
THR_COMP_ZERO_ZEROL3A,
|
||||
#endif // CONFIG_EXT_REFS
|
||||
|
||||
THR_COMP_NEAR_NEARESTGA,
|
||||
THR_COMP_NEAREST_NEARGA,
|
||||
THR_COMP_NEAR_NEARGA,
|
||||
THR_COMP_NEW_NEARESTGA,
|
||||
THR_COMP_NEAREST_NEWGA,
|
||||
|
|
@ -181,8 +234,6 @@ typedef enum {
|
|||
THR_COMP_ZERO_ZEROGA,
|
||||
|
||||
#if CONFIG_EXT_REFS
|
||||
THR_COMP_NEAR_NEARESTLB,
|
||||
THR_COMP_NEAREST_NEARLB,
|
||||
THR_COMP_NEAR_NEARLB,
|
||||
THR_COMP_NEW_NEARESTLB,
|
||||
THR_COMP_NEAREST_NEWLB,
|
||||
|
|
@ -191,8 +242,6 @@ typedef enum {
|
|||
THR_COMP_NEW_NEWLB,
|
||||
THR_COMP_ZERO_ZEROLB,
|
||||
|
||||
THR_COMP_NEAR_NEARESTL2B,
|
||||
THR_COMP_NEAREST_NEARL2B,
|
||||
THR_COMP_NEAR_NEARL2B,
|
||||
THR_COMP_NEW_NEARESTL2B,
|
||||
THR_COMP_NEAREST_NEWL2B,
|
||||
|
|
@ -201,8 +250,6 @@ typedef enum {
|
|||
THR_COMP_NEW_NEWL2B,
|
||||
THR_COMP_ZERO_ZEROL2B,
|
||||
|
||||
THR_COMP_NEAR_NEARESTL3B,
|
||||
THR_COMP_NEAREST_NEARL3B,
|
||||
THR_COMP_NEAR_NEARL3B,
|
||||
THR_COMP_NEW_NEARESTL3B,
|
||||
THR_COMP_NEAREST_NEWL3B,
|
||||
|
|
@ -211,8 +258,6 @@ typedef enum {
|
|||
THR_COMP_NEW_NEWL3B,
|
||||
THR_COMP_ZERO_ZEROL3B,
|
||||
|
||||
THR_COMP_NEAR_NEARESTGB,
|
||||
THR_COMP_NEAREST_NEARGB,
|
||||
THR_COMP_NEAR_NEARGB,
|
||||
THR_COMP_NEW_NEARESTGB,
|
||||
THR_COMP_NEAREST_NEWGB,
|
||||
|
|
@ -220,6 +265,40 @@ typedef enum {
|
|||
THR_COMP_NEAR_NEWGB,
|
||||
THR_COMP_NEW_NEWGB,
|
||||
THR_COMP_ZERO_ZEROGB,
|
||||
|
||||
#if CONFIG_EXT_COMP_REFS
|
||||
THR_COMP_NEAR_NEARLL2,
|
||||
THR_COMP_NEW_NEARESTLL2,
|
||||
THR_COMP_NEAREST_NEWLL2,
|
||||
THR_COMP_NEW_NEARLL2,
|
||||
THR_COMP_NEAR_NEWLL2,
|
||||
THR_COMP_NEW_NEWLL2,
|
||||
THR_COMP_ZERO_ZEROLL2,
|
||||
|
||||
THR_COMP_NEAR_NEARLL3,
|
||||
THR_COMP_NEW_NEARESTLL3,
|
||||
THR_COMP_NEAREST_NEWLL3,
|
||||
THR_COMP_NEW_NEARLL3,
|
||||
THR_COMP_NEAR_NEWLL3,
|
||||
THR_COMP_NEW_NEWLL3,
|
||||
THR_COMP_ZERO_ZEROLL3,
|
||||
|
||||
THR_COMP_NEAR_NEARLG,
|
||||
THR_COMP_NEW_NEARESTLG,
|
||||
THR_COMP_NEAREST_NEWLG,
|
||||
THR_COMP_NEW_NEARLG,
|
||||
THR_COMP_NEAR_NEWLG,
|
||||
THR_COMP_NEW_NEWLG,
|
||||
THR_COMP_ZERO_ZEROLG,
|
||||
|
||||
THR_COMP_NEAR_NEARBA,
|
||||
THR_COMP_NEW_NEARESTBA,
|
||||
THR_COMP_NEAREST_NEWBA,
|
||||
THR_COMP_NEW_NEARBA,
|
||||
THR_COMP_NEAR_NEWBA,
|
||||
THR_COMP_NEW_NEWBA,
|
||||
THR_COMP_ZERO_ZEROBA,
|
||||
#endif // CONFIG_EXT_COMP_REFS
|
||||
#endif // CONFIG_EXT_REFS
|
||||
|
||||
#else // CONFIG_EXT_INTER
|
||||
|
|
@ -244,6 +323,17 @@ typedef enum {
|
|||
THR_COMP_NEWL3B,
|
||||
THR_COMP_NEARGB,
|
||||
THR_COMP_NEWGB,
|
||||
|
||||
#if CONFIG_EXT_COMP_REFS
|
||||
THR_COMP_NEARLL2,
|
||||
THR_COMP_NEWLL2,
|
||||
THR_COMP_NEARLL3,
|
||||
THR_COMP_NEWLL3,
|
||||
THR_COMP_NEARLG,
|
||||
THR_COMP_NEWLG,
|
||||
THR_COMP_NEARBA,
|
||||
THR_COMP_NEWBA,
|
||||
#endif // CONFIG_EXT_COMP_REFS
|
||||
#endif // CONFIG_EXT_REFS
|
||||
|
||||
THR_COMP_ZEROLA,
|
||||
|
|
@ -258,6 +348,13 @@ typedef enum {
|
|||
THR_COMP_ZEROL2B,
|
||||
THR_COMP_ZEROL3B,
|
||||
THR_COMP_ZEROGB,
|
||||
|
||||
#if CONFIG_EXT_COMP_REFS
|
||||
THR_COMP_ZEROLL2,
|
||||
THR_COMP_ZEROLL3,
|
||||
THR_COMP_ZEROLG,
|
||||
THR_COMP_ZEROBA,
|
||||
#endif // CONFIG_EXT_COMP_REFS
|
||||
#endif // CONFIG_EXT_REFS
|
||||
|
||||
#endif // CONFIG_EXT_INTER
|
||||
|
|
@ -344,12 +441,11 @@ typedef struct RD_OPT {
|
|||
int thresh_mult[MAX_MODES];
|
||||
int thresh_mult_sub8x8[MAX_REFS];
|
||||
|
||||
int threshes[MAX_SEGMENTS][BLOCK_SIZES][MAX_MODES];
|
||||
int threshes[MAX_SEGMENTS][BLOCK_SIZES_ALL][MAX_MODES];
|
||||
|
||||
int64_t prediction_type_threshes[TOTAL_REFS_PER_FRAME][REFERENCE_MODES];
|
||||
|
||||
int RDMULT;
|
||||
int RDDIV;
|
||||
} RD_OPT;
|
||||
|
||||
static INLINE void av1_init_rd_stats(RD_STATS *rd_stats) {
|
||||
|
|
@ -361,7 +457,9 @@ static INLINE void av1_init_rd_stats(RD_STATS *rd_stats) {
|
|||
rd_stats->rdcost = 0;
|
||||
rd_stats->sse = 0;
|
||||
rd_stats->skip = 1;
|
||||
#if CONFIG_DAALA_DIST && CONFIG_CB4X4
|
||||
rd_stats->zero_rate = 0;
|
||||
rd_stats->ref_rdcost = INT64_MAX;
|
||||
#if CONFIG_DIST_8X8 && CONFIG_CB4X4
|
||||
rd_stats->dist_y = 0;
|
||||
#endif
|
||||
#if CONFIG_RD_DEBUG
|
||||
|
|
@ -388,7 +486,9 @@ static INLINE void av1_invalid_rd_stats(RD_STATS *rd_stats) {
|
|||
rd_stats->rdcost = INT64_MAX;
|
||||
rd_stats->sse = INT64_MAX;
|
||||
rd_stats->skip = 0;
|
||||
#if CONFIG_DAALA_DIST && CONFIG_CB4X4
|
||||
rd_stats->zero_rate = 0;
|
||||
rd_stats->ref_rdcost = INT64_MAX;
|
||||
#if CONFIG_DIST_8X8 && CONFIG_CB4X4
|
||||
rd_stats->dist_y = INT64_MAX;
|
||||
#endif
|
||||
#if CONFIG_RD_DEBUG
|
||||
|
|
@ -415,7 +515,7 @@ static INLINE void av1_merge_rd_stats(RD_STATS *rd_stats_dst,
|
|||
rd_stats_dst->dist += rd_stats_src->dist;
|
||||
rd_stats_dst->sse += rd_stats_src->sse;
|
||||
rd_stats_dst->skip &= rd_stats_src->skip;
|
||||
#if CONFIG_DAALA_DIST && CONFIG_CB4X4
|
||||
#if CONFIG_DIST_8X8 && CONFIG_CB4X4
|
||||
rd_stats_dst->dist_y += rd_stats_src->dist_y;
|
||||
#endif
|
||||
#if CONFIG_RD_DEBUG
|
||||
|
|
|
|||
3437
third_party/aom/av1/encoder/rdopt.c
vendored
3437
third_party/aom/av1/encoder/rdopt.c
vendored
File diff suppressed because it is too large
Load diff
25
third_party/aom/av1/encoder/rdopt.h
vendored
25
third_party/aom/av1/encoder/rdopt.h
vendored
|
|
@ -57,22 +57,33 @@ typedef enum OUTPUT_STATUS {
|
|||
OUTPUT_HAS_DECODED_PIXELS
|
||||
} OUTPUT_STATUS;
|
||||
|
||||
#if CONFIG_PALETTE || CONFIG_INTRABC
|
||||
// Returns the number of colors in 'src'.
|
||||
int av1_count_colors(const uint8_t *src, int stride, int rows, int cols);
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
// Same as av1_count_colors(), but for high-bitdepth mode.
|
||||
int av1_count_colors_highbd(const uint8_t *src8, int stride, int rows, int cols,
|
||||
int bit_depth);
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
#endif // CONFIG_PALETTE || CONFIG_INTRABC
|
||||
|
||||
void av1_dist_block(const AV1_COMP *cpi, MACROBLOCK *x, int plane,
|
||||
BLOCK_SIZE plane_bsize, int block, int blk_row, int blk_col,
|
||||
TX_SIZE tx_size, int64_t *out_dist, int64_t *out_sse,
|
||||
OUTPUT_STATUS output_status);
|
||||
|
||||
#if CONFIG_DAALA_DIST
|
||||
int64_t av1_daala_dist(const uint8_t *src, int src_stride, const uint8_t *dst,
|
||||
int dst_stride, int bsw, int bsh, int qm,
|
||||
int use_activity_masking, int qindex);
|
||||
#if CONFIG_DIST_8X8
|
||||
int64_t av1_dist_8x8(const AV1_COMP *const cpi, const MACROBLOCKD *xd,
|
||||
const uint8_t *src, int src_stride, const uint8_t *dst,
|
||||
int dst_stride, const BLOCK_SIZE tx_bsize, int bsw,
|
||||
int bsh, int visible_w, int visible_h, int qindex);
|
||||
#endif
|
||||
|
||||
#if !CONFIG_PVQ || CONFIG_VAR_TX
|
||||
int av1_cost_coeffs(const AV1_COMP *const cpi, MACROBLOCK *x, int plane,
|
||||
int block, TX_SIZE tx_size, const SCAN_ORDER *scan_order,
|
||||
const ENTROPY_CONTEXT *a, const ENTROPY_CONTEXT *l,
|
||||
int use_fast_coef_costing);
|
||||
int blk_row, int blk_col, int block, TX_SIZE tx_size,
|
||||
const SCAN_ORDER *scan_order, const ENTROPY_CONTEXT *a,
|
||||
const ENTROPY_CONTEXT *l, int use_fast_coef_costing);
|
||||
#endif
|
||||
void av1_rd_pick_intra_mode_sb(const struct AV1_COMP *cpi, struct macroblock *x,
|
||||
struct RD_STATS *rd_cost, BLOCK_SIZE bsize,
|
||||
|
|
|
|||
14
third_party/aom/av1/encoder/segmentation.c
vendored
14
third_party/aom/av1/encoder/segmentation.c
vendored
|
|
@ -299,12 +299,8 @@ void av1_choose_segmap_coding_method(AV1_COMMON *cm, MACROBLOCKD *xd) {
|
|||
int no_pred_cost;
|
||||
int t_pred_cost = INT_MAX;
|
||||
|
||||
int i, tile_col, tile_row, mi_row, mi_col;
|
||||
#if CONFIG_TILE_GROUPS
|
||||
int tile_col, tile_row, mi_row, mi_col;
|
||||
const int probwt = cm->num_tg;
|
||||
#else
|
||||
const int probwt = 1;
|
||||
#endif
|
||||
|
||||
unsigned(*temporal_predictor_count)[2] = cm->counts.seg.pred;
|
||||
unsigned *no_pred_segcounts = cm->counts.seg.tree_total;
|
||||
|
|
@ -312,7 +308,9 @@ void av1_choose_segmap_coding_method(AV1_COMMON *cm, MACROBLOCKD *xd) {
|
|||
|
||||
aom_prob no_pred_tree[SEG_TREE_PROBS];
|
||||
aom_prob t_pred_tree[SEG_TREE_PROBS];
|
||||
#if !CONFIG_NEW_MULTISYMBOL
|
||||
aom_prob t_nopred_prob[PREDICTION_PROBS];
|
||||
#endif
|
||||
|
||||
(void)xd;
|
||||
|
||||
|
|
@ -327,7 +325,7 @@ void av1_choose_segmap_coding_method(AV1_COMMON *cm, MACROBLOCKD *xd) {
|
|||
for (tile_col = 0; tile_col < cm->tile_cols; tile_col++) {
|
||||
MODE_INFO **mi_ptr;
|
||||
av1_tile_set_col(&tile_info, cm, tile_col);
|
||||
#if CONFIG_TILE_GROUPS && CONFIG_DEPENDENT_HORZTILES
|
||||
#if CONFIG_DEPENDENT_HORZTILES
|
||||
av1_tile_set_tg_boundary(&tile_info, cm, tile_row, tile_col);
|
||||
#endif
|
||||
mi_ptr = cm->mi_grid_visible + tile_info.mi_row_start * cm->mi_stride +
|
||||
|
|
@ -357,8 +355,9 @@ void av1_choose_segmap_coding_method(AV1_COMMON *cm, MACROBLOCKD *xd) {
|
|||
calc_segtree_probs(t_unpred_seg_counts, t_pred_tree, segp->tree_probs,
|
||||
probwt);
|
||||
t_pred_cost = cost_segmap(t_unpred_seg_counts, t_pred_tree);
|
||||
|
||||
#if !CONFIG_NEW_MULTISYMBOL
|
||||
// Add in the cost of the signaling for each prediction context.
|
||||
int i;
|
||||
for (i = 0; i < PREDICTION_PROBS; i++) {
|
||||
const int count0 = temporal_predictor_count[i][0];
|
||||
const int count1 = temporal_predictor_count[i][1];
|
||||
|
|
@ -372,6 +371,7 @@ void av1_choose_segmap_coding_method(AV1_COMMON *cm, MACROBLOCKD *xd) {
|
|||
t_pred_cost += count0 * av1_cost_zero(t_nopred_prob[i]) +
|
||||
count1 * av1_cost_one(t_nopred_prob[i]);
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
// Now choose which coding method to use.
|
||||
|
|
|
|||
34
third_party/aom/av1/encoder/speed_features.c
vendored
34
third_party/aom/av1/encoder/speed_features.c
vendored
|
|
@ -35,7 +35,7 @@ static unsigned char good_quality_max_mesh_pct[MAX_MESH_SPEED + 1] = {
|
|||
// TODO(aconverse@google.com): These settings are pretty relaxed, tune them for
|
||||
// each speed setting
|
||||
static MESH_PATTERN intrabc_mesh_patterns[MAX_MESH_SPEED + 1][MAX_MESH_STEP] = {
|
||||
{ { 64, 1 }, { 64, 1 }, { 0, 0 }, { 0, 0 } },
|
||||
{ { 256, 1 }, { 256, 1 }, { 0, 0 }, { 0, 0 } },
|
||||
{ { 64, 1 }, { 64, 1 }, { 0, 0 }, { 0, 0 } },
|
||||
{ { 64, 1 }, { 64, 1 }, { 0, 0 }, { 0, 0 } },
|
||||
{ { 64, 4 }, { 16, 1 }, { 0, 0 }, { 0, 0 } },
|
||||
|
|
@ -171,12 +171,24 @@ static void set_good_speed_features_framesize_independent(AV1_COMP *cpi,
|
|||
sf->recode_loop = ALLOW_RECODE_KFARFGF;
|
||||
#if CONFIG_TX64X64
|
||||
sf->intra_y_mode_mask[TX_64X64] = INTRA_DC_H_V;
|
||||
#if CONFIG_CFL
|
||||
sf->intra_uv_mode_mask[TX_64X64] = UV_INTRA_DC_H_V;
|
||||
#else
|
||||
sf->intra_uv_mode_mask[TX_64X64] = INTRA_DC_H_V;
|
||||
#endif // CONFIG_CFL
|
||||
#endif // CONFIG_TX64X64
|
||||
sf->intra_y_mode_mask[TX_32X32] = INTRA_DC_H_V;
|
||||
#if CONFIG_CFL
|
||||
sf->intra_uv_mode_mask[TX_32X32] = UV_INTRA_DC_H_V;
|
||||
#else
|
||||
sf->intra_uv_mode_mask[TX_32X32] = INTRA_DC_H_V;
|
||||
#endif
|
||||
sf->intra_y_mode_mask[TX_16X16] = INTRA_DC_H_V;
|
||||
#if CONFIG_CFL
|
||||
sf->intra_uv_mode_mask[TX_16X16] = UV_INTRA_DC_H_V;
|
||||
#else
|
||||
sf->intra_uv_mode_mask[TX_16X16] = INTRA_DC_H_V;
|
||||
#endif
|
||||
|
||||
sf->tx_size_search_breakout = 1;
|
||||
sf->partition_search_breakout_rate_thr = 80;
|
||||
|
|
@ -199,7 +211,7 @@ static void set_good_speed_features_framesize_independent(AV1_COMP *cpi,
|
|||
: FLAG_SKIP_INTRA_DIRMISMATCH | FLAG_SKIP_INTRA_BESTINTER |
|
||||
FLAG_SKIP_COMP_BESTINTRA | FLAG_SKIP_INTRA_LOWVAR;
|
||||
sf->disable_filter_search_var_thresh = 100;
|
||||
sf->comp_inter_joint_search_thresh = BLOCK_SIZES;
|
||||
sf->comp_inter_joint_search_thresh = BLOCK_SIZES_ALL;
|
||||
sf->auto_min_max_partition_size = RELAXED_NEIGHBORING_MIN_MAX;
|
||||
sf->allow_partition_search_skip = 1;
|
||||
sf->use_upsampled_references = 0;
|
||||
|
|
@ -227,10 +239,18 @@ static void set_good_speed_features_framesize_independent(AV1_COMP *cpi,
|
|||
sf->mode_skip_start = 6;
|
||||
#if CONFIG_TX64X64
|
||||
sf->intra_y_mode_mask[TX_64X64] = INTRA_DC;
|
||||
#if CONFIG_CFL
|
||||
sf->intra_uv_mode_mask[TX_64X64] = UV_INTRA_DC;
|
||||
#else
|
||||
sf->intra_uv_mode_mask[TX_64X64] = INTRA_DC;
|
||||
#endif // CONFIG_CFL
|
||||
#endif // CONFIG_TX64X64
|
||||
sf->intra_y_mode_mask[TX_32X32] = INTRA_DC;
|
||||
#if CONFIG_CFL
|
||||
sf->intra_uv_mode_mask[TX_32X32] = UV_INTRA_DC;
|
||||
#else
|
||||
sf->intra_uv_mode_mask[TX_32X32] = INTRA_DC;
|
||||
#endif // CONFIG_CFL
|
||||
sf->adaptive_interp_filter_search = 1;
|
||||
}
|
||||
|
||||
|
|
@ -255,7 +275,11 @@ static void set_good_speed_features_framesize_independent(AV1_COMP *cpi,
|
|||
sf->disable_filter_search_var_thresh = 500;
|
||||
for (i = 0; i < TX_SIZES; ++i) {
|
||||
sf->intra_y_mode_mask[i] = INTRA_DC;
|
||||
#if CONFIG_CFL
|
||||
sf->intra_uv_mode_mask[i] = UV_INTRA_DC;
|
||||
#else
|
||||
sf->intra_uv_mode_mask[i] = INTRA_DC;
|
||||
#endif // CONFIG_CFL
|
||||
}
|
||||
sf->partition_search_breakout_rate_thr = 500;
|
||||
sf->mv.reduce_first_step_size = 1;
|
||||
|
|
@ -405,7 +429,11 @@ void av1_set_speed_features_framesize_independent(AV1_COMP *cpi) {
|
|||
|
||||
for (i = 0; i < TX_SIZES; i++) {
|
||||
sf->intra_y_mode_mask[i] = INTRA_ALL;
|
||||
#if CONFIG_CFL
|
||||
sf->intra_uv_mode_mask[i] = UV_INTRA_ALL;
|
||||
#else
|
||||
sf->intra_uv_mode_mask[i] = INTRA_ALL;
|
||||
#endif // CONFIG_CFL
|
||||
}
|
||||
sf->use_rd_breakout = 0;
|
||||
sf->lpf_pick = LPF_PICK_FROM_FULL_IMAGE;
|
||||
|
|
@ -413,7 +441,7 @@ void av1_set_speed_features_framesize_independent(AV1_COMP *cpi) {
|
|||
sf->use_fast_coef_costing = 0;
|
||||
sf->mode_skip_start = MAX_MODES; // Mode index at which mode skip mask set
|
||||
sf->schedule_mode_search = 0;
|
||||
for (i = 0; i < BLOCK_SIZES; ++i) sf->inter_mode_mask[i] = INTER_ALL;
|
||||
for (i = 0; i < BLOCK_SIZES_ALL; ++i) sf->inter_mode_mask[i] = INTER_ALL;
|
||||
sf->max_intra_bsize = BLOCK_LARGEST;
|
||||
sf->reuse_inter_pred_sby = 0;
|
||||
// This setting only takes effect when partition_search_type is set
|
||||
|
|
|
|||
31
third_party/aom/av1/encoder/speed_features.h
vendored
31
third_party/aom/av1/encoder/speed_features.h
vendored
|
|
@ -29,6 +29,24 @@ enum {
|
|||
#endif // CONFIG_SMOOTH_HV
|
||||
#endif // CONFIG_ALT_INTRA
|
||||
(1 << TM_PRED),
|
||||
#if CONFIG_CFL
|
||||
UV_INTRA_ALL = (1 << UV_DC_PRED) | (1 << UV_V_PRED) | (1 << UV_H_PRED) |
|
||||
(1 << UV_D45_PRED) | (1 << UV_D135_PRED) |
|
||||
(1 << UV_D117_PRED) | (1 << UV_D153_PRED) |
|
||||
(1 << UV_D207_PRED) | (1 << UV_D63_PRED) |
|
||||
#if CONFIG_ALT_INTRA
|
||||
(1 << UV_SMOOTH_PRED) |
|
||||
#if CONFIG_SMOOTH_HV
|
||||
(1 << UV_SMOOTH_V_PRED) | (1 << UV_SMOOTH_H_PRED) |
|
||||
#endif // CONFIG_SMOOTH_HV
|
||||
#endif // CONFIG_ALT_INTRA
|
||||
(1 << UV_TM_PRED),
|
||||
UV_INTRA_DC = (1 << UV_DC_PRED),
|
||||
UV_INTRA_DC_TM = (1 << UV_DC_PRED) | (1 << UV_TM_PRED),
|
||||
UV_INTRA_DC_H_V = (1 << UV_DC_PRED) | (1 << UV_V_PRED) | (1 << UV_H_PRED),
|
||||
UV_INTRA_DC_TM_H_V = (1 << UV_DC_PRED) | (1 << UV_TM_PRED) |
|
||||
(1 << UV_V_PRED) | (1 << UV_H_PRED),
|
||||
#endif // CONFIG_CFL
|
||||
INTRA_DC = (1 << DC_PRED),
|
||||
INTRA_DC_TM = (1 << DC_PRED) | (1 << TM_PRED),
|
||||
INTRA_DC_H_V = (1 << DC_PRED) | (1 << V_PRED) | (1 << H_PRED),
|
||||
|
|
@ -38,6 +56,11 @@ enum {
|
|||
|
||||
#if CONFIG_EXT_INTER
|
||||
enum {
|
||||
#if CONFIG_COMPOUND_SINGLEREF
|
||||
// TODO(zoeliu): To further consider following single ref comp modes:
|
||||
// SR_NEAREST_NEARMV, SR_NEAREST_NEWMV, SR_NEAR_NEWMV,
|
||||
// SR_ZERO_NEWMV, and SR_NEW_NEWMV.
|
||||
#endif // CONFIG_COMPOUND_SINGLEREF
|
||||
INTER_ALL = (1 << NEARESTMV) | (1 << NEARMV) | (1 << ZEROMV) | (1 << NEWMV) |
|
||||
(1 << NEAREST_NEARESTMV) | (1 << NEAR_NEARMV) | (1 << NEW_NEWMV) |
|
||||
(1 << NEAREST_NEWMV) | (1 << NEAR_NEWMV) | (1 << NEW_NEARMV) |
|
||||
|
|
@ -67,7 +90,7 @@ enum {
|
|||
(1 << NEW_NEARMV) | (1 << NEAR_NEWMV) |
|
||||
(1 << NEAR_NEARMV),
|
||||
};
|
||||
#else
|
||||
#else // !CONFIG_EXT_INTER
|
||||
enum {
|
||||
INTER_ALL = (1 << NEARESTMV) | (1 << NEARMV) | (1 << ZEROMV) | (1 << NEWMV),
|
||||
INTER_NEAREST = (1 << NEARESTMV),
|
||||
|
|
@ -399,10 +422,6 @@ typedef struct SPEED_FEATURES {
|
|||
int intra_y_mode_mask[TX_SIZES];
|
||||
int intra_uv_mode_mask[TX_SIZES];
|
||||
|
||||
// These bit masks allow you to enable or disable intra modes for each
|
||||
// prediction block size separately.
|
||||
int intra_y_mode_bsize_mask[BLOCK_SIZES];
|
||||
|
||||
// This variable enables an early break out of mode testing if the model for
|
||||
// rd built from the prediction signal indicates a value that's much
|
||||
// higher than the best rd we've seen so far.
|
||||
|
|
@ -417,7 +436,7 @@ typedef struct SPEED_FEATURES {
|
|||
|
||||
// A binary mask indicating if NEARESTMV, NEARMV, ZEROMV, NEWMV
|
||||
// modes are used in order from LSB to MSB for each BLOCK_SIZE.
|
||||
int inter_mode_mask[BLOCK_SIZES];
|
||||
int inter_mode_mask[BLOCK_SIZES_ALL];
|
||||
|
||||
// This feature controls whether we do the expensive context update and
|
||||
// calculation in the rd coefficient costing loop.
|
||||
|
|
|
|||
83
third_party/aom/av1/encoder/temporal_filter.c
vendored
83
third_party/aom/av1/encoder/temporal_filter.c
vendored
|
|
@ -41,7 +41,7 @@ static void temporal_filter_predictors_mb_c(
|
|||
enum mv_precision mv_precision_uv;
|
||||
int uv_stride;
|
||||
// TODO(angiebird): change plane setting accordingly
|
||||
ConvolveParams conv_params = get_conv_params(which_mv, 0);
|
||||
ConvolveParams conv_params = get_conv_params(which_mv, which_mv, 0);
|
||||
|
||||
#if USE_TEMPORALFILTER_12TAP
|
||||
#if CONFIG_DUAL_FILTER
|
||||
|
|
@ -413,10 +413,10 @@ static void temporal_filter_iterate_c(AV1_COMP *cpi,
|
|||
mbd->mi[0]->bmi[0].as_mv[0].as_mv.col, predictor, scale,
|
||||
mb_col * 16, mb_row * 16);
|
||||
|
||||
// Apply the filter (YUV)
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
if (mbd->cur_buf->flags & YV12_FLAG_HIGHBITDEPTH) {
|
||||
int adj_strength = strength + 2 * (mbd->bd - 8);
|
||||
// Apply the filter (YUV)
|
||||
av1_highbd_temporal_filter_apply(
|
||||
f->y_buffer + mb_y_offset, f->y_stride, predictor, 16, 16,
|
||||
adj_strength, filter_weight, accumulator, count);
|
||||
|
|
@ -429,7 +429,7 @@ static void temporal_filter_iterate_c(AV1_COMP *cpi,
|
|||
mb_uv_width, mb_uv_height, adj_strength, filter_weight,
|
||||
accumulator + 512, count + 512);
|
||||
} else {
|
||||
// Apply the filter (YUV)
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
av1_temporal_filter_apply_c(f->y_buffer + mb_y_offset, f->y_stride,
|
||||
predictor, 16, 16, strength,
|
||||
filter_weight, accumulator, count);
|
||||
|
|
@ -441,29 +441,17 @@ static void temporal_filter_iterate_c(AV1_COMP *cpi,
|
|||
f->v_buffer + mb_uv_offset, f->uv_stride, predictor + 512,
|
||||
mb_uv_width, mb_uv_height, strength, filter_weight,
|
||||
accumulator + 512, count + 512);
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
}
|
||||
#else
|
||||
// Apply the filter (YUV)
|
||||
av1_temporal_filter_apply_c(f->y_buffer + mb_y_offset, f->y_stride,
|
||||
predictor, 16, 16, strength,
|
||||
filter_weight, accumulator, count);
|
||||
av1_temporal_filter_apply_c(f->u_buffer + mb_uv_offset, f->uv_stride,
|
||||
predictor + 256, mb_uv_width,
|
||||
mb_uv_height, strength, filter_weight,
|
||||
accumulator + 256, count + 256);
|
||||
av1_temporal_filter_apply_c(f->v_buffer + mb_uv_offset, f->uv_stride,
|
||||
predictor + 512, mb_uv_width,
|
||||
mb_uv_height, strength, filter_weight,
|
||||
accumulator + 512, count + 512);
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
}
|
||||
}
|
||||
|
||||
// Normalize filter output to produce AltRef frame
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
if (mbd->cur_buf->flags & YV12_FLAG_HIGHBITDEPTH) {
|
||||
uint16_t *dst1_16;
|
||||
uint16_t *dst2_16;
|
||||
// Normalize filter output to produce AltRef frame
|
||||
dst1 = cpi->alt_ref_buffer.y_buffer;
|
||||
dst1_16 = CONVERT_TO_SHORTPTR(dst1);
|
||||
stride = cpi->alt_ref_buffer.y_stride;
|
||||
|
|
@ -505,7 +493,7 @@ static void temporal_filter_iterate_c(AV1_COMP *cpi,
|
|||
byte += stride - mb_uv_width;
|
||||
}
|
||||
} else {
|
||||
// Normalize filter output to produce AltRef frame
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
dst1 = cpi->alt_ref_buffer.y_buffer;
|
||||
stride = cpi->alt_ref_buffer.y_stride;
|
||||
byte = mb_y_offset;
|
||||
|
|
@ -541,43 +529,7 @@ static void temporal_filter_iterate_c(AV1_COMP *cpi,
|
|||
}
|
||||
byte += stride - mb_uv_width;
|
||||
}
|
||||
}
|
||||
#else
|
||||
// Normalize filter output to produce AltRef frame
|
||||
dst1 = cpi->alt_ref_buffer.y_buffer;
|
||||
stride = cpi->alt_ref_buffer.y_stride;
|
||||
byte = mb_y_offset;
|
||||
for (i = 0, k = 0; i < 16; i++) {
|
||||
for (j = 0; j < 16; j++, k++) {
|
||||
dst1[byte] =
|
||||
(uint8_t)OD_DIVU(accumulator[k] + (count[k] >> 1), count[k]);
|
||||
|
||||
// move to next pixel
|
||||
byte++;
|
||||
}
|
||||
byte += stride - 16;
|
||||
}
|
||||
|
||||
dst1 = cpi->alt_ref_buffer.u_buffer;
|
||||
dst2 = cpi->alt_ref_buffer.v_buffer;
|
||||
stride = cpi->alt_ref_buffer.uv_stride;
|
||||
byte = mb_uv_offset;
|
||||
for (i = 0, k = 256; i < mb_uv_height; i++) {
|
||||
for (j = 0; j < mb_uv_width; j++, k++) {
|
||||
int m = k + 256;
|
||||
|
||||
// U
|
||||
dst1[byte] =
|
||||
(uint8_t)OD_DIVU(accumulator[k] + (count[k] >> 1), count[k]);
|
||||
|
||||
// V
|
||||
dst2[byte] =
|
||||
(uint8_t)OD_DIVU(accumulator[m] + (count[m] >> 1), count[m]);
|
||||
|
||||
// move to next pixel
|
||||
byte++;
|
||||
}
|
||||
byte += stride - mb_uv_width;
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
}
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
mb_y_offset += 16;
|
||||
|
|
@ -650,7 +602,11 @@ static void adjust_arnr_filter(AV1_COMP *cpi, int distance, int group_boost,
|
|||
*arnr_strength = strength;
|
||||
}
|
||||
|
||||
void av1_temporal_filter(AV1_COMP *cpi, int distance) {
|
||||
void av1_temporal_filter(AV1_COMP *cpi,
|
||||
#if CONFIG_BGSPRITE
|
||||
YV12_BUFFER_CONFIG *bg,
|
||||
#endif // CONFIG_BGSPRITE
|
||||
int distance) {
|
||||
RATE_CONTROL *const rc = &cpi->rc;
|
||||
int frame;
|
||||
int frames_to_blur;
|
||||
|
|
@ -692,9 +648,18 @@ void av1_temporal_filter(AV1_COMP *cpi, int distance) {
|
|||
// Setup frame pointers, NULL indicates frame not included in filter.
|
||||
for (frame = 0; frame < frames_to_blur; ++frame) {
|
||||
const int which_buffer = start_frame - frame;
|
||||
struct lookahead_entry *buf =
|
||||
av1_lookahead_peek(cpi->lookahead, which_buffer);
|
||||
frames[frames_to_blur - 1 - frame] = &buf->img;
|
||||
#if CONFIG_BGSPRITE
|
||||
if (frame == frames_to_blur_backward && bg != NULL) {
|
||||
// Insert bg into frames at ARF index.
|
||||
frames[frames_to_blur - 1 - frame] = bg;
|
||||
} else {
|
||||
#endif // CONFIG_BGSPRITE
|
||||
struct lookahead_entry *buf =
|
||||
av1_lookahead_peek(cpi->lookahead, which_buffer);
|
||||
frames[frames_to_blur - 1 - frame] = &buf->img;
|
||||
#if CONFIG_BGSPRITE
|
||||
}
|
||||
#endif // CONFIG_BGSPRITE
|
||||
}
|
||||
|
||||
if (frames_to_blur > 0) {
|
||||
|
|
|
|||
|
|
@ -16,7 +16,11 @@
|
|||
extern "C" {
|
||||
#endif
|
||||
|
||||
void av1_temporal_filter(AV1_COMP *cpi, int distance);
|
||||
void av1_temporal_filter(AV1_COMP *cpi,
|
||||
#if CONFIG_BGSPRITE
|
||||
YV12_BUFFER_CONFIG *bg,
|
||||
#endif // CONFIG_BGSPRITE
|
||||
int distance);
|
||||
|
||||
#ifdef __cplusplus
|
||||
} // extern "C"
|
||||
|
|
|
|||
107
third_party/aom/av1/encoder/tokenize.c
vendored
107
third_party/aom/av1/encoder/tokenize.c
vendored
|
|
@ -277,12 +277,12 @@ static void cost_coeffs_b(int plane, int block, int blk_row, int blk_col,
|
|||
struct macroblock_plane *p = &x->plane[plane];
|
||||
struct macroblockd_plane *pd = &xd->plane[plane];
|
||||
const PLANE_TYPE type = pd->plane_type;
|
||||
const int ref = is_inter_block(mbmi);
|
||||
const TX_TYPE tx_type = get_tx_type(type, xd, block, tx_size);
|
||||
const SCAN_ORDER *const scan_order = get_scan(cm, tx_size, tx_type, ref);
|
||||
const int rate = av1_cost_coeffs(cpi, x, plane, block, tx_size, scan_order,
|
||||
pd->above_context + blk_col,
|
||||
pd->left_context + blk_row, 0);
|
||||
const TX_TYPE tx_type =
|
||||
av1_get_tx_type(type, xd, blk_row, blk_col, block, tx_size);
|
||||
const SCAN_ORDER *const scan_order = get_scan(cm, tx_size, tx_type, mbmi);
|
||||
const int rate = av1_cost_coeffs(
|
||||
cpi, x, plane, blk_row, blk_col, block, tx_size, scan_order,
|
||||
pd->above_context + blk_col, pd->left_context + blk_row, 0);
|
||||
args->this_rate += rate;
|
||||
(void)plane_bsize;
|
||||
av1_set_contexts(xd, pd, plane, tx_size, p->eobs[block] > 0, blk_col,
|
||||
|
|
@ -323,42 +323,48 @@ void av1_tokenize_palette_sb(const AV1_COMP *cpi,
|
|||
const struct ThreadData *const td, int plane,
|
||||
TOKENEXTRA **t, RUN_TYPE dry_run, BLOCK_SIZE bsize,
|
||||
int *rate) {
|
||||
assert(plane == 0 || plane == 1);
|
||||
const MACROBLOCK *const x = &td->mb;
|
||||
const MACROBLOCKD *const xd = &x->e_mbd;
|
||||
const MB_MODE_INFO *const mbmi = &xd->mi[0]->mbmi;
|
||||
const uint8_t *const color_map = xd->plane[plane].color_index_map;
|
||||
const PALETTE_MODE_INFO *const pmi = &mbmi->palette_mode_info;
|
||||
const int n = pmi->palette_size[plane];
|
||||
int i, j;
|
||||
int this_rate = 0;
|
||||
uint8_t color_order[PALETTE_MAX_SIZE];
|
||||
const aom_prob(
|
||||
*const probs)[PALETTE_COLOR_INDEX_CONTEXTS][PALETTE_COLORS - 1] =
|
||||
plane == 0 ? av1_default_palette_y_color_index_prob
|
||||
: av1_default_palette_uv_color_index_prob;
|
||||
aom_cdf_prob(
|
||||
*palette_cdf)[PALETTE_COLOR_INDEX_CONTEXTS][CDF_SIZE(PALETTE_COLORS)] =
|
||||
plane ? xd->tile_ctx->palette_uv_color_index_cdf
|
||||
: xd->tile_ctx->palette_y_color_index_cdf;
|
||||
int plane_block_width, rows, cols;
|
||||
av1_get_block_dimensions(bsize, plane, xd, &plane_block_width, NULL, &rows,
|
||||
&cols);
|
||||
assert(plane == 0 || plane == 1);
|
||||
|
||||
// The first color index does not use context or entropy.
|
||||
(*t)->token = color_map[0];
|
||||
(*t)->palette_cdf = NULL;
|
||||
(*t)->skip_eob_node = 0;
|
||||
++(*t);
|
||||
|
||||
const int n = pmi->palette_size[plane];
|
||||
const int calc_rate = rate && dry_run == DRY_RUN_COSTCOEFFS;
|
||||
int this_rate = 0;
|
||||
uint8_t color_order[PALETTE_MAX_SIZE];
|
||||
#if CONFIG_PALETTE_THROUGHPUT
|
||||
int k;
|
||||
for (k = 1; k < rows + cols - 1; ++k) {
|
||||
for (j = AOMMIN(k, cols - 1); j >= AOMMAX(0, k - rows + 1); --j) {
|
||||
i = k - j;
|
||||
for (int k = 1; k < rows + cols - 1; ++k) {
|
||||
for (int j = AOMMIN(k, cols - 1); j >= AOMMAX(0, k - rows + 1); --j) {
|
||||
int i = k - j;
|
||||
#else
|
||||
for (i = 0; i < rows; ++i) {
|
||||
for (j = (i == 0 ? 1 : 0); j < cols; ++j) {
|
||||
for (int i = 0; i < rows; ++i) {
|
||||
for (int j = (i == 0 ? 1 : 0); j < cols; ++j) {
|
||||
#endif // CONFIG_PALETTE_THROUGHPUT
|
||||
int color_new_idx;
|
||||
const int color_ctx = av1_get_palette_color_index_context(
|
||||
color_map, plane_block_width, i, j, n, color_order, &color_new_idx);
|
||||
assert(color_new_idx >= 0 && color_new_idx < n);
|
||||
if (dry_run == DRY_RUN_COSTCOEFFS)
|
||||
if (calc_rate) {
|
||||
this_rate += cpi->palette_y_color_cost[n - PALETTE_MIN_SIZE][color_ctx]
|
||||
[color_new_idx];
|
||||
}
|
||||
(*t)->token = color_new_idx;
|
||||
(*t)->context_tree = probs[n - PALETTE_MIN_SIZE][color_ctx];
|
||||
(*t)->palette_cdf = palette_cdf[n - PALETTE_MIN_SIZE][color_ctx];
|
||||
(*t)->skip_eob_node = 0;
|
||||
++(*t);
|
||||
}
|
||||
|
|
@ -434,17 +440,13 @@ static void tokenize_b(int plane, int block, int blk_row, int blk_col,
|
|||
const int segment_id = mbmi->segment_id;
|
||||
#endif // CONFIG_SUEPRTX
|
||||
const int16_t *scan, *nb;
|
||||
const TX_TYPE tx_type = get_tx_type(type, xd, block, tx_size);
|
||||
const SCAN_ORDER *const scan_order =
|
||||
get_scan(cm, tx_size, tx_type, is_inter_block(mbmi));
|
||||
const TX_TYPE tx_type =
|
||||
av1_get_tx_type(type, xd, blk_row, blk_col, block, tx_size);
|
||||
const SCAN_ORDER *const scan_order = get_scan(cm, tx_size, tx_type, mbmi);
|
||||
const int ref = is_inter_block(mbmi);
|
||||
unsigned int(*const counts)[COEFF_CONTEXTS][ENTROPY_TOKENS] =
|
||||
td->rd_counts.coef_counts[txsize_sqr_map[tx_size]][type][ref];
|
||||
#if CONFIG_EC_ADAPT
|
||||
FRAME_CONTEXT *ec_ctx = xd->tile_ctx;
|
||||
#else
|
||||
FRAME_CONTEXT *ec_ctx = cpi->common.fc;
|
||||
#endif
|
||||
aom_cdf_prob(
|
||||
*const coef_head_cdfs)[COEFF_CONTEXTS][CDF_SIZE(ENTROPY_TOKENS)] =
|
||||
ec_ctx->coef_head_cdfs[txsize_sqr_map[tx_size]][type][ref];
|
||||
|
|
@ -595,16 +597,31 @@ void tokenize_vartx(ThreadData *td, TOKENEXTRA **t, RUN_TYPE dry_run,
|
|||
cost_coeffs_b(plane, block, blk_row, blk_col, plane_bsize, tx_size, arg);
|
||||
#endif
|
||||
} else {
|
||||
#if CONFIG_RECT_TX_EXT
|
||||
int is_qttx = plane_tx_size == quarter_txsize_lookup[plane_bsize];
|
||||
const TX_SIZE sub_txs = is_qttx ? plane_tx_size : sub_tx_size_map[tx_size];
|
||||
#else
|
||||
// Half the block size in transform block unit.
|
||||
const TX_SIZE sub_txs = sub_tx_size_map[tx_size];
|
||||
#endif
|
||||
const int bsl = tx_size_wide_unit[sub_txs];
|
||||
int i;
|
||||
|
||||
assert(bsl > 0);
|
||||
|
||||
for (i = 0; i < 4; ++i) {
|
||||
#if CONFIG_RECT_TX_EXT
|
||||
int is_wide_tx = tx_size_wide_unit[sub_txs] > tx_size_high_unit[sub_txs];
|
||||
const int offsetr =
|
||||
is_qttx ? (is_wide_tx ? i * tx_size_high_unit[sub_txs] : 0)
|
||||
: blk_row + ((i >> 1) * bsl);
|
||||
const int offsetc =
|
||||
is_qttx ? (is_wide_tx ? 0 : i * tx_size_wide_unit[sub_txs])
|
||||
: blk_col + ((i & 0x01) * bsl);
|
||||
#else
|
||||
const int offsetr = blk_row + ((i >> 1) * bsl);
|
||||
const int offsetc = blk_col + ((i & 0x01) * bsl);
|
||||
#endif
|
||||
|
||||
int step = tx_size_wide_unit[sub_txs] * tx_size_high_unit[sub_txs];
|
||||
|
||||
|
|
@ -666,7 +683,7 @@ void av1_tokenize_sb_vartx(const AV1_COMP *cpi, ThreadData *td, TOKENEXTRA **t,
|
|||
}
|
||||
#endif
|
||||
const struct macroblockd_plane *const pd = &xd->plane[plane];
|
||||
#if CONFIG_CB4X4 && !CONFIG_CHROMA_2X2
|
||||
#if CONFIG_CHROMA_SUB8X8
|
||||
const BLOCK_SIZE plane_bsize =
|
||||
AOMMAX(BLOCK_4X4, get_plane_block_size(bsize, pd));
|
||||
#else
|
||||
|
|
@ -681,14 +698,30 @@ void av1_tokenize_sb_vartx(const AV1_COMP *cpi, ThreadData *td, TOKENEXTRA **t,
|
|||
int idx, idy;
|
||||
int block = 0;
|
||||
int step = tx_size_wide_unit[max_tx_size] * tx_size_high_unit[max_tx_size];
|
||||
for (idy = 0; idy < mi_height; idy += bh) {
|
||||
for (idx = 0; idx < mi_width; idx += bw) {
|
||||
tokenize_vartx(td, t, dry_run, max_tx_size, plane_bsize, idy, idx,
|
||||
block, plane, &arg);
|
||||
block += step;
|
||||
|
||||
const BLOCK_SIZE max_unit_bsize = get_plane_block_size(BLOCK_64X64, pd);
|
||||
int mu_blocks_wide =
|
||||
block_size_wide[max_unit_bsize] >> tx_size_wide_log2[0];
|
||||
int mu_blocks_high =
|
||||
block_size_high[max_unit_bsize] >> tx_size_high_log2[0];
|
||||
|
||||
mu_blocks_wide = AOMMIN(mi_width, mu_blocks_wide);
|
||||
mu_blocks_high = AOMMIN(mi_height, mu_blocks_high);
|
||||
|
||||
for (idy = 0; idy < mi_height; idy += mu_blocks_high) {
|
||||
for (idx = 0; idx < mi_width; idx += mu_blocks_wide) {
|
||||
int blk_row, blk_col;
|
||||
const int unit_height = AOMMIN(mu_blocks_high + idy, mi_height);
|
||||
const int unit_width = AOMMIN(mu_blocks_wide + idx, mi_width);
|
||||
for (blk_row = idy; blk_row < unit_height; blk_row += bh) {
|
||||
for (blk_col = idx; blk_col < unit_width; blk_col += bw) {
|
||||
tokenize_vartx(td, t, dry_run, max_tx_size, plane_bsize, blk_row,
|
||||
blk_col, block, plane, &arg);
|
||||
block += step;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#if !CONFIG_LV_MAP
|
||||
if (!dry_run) {
|
||||
(*t)->token = EOSB_TOKEN;
|
||||
|
|
|
|||
3
third_party/aom/av1/encoder/tokenize.h
vendored
3
third_party/aom/av1/encoder/tokenize.h
vendored
|
|
@ -37,6 +37,9 @@ typedef struct {
|
|||
typedef struct {
|
||||
aom_cdf_prob (*tail_cdf)[CDF_SIZE(ENTROPY_TOKENS)];
|
||||
aom_cdf_prob (*head_cdf)[CDF_SIZE(ENTROPY_TOKENS)];
|
||||
#if CONFIG_PALETTE
|
||||
aom_cdf_prob *palette_cdf;
|
||||
#endif // CONFIG_PALETTE
|
||||
int eob_val;
|
||||
int first_val;
|
||||
const aom_prob *context_tree;
|
||||
|
|
|
|||
143
third_party/aom/av1/encoder/x86/av1_highbd_quantize_avx2.c
vendored
Normal file
143
third_party/aom/av1/encoder/x86/av1_highbd_quantize_avx2.c
vendored
Normal file
|
|
@ -0,0 +1,143 @@
|
|||
/*
|
||||
* Copyright (c) 2017, Alliance for Open Media. All rights reserved
|
||||
*
|
||||
* This source code is subject to the terms of the BSD 2 Clause License and
|
||||
* the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
|
||||
* was not distributed with this source code in the LICENSE file, you can
|
||||
* obtain it at www.aomedia.org/license/software. If the Alliance for Open
|
||||
* Media Patent License 1.0 was not distributed with this source code in the
|
||||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#include <immintrin.h>
|
||||
|
||||
#include "./av1_rtcd.h"
|
||||
#include "aom/aom_integer.h"
|
||||
#include "aom_dsp/aom_dsp_common.h"
|
||||
|
||||
static INLINE void init_one_qp(const __m128i *p, __m256i *qp) {
|
||||
const __m128i zero = _mm_setzero_si128();
|
||||
const __m128i dc = _mm_unpacklo_epi16(*p, zero);
|
||||
const __m128i ac = _mm_unpackhi_epi16(*p, zero);
|
||||
*qp = _mm256_insertf128_si256(_mm256_castsi128_si256(dc), ac, 1);
|
||||
}
|
||||
|
||||
static INLINE void update_qp(__m256i *qp) {
|
||||
qp[0] = _mm256_permute2x128_si256(qp[0], qp[0], 0x11);
|
||||
qp[1] = _mm256_permute2x128_si256(qp[1], qp[1], 0x11);
|
||||
qp[2] = _mm256_permute2x128_si256(qp[2], qp[2], 0x11);
|
||||
}
|
||||
|
||||
static INLINE void init_qp(const int16_t *round_ptr, const int16_t *quant_ptr,
|
||||
const int16_t *dequant_ptr, int log_scale,
|
||||
__m256i *qp) {
|
||||
__m128i round = _mm_loadu_si128((const __m128i *)round_ptr);
|
||||
round = _mm_srai_epi16(round, log_scale);
|
||||
const __m128i quant = _mm_loadu_si128((const __m128i *)quant_ptr);
|
||||
const __m128i dequant = _mm_loadu_si128((const __m128i *)dequant_ptr);
|
||||
|
||||
init_one_qp(&round, &qp[0]);
|
||||
init_one_qp(&quant, &qp[1]);
|
||||
init_one_qp(&dequant, &qp[2]);
|
||||
}
|
||||
|
||||
static INLINE void quantize(const __m256i *qp, __m256i *c,
|
||||
const int16_t *iscan_ptr, int log_scale,
|
||||
tran_low_t *qcoeff, tran_low_t *dqcoeff,
|
||||
__m256i *eob) {
|
||||
const __m256i abs = _mm256_abs_epi32(*c);
|
||||
__m256i q = _mm256_add_epi32(abs, qp[0]);
|
||||
|
||||
__m256i q_lo = _mm256_mul_epi32(q, qp[1]);
|
||||
__m256i q_hi = _mm256_srli_epi64(q, 32);
|
||||
const __m256i qp_hi = _mm256_srli_epi64(qp[1], 32);
|
||||
q_hi = _mm256_mul_epi32(q_hi, qp_hi);
|
||||
q_lo = _mm256_srli_epi64(q_lo, 16 - log_scale);
|
||||
q_hi = _mm256_srli_epi64(q_hi, 16 - log_scale);
|
||||
q_hi = _mm256_slli_epi64(q_hi, 32);
|
||||
q = _mm256_or_si256(q_lo, q_hi);
|
||||
|
||||
__m256i dq = _mm256_mullo_epi32(q, qp[2]);
|
||||
dq = _mm256_srai_epi32(dq, log_scale);
|
||||
q = _mm256_sign_epi32(q, *c);
|
||||
dq = _mm256_sign_epi32(dq, *c);
|
||||
|
||||
_mm256_storeu_si256((__m256i *)qcoeff, q);
|
||||
_mm256_storeu_si256((__m256i *)dqcoeff, dq);
|
||||
|
||||
const __m128i isc = _mm_loadu_si128((const __m128i *)iscan_ptr);
|
||||
const __m128i zr = _mm_setzero_si128();
|
||||
const __m128i lo = _mm_unpacklo_epi16(isc, zr);
|
||||
const __m128i hi = _mm_unpackhi_epi16(isc, zr);
|
||||
const __m256i iscan =
|
||||
_mm256_insertf128_si256(_mm256_castsi128_si256(lo), hi, 1);
|
||||
|
||||
const __m256i zero = _mm256_setzero_si256();
|
||||
const __m256i zc = _mm256_cmpeq_epi32(dq, zero);
|
||||
const __m256i nz = _mm256_cmpeq_epi32(zc, zero);
|
||||
__m256i cur_eob = _mm256_sub_epi32(iscan, nz);
|
||||
cur_eob = _mm256_and_si256(cur_eob, nz);
|
||||
*eob = _mm256_max_epi32(cur_eob, *eob);
|
||||
}
|
||||
|
||||
void av1_highbd_quantize_fp_avx2(
|
||||
const tran_low_t *coeff_ptr, intptr_t n_coeffs, int skip_block,
|
||||
const int16_t *zbin_ptr, const int16_t *round_ptr, const int16_t *quant_ptr,
|
||||
const int16_t *quant_shift_ptr, tran_low_t *qcoeff_ptr,
|
||||
tran_low_t *dqcoeff_ptr, const int16_t *dequant_ptr, uint16_t *eob_ptr,
|
||||
const int16_t *scan, const int16_t *iscan, int log_scale) {
|
||||
(void)scan;
|
||||
(void)zbin_ptr;
|
||||
(void)quant_shift_ptr;
|
||||
const unsigned int step = 8;
|
||||
|
||||
if (LIKELY(!skip_block)) {
|
||||
__m256i qp[3], coeff;
|
||||
|
||||
init_qp(round_ptr, quant_ptr, dequant_ptr, log_scale, qp);
|
||||
coeff = _mm256_loadu_si256((const __m256i *)coeff_ptr);
|
||||
|
||||
__m256i eob = _mm256_setzero_si256();
|
||||
quantize(qp, &coeff, iscan, log_scale, qcoeff_ptr, dqcoeff_ptr, &eob);
|
||||
|
||||
coeff_ptr += step;
|
||||
qcoeff_ptr += step;
|
||||
dqcoeff_ptr += step;
|
||||
iscan += step;
|
||||
n_coeffs -= step;
|
||||
|
||||
update_qp(qp);
|
||||
while (n_coeffs > 0) {
|
||||
coeff = _mm256_loadu_si256((const __m256i *)coeff_ptr);
|
||||
quantize(qp, &coeff, iscan, log_scale, qcoeff_ptr, dqcoeff_ptr, &eob);
|
||||
|
||||
coeff_ptr += step;
|
||||
qcoeff_ptr += step;
|
||||
dqcoeff_ptr += step;
|
||||
iscan += step;
|
||||
n_coeffs -= step;
|
||||
}
|
||||
{
|
||||
__m256i eob_s;
|
||||
eob_s = _mm256_shuffle_epi32(eob, 0xe);
|
||||
eob = _mm256_max_epi16(eob, eob_s);
|
||||
eob_s = _mm256_shufflelo_epi16(eob, 0xe);
|
||||
eob = _mm256_max_epi16(eob, eob_s);
|
||||
eob_s = _mm256_shufflelo_epi16(eob, 1);
|
||||
eob = _mm256_max_epi16(eob, eob_s);
|
||||
const __m128i final_eob = _mm_max_epi16(_mm256_castsi256_si128(eob),
|
||||
_mm256_extractf128_si256(eob, 1));
|
||||
*eob_ptr = _mm_extract_epi16(final_eob, 0);
|
||||
}
|
||||
} else {
|
||||
do {
|
||||
const __m256i zero = _mm256_setzero_si256();
|
||||
_mm256_storeu_si256((__m256i *)qcoeff_ptr, zero);
|
||||
_mm256_storeu_si256((__m256i *)dqcoeff_ptr, zero);
|
||||
qcoeff_ptr += step;
|
||||
dqcoeff_ptr += step;
|
||||
n_coeffs -= step;
|
||||
} while (n_coeffs > 0);
|
||||
*eob_ptr = 0;
|
||||
}
|
||||
}
|
||||
|
|
@ -133,9 +133,10 @@ void av1_highbd_quantize_fp_sse4_1(
|
|||
coeff[0] = _mm_loadu_si128((__m128i const *)src);
|
||||
|
||||
qparam[0] =
|
||||
_mm_set_epi32(round_ptr[1], round_ptr[1], round_ptr[1], round_ptr[0]);
|
||||
qparam[1] = _mm_set_epi64x(quant_ptr[1], quant_ptr[0]);
|
||||
qparam[2] = _mm_set_epi64x(dequant_ptr[1], dequant_ptr[0]);
|
||||
_mm_set_epi32(round_ptr[1] >> log_scale, round_ptr[1] >> log_scale,
|
||||
round_ptr[1] >> log_scale, round_ptr[0] >> log_scale);
|
||||
qparam[1] = _mm_set_epi32(0, quant_ptr[1], 0, quant_ptr[0]);
|
||||
qparam[2] = _mm_set_epi32(0, dequant_ptr[1], 0, dequant_ptr[0]);
|
||||
|
||||
// DC and first 3 AC
|
||||
quantize_coeff_phase1(&coeff[0], qparam, shift, log_scale, qcoeff, dequant,
|
||||
|
|
@ -143,8 +144,8 @@ void av1_highbd_quantize_fp_sse4_1(
|
|||
|
||||
// update round/quan/dquan for AC
|
||||
qparam[0] = _mm_unpackhi_epi64(qparam[0], qparam[0]);
|
||||
qparam[1] = _mm_set_epi64x(quant_ptr[1], quant_ptr[1]);
|
||||
qparam[2] = _mm_set_epi64x(dequant_ptr[1], dequant_ptr[1]);
|
||||
qparam[1] = _mm_set_epi32(0, quant_ptr[1], 0, quant_ptr[1]);
|
||||
qparam[2] = _mm_set_epi32(0, dequant_ptr[1], 0, dequant_ptr[1]);
|
||||
|
||||
quantize_coeff_phase2(qcoeff, dequant, &coeff_sign, qparam, shift,
|
||||
log_scale, quanAddr, dquanAddr);
|
||||
|
|
|
|||
289
third_party/aom/av1/encoder/x86/av1_quantize_avx2.c
vendored
Normal file
289
third_party/aom/av1/encoder/x86/av1_quantize_avx2.c
vendored
Normal file
|
|
@ -0,0 +1,289 @@
|
|||
/*
|
||||
* Copyright (c) 2017, Alliance for Open Media. All rights reserved
|
||||
*
|
||||
* This source code is subject to the terms of the BSD 2 Clause License and
|
||||
* the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
|
||||
* was not distributed with this source code in the LICENSE file, you can
|
||||
* obtain it at www.aomedia.org/license/software. If the Alliance for Open
|
||||
* Media Patent License 1.0 was not distributed with this source code in the
|
||||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#include <immintrin.h>
|
||||
|
||||
#include "./av1_rtcd.h"
|
||||
#include "aom/aom_integer.h"
|
||||
#include "aom_dsp/aom_dsp_common.h"
|
||||
|
||||
static INLINE void read_coeff(const tran_low_t *coeff, __m256i *c) {
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
const __m256i x0 = _mm256_loadu_si256((const __m256i *)coeff);
|
||||
const __m256i x1 = _mm256_loadu_si256((const __m256i *)coeff + 1);
|
||||
*c = _mm256_packs_epi32(x0, x1);
|
||||
*c = _mm256_permute4x64_epi64(*c, 0xD8);
|
||||
#else
|
||||
*c = _mm256_loadu_si256((const __m256i *)coeff);
|
||||
#endif
|
||||
}
|
||||
|
||||
static INLINE void write_zero(tran_low_t *qcoeff) {
|
||||
const __m256i zero = _mm256_setzero_si256();
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
_mm256_storeu_si256((__m256i *)qcoeff, zero);
|
||||
_mm256_storeu_si256((__m256i *)qcoeff + 1, zero);
|
||||
#else
|
||||
_mm256_storeu_si256((__m256i *)qcoeff, zero);
|
||||
#endif
|
||||
}
|
||||
|
||||
static INLINE void init_one_qp(const __m128i *p, __m256i *qp) {
|
||||
const __m128i ac = _mm_unpackhi_epi64(*p, *p);
|
||||
*qp = _mm256_insertf128_si256(_mm256_castsi128_si256(*p), ac, 1);
|
||||
}
|
||||
|
||||
static INLINE void init_qp(const int16_t *round_ptr, const int16_t *quant_ptr,
|
||||
const int16_t *dequant_ptr, int log_scale,
|
||||
__m256i *thr, __m256i *qp) {
|
||||
__m128i round = _mm_loadu_si128((const __m128i *)round_ptr);
|
||||
const __m128i quant = _mm_loadu_si128((const __m128i *)quant_ptr);
|
||||
const __m128i dequant = _mm_loadu_si128((const __m128i *)dequant_ptr);
|
||||
|
||||
if (log_scale > 0) {
|
||||
const __m128i rnd = _mm_set1_epi16((int16_t)1 << (log_scale - 1));
|
||||
round = _mm_add_epi16(round, rnd);
|
||||
round = _mm_srai_epi16(round, log_scale);
|
||||
}
|
||||
|
||||
init_one_qp(&round, &qp[0]);
|
||||
init_one_qp(&quant, &qp[1]);
|
||||
|
||||
if (log_scale > 0) {
|
||||
qp[1] = _mm256_slli_epi16(qp[1], log_scale);
|
||||
}
|
||||
|
||||
init_one_qp(&dequant, &qp[2]);
|
||||
*thr = _mm256_srai_epi16(qp[2], 1 + log_scale);
|
||||
}
|
||||
|
||||
static INLINE void update_qp(int log_scale, __m256i *thr, __m256i *qp) {
|
||||
qp[0] = _mm256_permute2x128_si256(qp[0], qp[0], 0x11);
|
||||
qp[1] = _mm256_permute2x128_si256(qp[1], qp[1], 0x11);
|
||||
qp[2] = _mm256_permute2x128_si256(qp[2], qp[2], 0x11);
|
||||
*thr = _mm256_srai_epi16(qp[2], 1 + log_scale);
|
||||
}
|
||||
|
||||
#define store_quan(q, addr) \
|
||||
do { \
|
||||
__m256i sign_bits = _mm256_srai_epi16(q, 15); \
|
||||
__m256i y0 = _mm256_unpacklo_epi16(q, sign_bits); \
|
||||
__m256i y1 = _mm256_unpackhi_epi16(q, sign_bits); \
|
||||
__m256i x0 = _mm256_permute2x128_si256(y0, y1, 0x20); \
|
||||
__m256i x1 = _mm256_permute2x128_si256(y0, y1, 0x31); \
|
||||
_mm256_storeu_si256((__m256i *)addr, x0); \
|
||||
_mm256_storeu_si256((__m256i *)addr + 1, x1); \
|
||||
} while (0)
|
||||
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
#define store_two_quan(q, addr1, dq, addr2) \
|
||||
do { \
|
||||
store_quan(q, addr1); \
|
||||
store_quan(dq, addr2); \
|
||||
} while (0)
|
||||
#else
|
||||
#define store_two_quan(q, addr1, dq, addr2) \
|
||||
do { \
|
||||
_mm256_storeu_si256((__m256i *)addr1, q); \
|
||||
_mm256_storeu_si256((__m256i *)addr2, dq); \
|
||||
} while (0)
|
||||
#endif
|
||||
|
||||
static INLINE void quantize(const __m256i *thr, const __m256i *qp, __m256i *c,
|
||||
const int16_t *iscan_ptr, tran_low_t *qcoeff,
|
||||
tran_low_t *dqcoeff, __m256i *eob) {
|
||||
const __m256i abs = _mm256_abs_epi16(*c);
|
||||
__m256i mask = _mm256_cmpgt_epi16(abs, *thr);
|
||||
mask = _mm256_or_si256(mask, _mm256_cmpeq_epi16(abs, *thr));
|
||||
const int nzflag = _mm256_movemask_epi8(mask);
|
||||
|
||||
if (nzflag) {
|
||||
__m256i q = _mm256_adds_epi16(abs, qp[0]);
|
||||
q = _mm256_mulhi_epi16(q, qp[1]);
|
||||
q = _mm256_sign_epi16(q, *c);
|
||||
const __m256i dq = _mm256_mullo_epi16(q, qp[2]);
|
||||
|
||||
store_two_quan(q, qcoeff, dq, dqcoeff);
|
||||
const __m256i zero = _mm256_setzero_si256();
|
||||
const __m256i iscan = _mm256_loadu_si256((const __m256i *)iscan_ptr);
|
||||
const __m256i zero_coeff = _mm256_cmpeq_epi16(dq, zero);
|
||||
const __m256i nzero_coeff = _mm256_cmpeq_epi16(zero_coeff, zero);
|
||||
__m256i cur_eob = _mm256_sub_epi16(iscan, nzero_coeff);
|
||||
cur_eob = _mm256_and_si256(cur_eob, nzero_coeff);
|
||||
*eob = _mm256_max_epi16(*eob, cur_eob);
|
||||
} else {
|
||||
write_zero(qcoeff);
|
||||
write_zero(dqcoeff);
|
||||
}
|
||||
}
|
||||
|
||||
void av1_quantize_fp_avx2(const tran_low_t *coeff_ptr, intptr_t n_coeffs,
|
||||
int skip_block, const int16_t *zbin_ptr,
|
||||
const int16_t *round_ptr, const int16_t *quant_ptr,
|
||||
const int16_t *quant_shift_ptr,
|
||||
tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr,
|
||||
const int16_t *dequant_ptr, uint16_t *eob_ptr,
|
||||
const int16_t *scan_ptr, const int16_t *iscan_ptr) {
|
||||
(void)scan_ptr;
|
||||
(void)zbin_ptr;
|
||||
(void)quant_shift_ptr;
|
||||
const unsigned int step = 16;
|
||||
|
||||
if (LIKELY(!skip_block)) {
|
||||
__m256i qp[3];
|
||||
__m256i coeff, thr;
|
||||
const int log_scale = 0;
|
||||
|
||||
init_qp(round_ptr, quant_ptr, dequant_ptr, log_scale, &thr, qp);
|
||||
read_coeff(coeff_ptr, &coeff);
|
||||
|
||||
__m256i eob = _mm256_setzero_si256();
|
||||
quantize(&thr, qp, &coeff, iscan_ptr, qcoeff_ptr, dqcoeff_ptr, &eob);
|
||||
|
||||
coeff_ptr += step;
|
||||
qcoeff_ptr += step;
|
||||
dqcoeff_ptr += step;
|
||||
iscan_ptr += step;
|
||||
n_coeffs -= step;
|
||||
|
||||
update_qp(log_scale, &thr, qp);
|
||||
|
||||
while (n_coeffs > 0) {
|
||||
read_coeff(coeff_ptr, &coeff);
|
||||
quantize(&thr, qp, &coeff, iscan_ptr, qcoeff_ptr, dqcoeff_ptr, &eob);
|
||||
|
||||
coeff_ptr += step;
|
||||
qcoeff_ptr += step;
|
||||
dqcoeff_ptr += step;
|
||||
iscan_ptr += step;
|
||||
n_coeffs -= step;
|
||||
}
|
||||
{
|
||||
__m256i eob_s;
|
||||
eob_s = _mm256_shuffle_epi32(eob, 0xe);
|
||||
eob = _mm256_max_epi16(eob, eob_s);
|
||||
eob_s = _mm256_shufflelo_epi16(eob, 0xe);
|
||||
eob = _mm256_max_epi16(eob, eob_s);
|
||||
eob_s = _mm256_shufflelo_epi16(eob, 1);
|
||||
eob = _mm256_max_epi16(eob, eob_s);
|
||||
const __m128i final_eob = _mm_max_epi16(_mm256_castsi256_si128(eob),
|
||||
_mm256_extractf128_si256(eob, 1));
|
||||
*eob_ptr = _mm_extract_epi16(final_eob, 0);
|
||||
}
|
||||
} else {
|
||||
do {
|
||||
write_zero(qcoeff_ptr);
|
||||
write_zero(dqcoeff_ptr);
|
||||
qcoeff_ptr += step;
|
||||
dqcoeff_ptr += step;
|
||||
n_coeffs -= step;
|
||||
} while (n_coeffs > 0);
|
||||
*eob_ptr = 0;
|
||||
}
|
||||
}
|
||||
|
||||
static INLINE void quantize_32x32(const __m256i *thr, const __m256i *qp,
|
||||
__m256i *c, const int16_t *iscan_ptr,
|
||||
tran_low_t *qcoeff, tran_low_t *dqcoeff,
|
||||
__m256i *eob) {
|
||||
const __m256i abs = _mm256_abs_epi16(*c);
|
||||
__m256i mask = _mm256_cmpgt_epi16(abs, *thr);
|
||||
mask = _mm256_or_si256(mask, _mm256_cmpeq_epi16(abs, *thr));
|
||||
const int nzflag = _mm256_movemask_epi8(mask);
|
||||
|
||||
if (nzflag) {
|
||||
__m256i q = _mm256_adds_epi16(abs, qp[0]);
|
||||
q = _mm256_mulhi_epu16(q, qp[1]);
|
||||
|
||||
__m256i dq = _mm256_mullo_epi16(q, qp[2]);
|
||||
dq = _mm256_srli_epi16(dq, 1);
|
||||
|
||||
q = _mm256_sign_epi16(q, *c);
|
||||
dq = _mm256_sign_epi16(dq, *c);
|
||||
|
||||
store_two_quan(q, qcoeff, dq, dqcoeff);
|
||||
const __m256i zero = _mm256_setzero_si256();
|
||||
const __m256i iscan = _mm256_loadu_si256((const __m256i *)iscan_ptr);
|
||||
const __m256i zero_coeff = _mm256_cmpeq_epi16(dq, zero);
|
||||
const __m256i nzero_coeff = _mm256_cmpeq_epi16(zero_coeff, zero);
|
||||
__m256i cur_eob = _mm256_sub_epi16(iscan, nzero_coeff);
|
||||
cur_eob = _mm256_and_si256(cur_eob, nzero_coeff);
|
||||
*eob = _mm256_max_epi16(*eob, cur_eob);
|
||||
} else {
|
||||
write_zero(qcoeff);
|
||||
write_zero(dqcoeff);
|
||||
}
|
||||
}
|
||||
|
||||
void av1_quantize_fp_32x32_avx2(
|
||||
const tran_low_t *coeff_ptr, intptr_t n_coeffs, int skip_block,
|
||||
const int16_t *zbin_ptr, const int16_t *round_ptr, const int16_t *quant_ptr,
|
||||
const int16_t *quant_shift_ptr, tran_low_t *qcoeff_ptr,
|
||||
tran_low_t *dqcoeff_ptr, const int16_t *dequant_ptr, uint16_t *eob_ptr,
|
||||
const int16_t *scan_ptr, const int16_t *iscan_ptr) {
|
||||
(void)scan_ptr;
|
||||
(void)zbin_ptr;
|
||||
(void)quant_shift_ptr;
|
||||
const unsigned int step = 16;
|
||||
|
||||
if (LIKELY(!skip_block)) {
|
||||
__m256i qp[3];
|
||||
__m256i coeff, thr;
|
||||
const int log_scale = 1;
|
||||
|
||||
init_qp(round_ptr, quant_ptr, dequant_ptr, log_scale, &thr, qp);
|
||||
read_coeff(coeff_ptr, &coeff);
|
||||
|
||||
__m256i eob = _mm256_setzero_si256();
|
||||
quantize_32x32(&thr, qp, &coeff, iscan_ptr, qcoeff_ptr, dqcoeff_ptr, &eob);
|
||||
|
||||
coeff_ptr += step;
|
||||
qcoeff_ptr += step;
|
||||
dqcoeff_ptr += step;
|
||||
iscan_ptr += step;
|
||||
n_coeffs -= step;
|
||||
|
||||
update_qp(log_scale, &thr, qp);
|
||||
|
||||
while (n_coeffs > 0) {
|
||||
read_coeff(coeff_ptr, &coeff);
|
||||
quantize_32x32(&thr, qp, &coeff, iscan_ptr, qcoeff_ptr, dqcoeff_ptr,
|
||||
&eob);
|
||||
|
||||
coeff_ptr += step;
|
||||
qcoeff_ptr += step;
|
||||
dqcoeff_ptr += step;
|
||||
iscan_ptr += step;
|
||||
n_coeffs -= step;
|
||||
}
|
||||
{
|
||||
__m256i eob_s;
|
||||
eob_s = _mm256_shuffle_epi32(eob, 0xe);
|
||||
eob = _mm256_max_epi16(eob, eob_s);
|
||||
eob_s = _mm256_shufflelo_epi16(eob, 0xe);
|
||||
eob = _mm256_max_epi16(eob, eob_s);
|
||||
eob_s = _mm256_shufflelo_epi16(eob, 1);
|
||||
eob = _mm256_max_epi16(eob, eob_s);
|
||||
const __m128i final_eob = _mm_max_epi16(_mm256_castsi256_si128(eob),
|
||||
_mm256_extractf128_si256(eob, 1));
|
||||
*eob_ptr = _mm_extract_epi16(final_eob, 0);
|
||||
}
|
||||
} else {
|
||||
do {
|
||||
write_zero(qcoeff_ptr);
|
||||
write_zero(dqcoeff_ptr);
|
||||
qcoeff_ptr += step;
|
||||
dqcoeff_ptr += step;
|
||||
n_coeffs -= step;
|
||||
} while (n_coeffs > 0);
|
||||
*eob_ptr = 0;
|
||||
}
|
||||
}
|
||||
|
|
@ -203,8 +203,12 @@ static void fidtx4_sse2(__m128i *in) {
|
|||
#endif // CONFIG_EXT_TX
|
||||
|
||||
void av1_fht4x4_sse2(const int16_t *input, tran_low_t *output, int stride,
|
||||
int tx_type) {
|
||||
TxfmParam *txfm_param) {
|
||||
__m128i in[4];
|
||||
int tx_type = txfm_param->tx_type;
|
||||
#if CONFIG_MRC_TX
|
||||
assert(tx_type != MRC_DCT && "Invalid tx type for tx size");
|
||||
#endif
|
||||
|
||||
switch (tx_type) {
|
||||
case DCT_DCT: aom_fdct4x4_sse2(input, output, stride); break;
|
||||
|
|
@ -1301,8 +1305,12 @@ static void fidtx8_sse2(__m128i *in) {
|
|||
#endif // CONFIG_EXT_TX
|
||||
|
||||
void av1_fht8x8_sse2(const int16_t *input, tran_low_t *output, int stride,
|
||||
int tx_type) {
|
||||
TxfmParam *txfm_param) {
|
||||
__m128i in[8];
|
||||
int tx_type = txfm_param->tx_type;
|
||||
#if CONFIG_MRC_TX
|
||||
assert(tx_type != MRC_DCT && "Invalid tx type for tx size");
|
||||
#endif
|
||||
|
||||
switch (tx_type) {
|
||||
case DCT_DCT: aom_fdct8x8_sse2(input, output, stride); break;
|
||||
|
|
@ -2334,8 +2342,12 @@ static void fidtx16_sse2(__m128i *in0, __m128i *in1) {
|
|||
#endif // CONFIG_EXT_TX
|
||||
|
||||
void av1_fht16x16_sse2(const int16_t *input, tran_low_t *output, int stride,
|
||||
int tx_type) {
|
||||
TxfmParam *txfm_param) {
|
||||
__m128i in0[16], in1[16];
|
||||
int tx_type = txfm_param->tx_type;
|
||||
#if CONFIG_MRC_TX
|
||||
assert(tx_type != MRC_DCT && "Invalid tx type for tx size");
|
||||
#endif
|
||||
|
||||
switch (tx_type) {
|
||||
case DCT_DCT:
|
||||
|
|
@ -2550,8 +2562,12 @@ static INLINE void write_buffer_4x8(tran_low_t *output, __m128i *res) {
|
|||
}
|
||||
|
||||
void av1_fht4x8_sse2(const int16_t *input, tran_low_t *output, int stride,
|
||||
int tx_type) {
|
||||
TxfmParam *txfm_param) {
|
||||
__m128i in[8];
|
||||
int tx_type = txfm_param->tx_type;
|
||||
#if CONFIG_MRC_TX
|
||||
assert(tx_type != MRC_DCT && "Invalid tx type for tx size");
|
||||
#endif
|
||||
|
||||
switch (tx_type) {
|
||||
case DCT_DCT:
|
||||
|
|
@ -2724,8 +2740,12 @@ static INLINE void write_buffer_8x4(tran_low_t *output, __m128i *res) {
|
|||
}
|
||||
|
||||
void av1_fht8x4_sse2(const int16_t *input, tran_low_t *output, int stride,
|
||||
int tx_type) {
|
||||
TxfmParam *txfm_param) {
|
||||
__m128i in[8];
|
||||
int tx_type = txfm_param->tx_type;
|
||||
#if CONFIG_MRC_TX
|
||||
assert(tx_type != MRC_DCT && "Invalid tx type for tx size");
|
||||
#endif
|
||||
|
||||
switch (tx_type) {
|
||||
case DCT_DCT:
|
||||
|
|
@ -2864,8 +2884,12 @@ static void row_8x16_rounding(__m128i *in, int bits) {
|
|||
}
|
||||
|
||||
void av1_fht8x16_sse2(const int16_t *input, tran_low_t *output, int stride,
|
||||
int tx_type) {
|
||||
TxfmParam *txfm_param) {
|
||||
__m128i in[16];
|
||||
int tx_type = txfm_param->tx_type;
|
||||
#if CONFIG_MRC_TX
|
||||
assert(tx_type != MRC_DCT && "Invalid tx type for tx size");
|
||||
#endif
|
||||
|
||||
__m128i *const t = in; // Alias to top 8x8 sub block
|
||||
__m128i *const b = in + 8; // Alias to bottom 8x8 sub block
|
||||
|
|
@ -3045,8 +3069,12 @@ static INLINE void load_buffer_16x8(const int16_t *input, __m128i *in,
|
|||
#define col_16x8_rounding row_8x16_rounding
|
||||
|
||||
void av1_fht16x8_sse2(const int16_t *input, tran_low_t *output, int stride,
|
||||
int tx_type) {
|
||||
TxfmParam *txfm_param) {
|
||||
__m128i in[16];
|
||||
int tx_type = txfm_param->tx_type;
|
||||
#if CONFIG_MRC_TX
|
||||
assert(tx_type != MRC_DCT && "Invalid tx type for tx size");
|
||||
#endif
|
||||
|
||||
__m128i *const l = in; // Alias to left 8x8 sub block
|
||||
__m128i *const r = in + 8; // Alias to right 8x8 sub block, which we store
|
||||
|
|
@ -3355,8 +3383,12 @@ static INLINE void fhalfright32_16col(__m128i *tl, __m128i *tr, __m128i *bl,
|
|||
// For 16x32, this means the input is a 2x2 grid of such blocks.
|
||||
// For 32x16, it means the input is a 4x1 grid.
|
||||
void av1_fht16x32_sse2(const int16_t *input, tran_low_t *output, int stride,
|
||||
int tx_type) {
|
||||
TxfmParam *txfm_param) {
|
||||
__m128i intl[16], intr[16], inbl[16], inbr[16];
|
||||
int tx_type = txfm_param->tx_type;
|
||||
#if CONFIG_MRC_TX
|
||||
assert(tx_type != MRC_DCT && "Invalid tx type for tx size");
|
||||
#endif
|
||||
|
||||
switch (tx_type) {
|
||||
case DCT_DCT:
|
||||
|
|
@ -3544,8 +3576,12 @@ static INLINE void write_buffer_32x16(tran_low_t *output, __m128i *res0,
|
|||
}
|
||||
|
||||
void av1_fht32x16_sse2(const int16_t *input, tran_low_t *output, int stride,
|
||||
int tx_type) {
|
||||
TxfmParam *txfm_param) {
|
||||
__m128i in0[16], in1[16], in2[16], in3[16];
|
||||
int tx_type = txfm_param->tx_type;
|
||||
#if CONFIG_MRC_TX
|
||||
assert(tx_type != MRC_DCT && "Invalid tx type for tx size");
|
||||
#endif
|
||||
|
||||
load_buffer_32x16(input, in0, in1, in2, in3, stride, 0, 0);
|
||||
switch (tx_type) {
|
||||
|
|
@ -3784,8 +3820,12 @@ static INLINE void write_buffer_32x32(__m128i *in0, __m128i *in1, __m128i *in2,
|
|||
}
|
||||
|
||||
void av1_fht32x32_sse2(const int16_t *input, tran_low_t *output, int stride,
|
||||
int tx_type) {
|
||||
TxfmParam *txfm_param) {
|
||||
__m128i in0[32], in1[32], in2[32], in3[32];
|
||||
int tx_type = txfm_param->tx_type;
|
||||
#if CONFIG_MRC_TX
|
||||
assert(tx_type != MRC_DCT && "No 32x32 sse2 MRC_DCT implementation");
|
||||
#endif
|
||||
|
||||
load_buffer_32x32(input, in0, in1, in2, in3, stride, 0, 0);
|
||||
switch (tx_type) {
|
||||
|
|
|
|||
|
|
@ -14,7 +14,20 @@
|
|||
#include "./av1_rtcd.h"
|
||||
#include "aom/aom_integer.h"
|
||||
|
||||
int64_t av1_block_error_avx2(const int16_t *coeff, const int16_t *dqcoeff,
|
||||
static INLINE void read_coeff(const tran_low_t *coeff, intptr_t offset,
|
||||
__m256i *c) {
|
||||
const tran_low_t *addr = coeff + offset;
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
const __m256i x0 = _mm256_loadu_si256((const __m256i *)addr);
|
||||
const __m256i x1 = _mm256_loadu_si256((const __m256i *)addr + 1);
|
||||
const __m256i y = _mm256_packs_epi32(x0, x1);
|
||||
*c = _mm256_permute4x64_epi64(y, 0xD8);
|
||||
#else
|
||||
*c = _mm256_loadu_si256((const __m256i *)addr);
|
||||
#endif
|
||||
}
|
||||
|
||||
int64_t av1_block_error_avx2(const tran_low_t *coeff, const tran_low_t *dqcoeff,
|
||||
intptr_t block_size, int64_t *ssz) {
|
||||
__m256i sse_reg, ssz_reg, coeff_reg, dqcoeff_reg;
|
||||
__m256i exp_dqcoeff_lo, exp_dqcoeff_hi, exp_coeff_lo, exp_coeff_hi;
|
||||
|
|
@ -22,16 +35,16 @@ int64_t av1_block_error_avx2(const int16_t *coeff, const int16_t *dqcoeff,
|
|||
__m128i sse_reg128, ssz_reg128;
|
||||
int64_t sse;
|
||||
int i;
|
||||
const __m256i zero_reg = _mm256_set1_epi16(0);
|
||||
const __m256i zero_reg = _mm256_setzero_si256();
|
||||
|
||||
// init sse and ssz registerd to zero
|
||||
sse_reg = _mm256_set1_epi16(0);
|
||||
ssz_reg = _mm256_set1_epi16(0);
|
||||
sse_reg = _mm256_setzero_si256();
|
||||
ssz_reg = _mm256_setzero_si256();
|
||||
|
||||
for (i = 0; i < block_size; i += 16) {
|
||||
// load 32 bytes from coeff and dqcoeff
|
||||
coeff_reg = _mm256_loadu_si256((const __m256i *)(coeff + i));
|
||||
dqcoeff_reg = _mm256_loadu_si256((const __m256i *)(dqcoeff + i));
|
||||
read_coeff(coeff, i, &coeff_reg);
|
||||
read_coeff(dqcoeff, i, &dqcoeff_reg);
|
||||
// dqcoeff - coeff
|
||||
dqcoeff_reg = _mm256_sub_epi16(dqcoeff_reg, coeff_reg);
|
||||
// madd (dqcoeff - coeff)
|
||||
|
|
|
|||
|
|
@ -113,25 +113,13 @@ static void fdct4x4_sse4_1(__m128i *in, int bit) {
|
|||
in[3] = _mm_unpackhi_epi64(v1, v3);
|
||||
}
|
||||
|
||||
static INLINE void write_buffer_4x4(__m128i *res, tran_low_t *output) {
|
||||
static INLINE void write_buffer_4x4(__m128i *res, int32_t *output) {
|
||||
_mm_store_si128((__m128i *)(output + 0 * 4), res[0]);
|
||||
_mm_store_si128((__m128i *)(output + 1 * 4), res[1]);
|
||||
_mm_store_si128((__m128i *)(output + 2 * 4), res[2]);
|
||||
_mm_store_si128((__m128i *)(output + 3 * 4), res[3]);
|
||||
}
|
||||
|
||||
// Note:
|
||||
// We implement av1_fwd_txfm2d_4x4(). This function is kept here since
|
||||
// av1_highbd_fht4x4_c() is not removed yet
|
||||
void av1_highbd_fht4x4_sse4_1(const int16_t *input, tran_low_t *output,
|
||||
int stride, int tx_type) {
|
||||
(void)input;
|
||||
(void)output;
|
||||
(void)stride;
|
||||
(void)tx_type;
|
||||
assert(0);
|
||||
}
|
||||
|
||||
static void fadst4x4_sse4_1(__m128i *in, int bit) {
|
||||
const int32_t *cospi = cospi_arr(bit);
|
||||
const __m128i cospi8 = _mm_set1_epi32(cospi[8]);
|
||||
|
|
@ -416,7 +404,7 @@ static INLINE void col_txfm_8x8_rounding(__m128i *in, int shift) {
|
|||
in[15] = _mm_srai_epi32(in[15], shift);
|
||||
}
|
||||
|
||||
static INLINE void write_buffer_8x8(const __m128i *res, tran_low_t *output) {
|
||||
static INLINE void write_buffer_8x8(const __m128i *res, int32_t *output) {
|
||||
_mm_store_si128((__m128i *)(output + 0 * 4), res[0]);
|
||||
_mm_store_si128((__m128i *)(output + 1 * 4), res[1]);
|
||||
_mm_store_si128((__m128i *)(output + 2 * 4), res[2]);
|
||||
|
|
@ -1800,7 +1788,7 @@ static void col_txfm_16x16_rounding(__m128i *in, int shift) {
|
|||
col_txfm_8x8_rounding(&in[48], shift);
|
||||
}
|
||||
|
||||
static void write_buffer_16x16(const __m128i *in, tran_low_t *output) {
|
||||
static void write_buffer_16x16(const __m128i *in, int32_t *output) {
|
||||
const int size_8x8 = 16 * 4;
|
||||
write_buffer_8x8(&in[0], output);
|
||||
output += size_8x8;
|
||||
|
|
|
|||
|
|
@ -18,51 +18,6 @@
|
|||
#include "aom_dsp/txfm_common.h"
|
||||
#include "aom_dsp/x86/txfm_common_avx2.h"
|
||||
|
||||
static int32_t get_16x16_sum(const int16_t *input, int stride) {
|
||||
__m256i r0, r1, r2, r3, u0, u1;
|
||||
__m256i zero = _mm256_setzero_si256();
|
||||
__m256i sum = _mm256_setzero_si256();
|
||||
const int16_t *blockBound = input + (stride << 4);
|
||||
__m128i v0, v1;
|
||||
|
||||
while (input < blockBound) {
|
||||
r0 = _mm256_loadu_si256((__m256i const *)input);
|
||||
r1 = _mm256_loadu_si256((__m256i const *)(input + stride));
|
||||
r2 = _mm256_loadu_si256((__m256i const *)(input + 2 * stride));
|
||||
r3 = _mm256_loadu_si256((__m256i const *)(input + 3 * stride));
|
||||
|
||||
u0 = _mm256_add_epi16(r0, r1);
|
||||
u1 = _mm256_add_epi16(r2, r3);
|
||||
sum = _mm256_add_epi16(sum, u0);
|
||||
sum = _mm256_add_epi16(sum, u1);
|
||||
|
||||
input += stride << 2;
|
||||
}
|
||||
|
||||
// unpack 16 int16_t into 2x8 int32_t
|
||||
u0 = _mm256_unpacklo_epi16(zero, sum);
|
||||
u1 = _mm256_unpackhi_epi16(zero, sum);
|
||||
u0 = _mm256_srai_epi32(u0, 16);
|
||||
u1 = _mm256_srai_epi32(u1, 16);
|
||||
sum = _mm256_add_epi32(u0, u1);
|
||||
|
||||
u0 = _mm256_srli_si256(sum, 8);
|
||||
u1 = _mm256_add_epi32(sum, u0);
|
||||
|
||||
v0 = _mm_add_epi32(_mm256_extracti128_si256(u1, 1),
|
||||
_mm256_castsi256_si128(u1));
|
||||
v1 = _mm_srli_si128(v0, 4);
|
||||
v0 = _mm_add_epi32(v0, v1);
|
||||
return (int32_t)_mm_extract_epi32(v0, 0);
|
||||
}
|
||||
|
||||
void aom_fdct16x16_1_avx2(const int16_t *input, tran_low_t *output,
|
||||
int stride) {
|
||||
int32_t dc = get_16x16_sum(input, stride);
|
||||
output[0] = (tran_low_t)(dc >> 1);
|
||||
_mm256_zeroupper();
|
||||
}
|
||||
|
||||
static INLINE void load_buffer_16x16(const int16_t *input, int stride,
|
||||
int flipud, int fliplr, __m256i *in) {
|
||||
if (!flipud) {
|
||||
|
|
@ -959,8 +914,12 @@ static void fidtx16_avx2(__m256i *in) {
|
|||
#endif
|
||||
|
||||
void av1_fht16x16_avx2(const int16_t *input, tran_low_t *output, int stride,
|
||||
int tx_type) {
|
||||
TxfmParam *txfm_param) {
|
||||
__m256i in[16];
|
||||
int tx_type = txfm_param->tx_type;
|
||||
#if CONFIG_MRC_TX
|
||||
assert(tx_type != MRC_DCT && "Invalid tx type for tx size");
|
||||
#endif
|
||||
|
||||
switch (tx_type) {
|
||||
case DCT_DCT:
|
||||
|
|
@ -1084,22 +1043,6 @@ void av1_fht16x16_avx2(const int16_t *input, tran_low_t *output, int stride,
|
|||
_mm256_zeroupper();
|
||||
}
|
||||
|
||||
void aom_fdct32x32_1_avx2(const int16_t *input, tran_low_t *output,
|
||||
int stride) {
|
||||
// left and upper corner
|
||||
int32_t sum = get_16x16_sum(input, stride);
|
||||
// right and upper corner
|
||||
sum += get_16x16_sum(input + 16, stride);
|
||||
// left and lower corner
|
||||
sum += get_16x16_sum(input + (stride << 4), stride);
|
||||
// right and lower corner
|
||||
sum += get_16x16_sum(input + (stride << 4) + 16, stride);
|
||||
|
||||
sum >>= 3;
|
||||
output[0] = (tran_low_t)sum;
|
||||
_mm256_zeroupper();
|
||||
}
|
||||
|
||||
static void mm256_vectors_swap(__m256i *a0, __m256i *a1, const int size) {
|
||||
int i = 0;
|
||||
__m256i temp;
|
||||
|
|
@ -1570,9 +1513,13 @@ static void fidtx32_avx2(__m256i *in0, __m256i *in1) {
|
|||
#endif
|
||||
|
||||
void av1_fht32x32_avx2(const int16_t *input, tran_low_t *output, int stride,
|
||||
int tx_type) {
|
||||
TxfmParam *txfm_param) {
|
||||
__m256i in0[32]; // left 32 columns
|
||||
__m256i in1[32]; // right 32 columns
|
||||
int tx_type = txfm_param->tx_type;
|
||||
#if CONFIG_MRC_TX
|
||||
assert(tx_type != MRC_DCT && "No avx2 32x32 implementation of MRC_DCT");
|
||||
#endif
|
||||
|
||||
switch (tx_type) {
|
||||
case DCT_DCT:
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue