mirror of
https://repo.dactyloidae.xyz/Dactyloidae/UXP.git
synced 2026-10-06 15:27:30 +09:00
Update aom to commit id e87fb2378f01103d5d6e477a4ef6892dc714e614
This commit is contained in:
parent
debbee1e2a
commit
992c6637e3
429 changed files with 76047 additions and 40937 deletions
160
third_party/aom/av1/av1.cmake
vendored
160
third_party/aom/av1/av1.cmake
vendored
|
|
@ -89,7 +89,8 @@ set(AOM_AV1_DECODER_SOURCES
|
|||
"${AOM_ROOT}/av1/decoder/dsubexp.c"
|
||||
"${AOM_ROOT}/av1/decoder/dsubexp.h"
|
||||
"${AOM_ROOT}/av1/decoder/dthread.c"
|
||||
"${AOM_ROOT}/av1/decoder/dthread.h")
|
||||
"${AOM_ROOT}/av1/decoder/dthread.h"
|
||||
"${AOM_ROOT}/av1/decoder/symbolrate.h")
|
||||
|
||||
set(AOM_AV1_ENCODER_SOURCES
|
||||
"${AOM_ROOT}/av1/av1_cx_iface.c"
|
||||
|
|
@ -123,6 +124,8 @@ set(AOM_AV1_ENCODER_SOURCES
|
|||
"${AOM_ROOT}/av1/encoder/extend.h"
|
||||
"${AOM_ROOT}/av1/encoder/firstpass.c"
|
||||
"${AOM_ROOT}/av1/encoder/firstpass.h"
|
||||
"${AOM_ROOT}/av1/encoder/hash.c"
|
||||
"${AOM_ROOT}/av1/encoder/hash.h"
|
||||
"${AOM_ROOT}/av1/encoder/hybrid_fwd_txfm.c"
|
||||
"${AOM_ROOT}/av1/encoder/hybrid_fwd_txfm.h"
|
||||
"${AOM_ROOT}/av1/encoder/lookahead.c"
|
||||
|
|
@ -131,6 +134,8 @@ set(AOM_AV1_ENCODER_SOURCES
|
|||
"${AOM_ROOT}/av1/encoder/mbgraph.h"
|
||||
"${AOM_ROOT}/av1/encoder/mcomp.c"
|
||||
"${AOM_ROOT}/av1/encoder/mcomp.h"
|
||||
"${AOM_ROOT}/av1/encoder/palette.c"
|
||||
"${AOM_ROOT}/av1/encoder/palette.h"
|
||||
"${AOM_ROOT}/av1/encoder/picklpf.c"
|
||||
"${AOM_ROOT}/av1/encoder/picklpf.h"
|
||||
"${AOM_ROOT}/av1/encoder/ratectrl.c"
|
||||
|
|
@ -167,11 +172,6 @@ set(AOM_AV1_COMMON_INTRIN_AVX2
|
|||
"${AOM_ROOT}/av1/common/x86/highbd_inv_txfm_avx2.c"
|
||||
"${AOM_ROOT}/av1/common/x86/hybrid_inv_txfm_avx2.c")
|
||||
|
||||
set(AOM_AV1_COMMON_INTRIN_DSPR2
|
||||
"${AOM_ROOT}/av1/common/mips/dspr2/av1_itrans16_dspr2.c"
|
||||
"${AOM_ROOT}/av1/common/mips/dspr2/av1_itrans4_dspr2.c"
|
||||
"${AOM_ROOT}/av1/common/mips/dspr2/av1_itrans8_dspr2.c")
|
||||
|
||||
set(AOM_AV1_COMMON_INTRIN_MSA
|
||||
"${AOM_ROOT}/av1/common/mips/msa/av1_idct16x16_msa.c"
|
||||
"${AOM_ROOT}/av1/common/mips/msa/av1_idct4x4_msa.c"
|
||||
|
|
@ -190,9 +190,6 @@ set(AOM_AV1_ENCODER_INTRIN_SSE2
|
|||
set(AOM_AV1_ENCODER_ASM_SSSE3_X86_64
|
||||
"${AOM_ROOT}/av1/encoder/x86/av1_quantize_ssse3_x86_64.asm")
|
||||
|
||||
set(AOM_AV1_ENCODER_INTRIN_SSSE3
|
||||
"${AOM_ROOT}/av1/encoder/x86/dct_ssse3.c")
|
||||
|
||||
set(AOM_AV1_ENCODER_INTRIN_SSE4_1
|
||||
${AOM_AV1_ENCODER_INTRIN_SSE4_1}
|
||||
"${AOM_ROOT}/av1/encoder/x86/av1_highbd_quantize_sse4.c"
|
||||
|
|
@ -222,7 +219,6 @@ if (CONFIG_HIGHBITDEPTH)
|
|||
else ()
|
||||
set(AOM_AV1_COMMON_INTRIN_NEON
|
||||
${AOM_AV1_COMMON_INTRIN_NEON}
|
||||
"${AOM_ROOT}/av1/encoder/arm/neon/dct_neon.c"
|
||||
"${AOM_ROOT}/av1/common/arm/neon/iht4x4_add_neon.c"
|
||||
"${AOM_ROOT}/av1/common/arm/neon/iht8x8_add_neon.c")
|
||||
|
||||
|
|
@ -234,14 +230,10 @@ endif ()
|
|||
if (CONFIG_CDEF)
|
||||
set(AOM_AV1_COMMON_SOURCES
|
||||
${AOM_AV1_COMMON_SOURCES}
|
||||
"${AOM_ROOT}/av1/common/clpf.c"
|
||||
"${AOM_ROOT}/av1/common/clpf_simd.h"
|
||||
"${AOM_ROOT}/av1/common/cdef_simd.h"
|
||||
"${AOM_ROOT}/av1/common/cdef.c"
|
||||
"${AOM_ROOT}/av1/common/cdef.h"
|
||||
"${AOM_ROOT}/av1/common/od_dering.c"
|
||||
"${AOM_ROOT}/av1/common/od_dering.h"
|
||||
"${AOM_ROOT}/av1/common/od_dering_simd.h")
|
||||
"${AOM_ROOT}/av1/common/cdef_block.c"
|
||||
"${AOM_ROOT}/av1/common/cdef_block.h")
|
||||
|
||||
set(AOM_AV1_ENCODER_SOURCES
|
||||
${AOM_AV1_ENCODER_SOURCES}
|
||||
|
|
@ -249,32 +241,70 @@ if (CONFIG_CDEF)
|
|||
|
||||
set(AOM_AV1_COMMON_INTRIN_SSE2
|
||||
${AOM_AV1_COMMON_INTRIN_SSE2}
|
||||
"${AOM_ROOT}/av1/common/clpf_sse2.c"
|
||||
"${AOM_ROOT}/av1/common/od_dering_sse2.c")
|
||||
"${AOM_ROOT}/av1/common/cdef_block_sse2.c")
|
||||
|
||||
set(AOM_AV1_COMMON_INTRIN_SSSE3
|
||||
${AOM_AV1_COMMON_INTRIN_SSSE3}
|
||||
"${AOM_ROOT}/av1/common/clpf_ssse3.c"
|
||||
"${AOM_ROOT}/av1/common/od_dering_ssse3.c")
|
||||
"${AOM_ROOT}/av1/common/cdef_block_ssse3.c")
|
||||
|
||||
set(AOM_AV1_COMMON_INTRIN_SSE4_1
|
||||
${AOM_AV1_COMMON_INTRIN_SSE4_1}
|
||||
"${AOM_ROOT}/av1/common/clpf_sse4.c"
|
||||
"${AOM_ROOT}/av1/common/od_dering_sse4.c")
|
||||
"${AOM_ROOT}/av1/common/cdef_block_sse4.c")
|
||||
|
||||
set(AOM_AV1_COMMON_INTRIN_AVX2
|
||||
${AOM_AV1_COMMON_INTRIN_AVX2}
|
||||
"${AOM_ROOT}/av1/common/cdef_block_avx2.c")
|
||||
|
||||
set(AOM_AV1_COMMON_INTRIN_NEON
|
||||
${AOM_AV1_COMMON_INTRIN_NEON}
|
||||
"${AOM_ROOT}/av1/common/clpf_neon.c"
|
||||
"${AOM_ROOT}/av1/common/od_dering_neon.c")
|
||||
"${AOM_ROOT}/av1/common/cdef_block_neon.c")
|
||||
|
||||
if (NOT CONFIG_CDEF_SINGLEPASS)
|
||||
set(AOM_AV1_COMMON_SOURCES
|
||||
${AOM_AV1_COMMON_SOURCES}
|
||||
"${AOM_ROOT}/av1/common/clpf.c"
|
||||
"${AOM_ROOT}/av1/common/clpf_simd.h"
|
||||
"${AOM_ROOT}/av1/common/cdef_block_simd.h")
|
||||
|
||||
set(AOM_AV1_COMMON_INTRIN_SSE2
|
||||
${AOM_AV1_COMMON_INTRIN_SSE2}
|
||||
"${AOM_ROOT}/av1/common/clpf_sse2.c")
|
||||
|
||||
set(AOM_AV1_COMMON_INTRIN_SSSE3
|
||||
${AOM_AV1_COMMON_INTRIN_SSSE3}
|
||||
"${AOM_ROOT}/av1/common/clpf_ssse3.c")
|
||||
|
||||
set(AOM_AV1_COMMON_INTRIN_SSE4_1
|
||||
${AOM_AV1_COMMON_INTRIN_SSE4_1}
|
||||
"${AOM_ROOT}/av1/common/clpf_sse4.c")
|
||||
|
||||
set(AOM_AV1_COMMON_INTRIN_NEON
|
||||
${AOM_AV1_COMMON_INTRIN_NEON}
|
||||
"${AOM_ROOT}/av1/common/clpf_neon.c")
|
||||
endif ()
|
||||
endif ()
|
||||
|
||||
if (CONFIG_CONVOLVE_ROUND)
|
||||
set(AOM_AV1_COMMON_INTRIN_SSE2
|
||||
${AOM_AV1_COMMON_INTRIN_SSE2}
|
||||
"${AOM_ROOT}/av1/common/x86/convolve_2d_sse2.c")
|
||||
if (CONFIG_HIGHBITDEPTH)
|
||||
set(AOM_AV1_COMMON_INTRIN_SSSE3
|
||||
${AOM_AV1_COMMON_INTRIN_SSSE3}
|
||||
"${AOM_ROOT}/av1/common/x86/highbd_convolve_2d_ssse3.c")
|
||||
endif ()
|
||||
|
||||
if(NOT CONFIG_COMPOUND_ROUND)
|
||||
set(AOM_AV1_COMMON_INTRIN_SSE4_1
|
||||
${AOM_AV1_COMMON_INTRIN_SSE4_1}
|
||||
"${AOM_ROOT}/av1/common/x86/av1_convolve_scale_sse4.c")
|
||||
endif()
|
||||
|
||||
set(AOM_AV1_COMMON_INTRIN_AVX2
|
||||
${AOM_AV1_COMMON_INTRIN_AVX2}
|
||||
"${AOM_ROOT}/av1/common/x86/convolve_avx2.c")
|
||||
endif ()
|
||||
|
||||
if (CONFIG_EXT_INTER)
|
||||
set(AOM_AV1_ENCODER_SOURCES
|
||||
${AOM_AV1_ENCODER_SOURCES}
|
||||
"${AOM_ROOT}/av1/encoder/wedge_utils.c")
|
||||
|
|
@ -282,7 +312,6 @@ if (CONFIG_EXT_INTER)
|
|||
set(AOM_AV1_ENCODER_INTRIN_SSE2
|
||||
${AOM_AV1_ENCODER_INTRIN_SSE2}
|
||||
"${AOM_ROOT}/av1/encoder/x86/wedge_utils_sse2.c")
|
||||
endif ()
|
||||
|
||||
if (CONFIG_FILTER_INTRA)
|
||||
set(AOM_AV1_COMMON_INTRIN_SSE4_1
|
||||
|
|
@ -297,6 +326,13 @@ if (CONFIG_ACCOUNTING)
|
|||
"${AOM_ROOT}/av1/decoder/accounting.h")
|
||||
endif ()
|
||||
|
||||
if (CONFIG_BGSPRITE)
|
||||
set(AOM_AV1_ENCODER_SOURCES
|
||||
${AOM_AV1_ENCODER_SOURCES}
|
||||
"${AOM_ROOT}/av1/encoder/bgsprite.c"
|
||||
"${AOM_ROOT}/av1/encoder/bgsprite.h")
|
||||
endif ()
|
||||
|
||||
if (CONFIG_GLOBAL_MOTION)
|
||||
set(AOM_AV1_ENCODER_SOURCES
|
||||
${AOM_AV1_ENCODER_SOURCES}
|
||||
|
|
@ -331,11 +367,21 @@ if (CONFIG_INTERNAL_STATS)
|
|||
"${AOM_ROOT}/av1/encoder/blockiness.c")
|
||||
endif ()
|
||||
|
||||
if (CONFIG_PALETTE)
|
||||
if (CONFIG_LV_MAP)
|
||||
set(AOM_AV1_COMMON_SOURCES
|
||||
${AOM_AV1_COMMON_SOURCES}
|
||||
"${AOM_ROOT}/av1/common/txb_common.c"
|
||||
"${AOM_ROOT}/av1/common/txb_common.h")
|
||||
|
||||
set(AOM_AV1_DECODER_SOURCES
|
||||
${AOM_AV1_DECODER_SOURCES}
|
||||
"${AOM_ROOT}/av1/decoder/decodetxb.c"
|
||||
"${AOM_ROOT}/av1/decoder/decodetxb.h")
|
||||
|
||||
set(AOM_AV1_ENCODER_SOURCES
|
||||
${AOM_AV1_ENCODER_SOURCES}
|
||||
"${AOM_ROOT}/av1/encoder/palette.c"
|
||||
"${AOM_ROOT}/av1/encoder/palette.h")
|
||||
"${AOM_ROOT}/av1/encoder/encodetxb.c"
|
||||
"${AOM_ROOT}/av1/encoder/encodetxb.h")
|
||||
endif ()
|
||||
|
||||
if (CONFIG_CFL)
|
||||
|
|
@ -361,6 +407,19 @@ if (CONFIG_LOOP_RESTORATION)
|
|||
"${AOM_ROOT}/av1/encoder/pickrst.h")
|
||||
endif ()
|
||||
|
||||
if (CONFIG_INTRA_EDGE)
|
||||
set(AOM_AV1_COMMON_INTRIN_SSE4_1
|
||||
${AOM_AV1_COMMON_INTRIN_SSE4_1}
|
||||
"${AOM_ROOT}/av1/common/x86/intra_edge_sse4.c")
|
||||
endif ()
|
||||
|
||||
if (CONFIG_NCOBMC_ADAPT_WEIGHT)
|
||||
set(AOM_AV1_COMMON_SOURCES
|
||||
${AOM_AV1_COMMON_SOURCES}
|
||||
"${AOM_ROOT}/av1/common/ncobmc_kernels.c"
|
||||
"${AOM_ROOT}/av1/common/ncobmc_kernels.h")
|
||||
endif ()
|
||||
|
||||
if (CONFIG_PVQ)
|
||||
set(AOM_AV1_COMMON_SOURCES
|
||||
${AOM_AV1_COMMON_SOURCES}
|
||||
|
|
@ -417,9 +476,6 @@ if (CONFIG_PVQ)
|
|||
${AOM_AV1_DECODER_INTRIN_SSE2}
|
||||
"${AOM_ROOT}/av1/encoder/x86/dct_intrin_sse2.c")
|
||||
|
||||
set(AOM_AV1_DECODER_INTRIN_SSSE3
|
||||
${AOM_AV1_DECODER_INTRIN_SSSE3}
|
||||
"${AOM_ROOT}/av1/encoder/x86/dct_ssse3.c")
|
||||
endif ()
|
||||
endif ()
|
||||
|
||||
|
|
@ -444,6 +500,28 @@ if (CONFIG_WARPED_MOTION OR CONFIG_GLOBAL_MOTION)
|
|||
endif ()
|
||||
endif ()
|
||||
|
||||
if (CONFIG_HASH_ME)
|
||||
set(AOM_AV1_ENCODER_SOURCES
|
||||
${AOM_AV1_ENCODER_SOURCES}
|
||||
"${AOM_ROOT}/av1/encoder/hash_motion.h"
|
||||
"${AOM_ROOT}/av1/encoder/hash_motion.c"
|
||||
"${AOM_ROOT}/third_party/vector/vector.h"
|
||||
"${AOM_ROOT}/third_party/vector/vector.c")
|
||||
endif ()
|
||||
|
||||
if (CONFIG_Q_ADAPT_PROBS)
|
||||
set(AOM_AV1_COMMON_SOURCES
|
||||
${AOM_AV1_COMMON_SOURCES}
|
||||
"${AOM_ROOT}/av1/common/token_cdfs.h")
|
||||
endif ()
|
||||
|
||||
if (CONFIG_XIPHRC)
|
||||
set(AOM_AV1_ENCODER_SOURCES
|
||||
${AOM_AV1_ENCODER_SOURCES}
|
||||
"${AOM_ROOT}/av1/encoder/ratectrl_xiph.c"
|
||||
"${AOM_ROOT}/av1/encoder/ratectrl_xiph.h")
|
||||
endif ()
|
||||
|
||||
# Setup AV1 common/decoder/encoder targets. The libaom target must exist before
|
||||
# this function is called.
|
||||
function (setup_av1_targets)
|
||||
|
|
@ -472,7 +550,7 @@ function (setup_av1_targets)
|
|||
endif ()
|
||||
|
||||
if (HAVE_SSE2)
|
||||
require_flag_nomsvc("-msse2" NO)
|
||||
require_compiler_flag_nomsvc("-msse2" NO)
|
||||
add_intrinsics_object_library("-msse2" "sse2" "aom_av1_common"
|
||||
"AOM_AV1_COMMON_INTRIN_SSE2" "aom")
|
||||
if (CONFIG_AV1_DECODER)
|
||||
|
|
@ -494,7 +572,7 @@ function (setup_av1_targets)
|
|||
endif ()
|
||||
|
||||
if (HAVE_SSSE3)
|
||||
require_flag_nomsvc("-mssse3" NO)
|
||||
require_compiler_flag_nomsvc("-mssse3" NO)
|
||||
add_intrinsics_object_library("-mssse3" "ssse3" "aom_av1_common"
|
||||
"AOM_AV1_COMMON_INTRIN_SSSE3" "aom")
|
||||
|
||||
|
|
@ -504,15 +582,10 @@ function (setup_av1_targets)
|
|||
"AOM_AV1_DECODER_INTRIN_SSSE3" "aom")
|
||||
endif ()
|
||||
endif ()
|
||||
|
||||
if (CONFIG_AV1_ENCODER)
|
||||
add_intrinsics_object_library("-mssse3" "ssse3" "aom_av1_encoder"
|
||||
"AOM_AV1_ENCODER_INTRIN_SSSE3" "aom")
|
||||
endif ()
|
||||
endif ()
|
||||
|
||||
if (HAVE_SSE4_1)
|
||||
require_flag_nomsvc("-msse4.1" NO)
|
||||
require_compiler_flag_nomsvc("-msse4.1" NO)
|
||||
add_intrinsics_object_library("-msse4.1" "sse4" "aom_av1_common"
|
||||
"AOM_AV1_COMMON_INTRIN_SSE4_1" "aom")
|
||||
|
||||
|
|
@ -530,7 +603,7 @@ function (setup_av1_targets)
|
|||
endif ()
|
||||
|
||||
if (HAVE_AVX2)
|
||||
require_flag_nomsvc("-mavx2" NO)
|
||||
require_compiler_flag_nomsvc("-mavx2" NO)
|
||||
add_intrinsics_object_library("-mavx2" "avx2" "aom_av1_common"
|
||||
"AOM_AV1_COMMON_INTRIN_AVX2" "aom")
|
||||
|
||||
|
|
@ -556,11 +629,6 @@ function (setup_av1_targets)
|
|||
endif ()
|
||||
endif ()
|
||||
|
||||
if (HAVE_DSPR2)
|
||||
add_intrinsics_object_library("" "dspr2" "aom_av1_common"
|
||||
"AOM_AV1_COMMON_INTRIN_DSPR2" "aom")
|
||||
endif ()
|
||||
|
||||
if (HAVE_MSA)
|
||||
add_intrinsics_object_library("" "msa" "aom_av1_common"
|
||||
"AOM_AV1_COMMON_INTRIN_MSA" "aom")
|
||||
|
|
|
|||
45
third_party/aom/av1/av1_common.mk
vendored
45
third_party/aom/av1/av1_common.mk
vendored
|
|
@ -75,6 +75,9 @@ AV1_COMMON_SRCS-yes += common/av1_inv_txfm2d.c
|
|||
AV1_COMMON_SRCS-yes += common/av1_inv_txfm1d_cfg.h
|
||||
AV1_COMMON_SRCS-$(HAVE_AVX2) += common/x86/convolve_avx2.c
|
||||
AV1_COMMON_SRCS-$(HAVE_SSSE3) += common/x86/av1_convolve_ssse3.c
|
||||
ifeq ($(CONFIG_CONVOLVE_ROUND)x$(CONFIG_COMPOUND_ROUND),yesx)
|
||||
AV1_COMMON_SRCS-$(HAVE_SSE4_1) += common/x86/av1_convolve_scale_sse4.c
|
||||
endif
|
||||
ifeq ($(CONFIG_HIGHBITDEPTH),yes)
|
||||
AV1_COMMON_SRCS-$(HAVE_SSE4_1) += common/x86/av1_highbd_convolve_sse4.c
|
||||
endif
|
||||
|
|
@ -85,25 +88,31 @@ AV1_COMMON_SRCS-yes += common/restoration.h
|
|||
AV1_COMMON_SRCS-yes += common/restoration.c
|
||||
AV1_COMMON_SRCS-$(HAVE_SSE4_1) += common/x86/selfguided_sse4.c
|
||||
endif
|
||||
ifeq ($(CONFIG_INTRA_EDGE),yes)
|
||||
AV1_COMMON_SRCS-$(HAVE_SSE4_1) += common/x86/intra_edge_sse4.c
|
||||
endif
|
||||
ifeq (yes,$(filter $(CONFIG_GLOBAL_MOTION) $(CONFIG_WARPED_MOTION),yes))
|
||||
AV1_COMMON_SRCS-yes += common/warped_motion.h
|
||||
AV1_COMMON_SRCS-yes += common/warped_motion.c
|
||||
endif
|
||||
ifeq ($(CONFIG_CDEF),yes)
|
||||
ifeq ($(CONFIG_CDEF_SINGLEPASS),yes)
|
||||
AV1_COMMON_SRCS-$(HAVE_AVX2) += common/cdef_block_avx2.c
|
||||
else
|
||||
AV1_COMMON_SRCS-yes += common/clpf.c
|
||||
AV1_COMMON_SRCS-yes += common/clpf_simd.h
|
||||
AV1_COMMON_SRCS-yes += common/cdef_simd.h
|
||||
AV1_COMMON_SRCS-$(HAVE_SSE2) += common/clpf_sse2.c
|
||||
AV1_COMMON_SRCS-$(HAVE_SSSE3) += common/clpf_ssse3.c
|
||||
AV1_COMMON_SRCS-$(HAVE_SSE4_1) += common/clpf_sse4.c
|
||||
AV1_COMMON_SRCS-$(HAVE_NEON) += common/clpf_neon.c
|
||||
AV1_COMMON_SRCS-$(HAVE_SSE2) += common/od_dering_sse2.c
|
||||
AV1_COMMON_SRCS-$(HAVE_SSSE3) += common/od_dering_ssse3.c
|
||||
AV1_COMMON_SRCS-$(HAVE_SSE4_1) += common/od_dering_sse4.c
|
||||
AV1_COMMON_SRCS-$(HAVE_NEON) += common/od_dering_neon.c
|
||||
AV1_COMMON_SRCS-yes += common/od_dering.c
|
||||
AV1_COMMON_SRCS-yes += common/od_dering.h
|
||||
AV1_COMMON_SRCS-yes += common/od_dering_simd.h
|
||||
endif
|
||||
AV1_COMMON_SRCS-$(HAVE_SSE2) += common/cdef_block_sse2.c
|
||||
AV1_COMMON_SRCS-$(HAVE_SSSE3) += common/cdef_block_ssse3.c
|
||||
AV1_COMMON_SRCS-$(HAVE_SSE4_1) += common/cdef_block_sse4.c
|
||||
AV1_COMMON_SRCS-$(HAVE_NEON) += common/cdef_block_neon.c
|
||||
AV1_COMMON_SRCS-yes += common/cdef_block.c
|
||||
AV1_COMMON_SRCS-yes += common/cdef_block.h
|
||||
AV1_COMMON_SRCS-yes += common/cdef_block_simd.h
|
||||
AV1_COMMON_SRCS-yes += common/cdef.c
|
||||
AV1_COMMON_SRCS-yes += common/cdef.h
|
||||
endif
|
||||
|
|
@ -115,6 +124,10 @@ AV1_COMMON_SRCS-yes += common/cfl.h
|
|||
AV1_COMMON_SRCS-yes += common/cfl.c
|
||||
endif
|
||||
|
||||
ifeq ($(CONFIG_MOTION_VAR),yes)
|
||||
AV1_COMMON_SRCS-yes += common/obmc.h
|
||||
endif
|
||||
|
||||
ifeq ($(CONFIG_PVQ),yes)
|
||||
# PVQ from daala
|
||||
AV1_COMMON_SRCS-yes += common/pvq.c
|
||||
|
|
@ -137,12 +150,6 @@ AV1_COMMON_SRCS-yes += common/pvq_state.h
|
|||
AV1_COMMON_SRCS-yes += common/generic_code.h
|
||||
endif
|
||||
|
||||
ifneq ($(CONFIG_HIGHBITDEPTH),yes)
|
||||
AV1_COMMON_SRCS-$(HAVE_DSPR2) += common/mips/dspr2/av1_itrans4_dspr2.c
|
||||
AV1_COMMON_SRCS-$(HAVE_DSPR2) += common/mips/dspr2/av1_itrans8_dspr2.c
|
||||
AV1_COMMON_SRCS-$(HAVE_DSPR2) += common/mips/dspr2/av1_itrans16_dspr2.c
|
||||
endif
|
||||
|
||||
# common (msa)
|
||||
AV1_COMMON_SRCS-$(HAVE_MSA) += common/mips/msa/av1_idct4x4_msa.c
|
||||
AV1_COMMON_SRCS-$(HAVE_MSA) += common/mips/msa/av1_idct8x8_msa.c
|
||||
|
|
@ -185,4 +192,14 @@ AV1_COMMON_SRCS-$(HAVE_SSSE3) += common/x86/highbd_convolve_2d_ssse3.c
|
|||
endif
|
||||
endif
|
||||
|
||||
|
||||
ifeq ($(CONFIG_Q_ADAPT_PROBS),yes)
|
||||
AV1_COMMON_SRCS-yes += common/token_cdfs.h
|
||||
endif
|
||||
|
||||
ifeq ($(CONFIG_NCOBMC_ADAPT_WEIGHT),yes)
|
||||
AV1_COMMON_SRCS-yes += common/ncobmc_kernels.h
|
||||
AV1_COMMON_SRCS-yes += common/ncobmc_kernels.c
|
||||
endif
|
||||
|
||||
$(eval $(call rtcd_h_template,av1_rtcd,av1/common/av1_rtcd_defs.pl))
|
||||
|
|
|
|||
16
third_party/aom/av1/av1_cx.mk
vendored
16
third_party/aom/av1/av1_cx.mk
vendored
|
|
@ -63,6 +63,7 @@ AV1_CX_SRCS-yes += encoder/lookahead.c
|
|||
AV1_CX_SRCS-yes += encoder/lookahead.h
|
||||
AV1_CX_SRCS-yes += encoder/mcomp.h
|
||||
AV1_CX_SRCS-yes += encoder/encoder.h
|
||||
AV1_CX_SRCS-yes += encoder/random.h
|
||||
AV1_CX_SRCS-yes += encoder/ratectrl.h
|
||||
ifeq ($(CONFIG_XIPHRC),yes)
|
||||
AV1_CX_SRCS-yes += encoder/ratectrl_xiph.h
|
||||
|
|
@ -73,10 +74,9 @@ AV1_CX_SRCS-yes += encoder/tokenize.h
|
|||
AV1_CX_SRCS-yes += encoder/treewriter.h
|
||||
AV1_CX_SRCS-yes += encoder/mcomp.c
|
||||
AV1_CX_SRCS-yes += encoder/encoder.c
|
||||
ifeq ($(CONFIG_PALETTE),yes)
|
||||
AV1_CX_SRCS-yes += encoder/k_means_template.h
|
||||
AV1_CX_SRCS-yes += encoder/palette.h
|
||||
AV1_CX_SRCS-yes += encoder/palette.c
|
||||
endif
|
||||
AV1_CX_SRCS-yes += encoder/picklpf.c
|
||||
AV1_CX_SRCS-yes += encoder/picklpf.h
|
||||
AV1_CX_SRCS-$(CONFIG_LOOP_RESTORATION) += encoder/pickrst.c
|
||||
|
|
@ -107,6 +107,14 @@ AV1_CX_SRCS-yes += encoder/temporal_filter.c
|
|||
AV1_CX_SRCS-yes += encoder/temporal_filter.h
|
||||
AV1_CX_SRCS-yes += encoder/mbgraph.c
|
||||
AV1_CX_SRCS-yes += encoder/mbgraph.h
|
||||
AV1_CX_SRCS-yes += encoder/hash.c
|
||||
AV1_CX_SRCS-yes += encoder/hash.h
|
||||
ifeq ($(CONFIG_HASH_ME),yes)
|
||||
AV1_CX_SRCS-yes += ../third_party/vector/vector.h
|
||||
AV1_CX_SRCS-yes += ../third_party/vector/vector.c
|
||||
AV1_CX_SRCS-yes += encoder/hash_motion.c
|
||||
AV1_CX_SRCS-yes += encoder/hash_motion.h
|
||||
endif
|
||||
ifeq ($(CONFIG_CDEF),yes)
|
||||
AV1_CX_SRCS-yes += encoder/pickcdef.c
|
||||
endif
|
||||
|
|
@ -138,22 +146,18 @@ AV1_CX_SRCS-$(HAVE_SSSE3) += encoder/x86/av1_quantize_ssse3_x86_64.asm
|
|||
endif
|
||||
|
||||
AV1_CX_SRCS-$(HAVE_SSE2) += encoder/x86/dct_intrin_sse2.c
|
||||
AV1_CX_SRCS-$(HAVE_SSSE3) += encoder/x86/dct_ssse3.c
|
||||
AV1_CX_SRCS-$(HAVE_AVX2) += encoder/x86/hybrid_fwd_txfm_avx2.c
|
||||
|
||||
AV1_CX_SRCS-$(HAVE_SSE4_1) += encoder/x86/av1_highbd_quantize_sse4.c
|
||||
|
||||
AV1_CX_SRCS-$(HAVE_SSE4_1) += encoder/x86/highbd_fwd_txfm_sse4.c
|
||||
|
||||
ifeq ($(CONFIG_EXT_INTER),yes)
|
||||
AV1_CX_SRCS-yes += encoder/wedge_utils.c
|
||||
AV1_CX_SRCS-$(HAVE_SSE2) += encoder/x86/wedge_utils_sse2.c
|
||||
endif
|
||||
|
||||
AV1_CX_SRCS-$(HAVE_AVX2) += encoder/x86/error_intrin_avx2.c
|
||||
|
||||
ifneq ($(CONFIG_HIGHBITDEPTH),yes)
|
||||
AV1_CX_SRCS-$(HAVE_NEON) += encoder/arm/neon/dct_neon.c
|
||||
AV1_CX_SRCS-$(HAVE_NEON) += encoder/arm/neon/error_neon.c
|
||||
endif
|
||||
AV1_CX_SRCS-$(HAVE_NEON) += encoder/arm/neon/quantize_neon.c
|
||||
|
|
|
|||
270
third_party/aom/av1/av1_cx_iface.c
vendored
270
third_party/aom/av1/av1_cx_iface.c
vendored
|
|
@ -8,7 +8,6 @@
|
|||
* Media Patent License 1.0 was not distributed with this source code in the
|
||||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
|
||||
|
|
@ -23,6 +22,9 @@
|
|||
#include "av1/encoder/firstpass.h"
|
||||
#include "av1/av1_iface_common.h"
|
||||
|
||||
#define MAG_SIZE (4)
|
||||
#define MAX_INDEX_SIZE (256)
|
||||
|
||||
struct av1_extracfg {
|
||||
int cpu_used; // available cpu percentage in 1/16
|
||||
unsigned int enable_auto_alt_ref;
|
||||
|
|
@ -32,8 +34,8 @@ struct av1_extracfg {
|
|||
unsigned int noise_sensitivity;
|
||||
unsigned int sharpness;
|
||||
unsigned int static_thresh;
|
||||
unsigned int tile_columns;
|
||||
unsigned int tile_rows;
|
||||
unsigned int tile_columns; // log2 number of tile columns
|
||||
unsigned int tile_rows; // log2 number of tile rows
|
||||
#if CONFIG_DEPENDENT_HORZTILES
|
||||
unsigned int dependent_horz_tiles;
|
||||
#endif
|
||||
|
|
@ -54,6 +56,9 @@ struct av1_extracfg {
|
|||
unsigned int enable_qm;
|
||||
unsigned int qm_min;
|
||||
unsigned int qm_max;
|
||||
#endif
|
||||
#if CONFIG_DIST_8X8
|
||||
unsigned int enable_dist_8x8;
|
||||
#endif
|
||||
unsigned int num_tg;
|
||||
unsigned int mtu_size;
|
||||
|
|
@ -69,10 +74,8 @@ struct av1_extracfg {
|
|||
aom_bit_depth_t bit_depth;
|
||||
aom_tune_content content;
|
||||
aom_color_space_t color_space;
|
||||
#if CONFIG_COLORSPACE_HEADERS
|
||||
aom_transfer_function_t transfer_function;
|
||||
aom_chroma_sample_position_t chroma_sample_position;
|
||||
#endif
|
||||
int color_range;
|
||||
int render_width;
|
||||
int render_height;
|
||||
|
|
@ -118,6 +121,9 @@ static struct av1_extracfg default_extra_cfg = {
|
|||
0, // enable_qm
|
||||
DEFAULT_QM_FIRST, // qm_min
|
||||
DEFAULT_QM_LAST, // qm_max
|
||||
#endif
|
||||
#if CONFIG_DIST_8X8
|
||||
0,
|
||||
#endif
|
||||
1, // max number of tile groups
|
||||
0, // mtu_size
|
||||
|
|
@ -129,14 +135,12 @@ static struct av1_extracfg default_extra_cfg = {
|
|||
#if CONFIG_EXT_DELTA_Q
|
||||
NO_DELTA_Q, // deltaq_mode
|
||||
#endif
|
||||
CONFIG_XIPHRC, // frame_periodic_delta_q
|
||||
AOM_BITS_8, // Bit depth
|
||||
AOM_CONTENT_DEFAULT, // content
|
||||
AOM_CS_UNKNOWN, // color space
|
||||
#if CONFIG_COLORSPACE_HEADERS
|
||||
AOM_TF_UNKNOWN, // transfer function
|
||||
AOM_CSP_UNKNOWN, // chroma sample position
|
||||
#endif
|
||||
CONFIG_XIPHRC, // frame_periodic_delta_q
|
||||
AOM_BITS_8, // Bit depth
|
||||
AOM_CONTENT_DEFAULT, // content
|
||||
AOM_CS_UNKNOWN, // color space
|
||||
AOM_TF_UNKNOWN, // transfer function
|
||||
AOM_CSP_UNKNOWN, // chroma sample position
|
||||
0, // color range
|
||||
0, // render width
|
||||
0, // render height
|
||||
|
|
@ -222,9 +226,9 @@ static aom_codec_err_t validate_config(aom_codec_alg_priv_t *ctx,
|
|||
RANGE_CHECK_HI(cfg, rc_max_quantizer, 63);
|
||||
RANGE_CHECK_HI(cfg, rc_min_quantizer, cfg->rc_max_quantizer);
|
||||
RANGE_CHECK_BOOL(extra_cfg, lossless);
|
||||
RANGE_CHECK(extra_cfg, aq_mode, 0, AQ_MODE_COUNT - 1);
|
||||
RANGE_CHECK_HI(extra_cfg, aq_mode, AQ_MODE_COUNT - 1);
|
||||
#if CONFIG_EXT_DELTA_Q
|
||||
RANGE_CHECK(extra_cfg, deltaq_mode, 0, DELTAQ_MODE_COUNT - 1);
|
||||
RANGE_CHECK_HI(extra_cfg, deltaq_mode, DELTAQ_MODE_COUNT - 1);
|
||||
#endif
|
||||
RANGE_CHECK_HI(extra_cfg, frame_periodic_boost, 1);
|
||||
RANGE_CHECK_HI(cfg, g_threads, 64);
|
||||
|
|
@ -246,17 +250,19 @@ static aom_codec_err_t validate_config(aom_codec_alg_priv_t *ctx,
|
|||
(MAX_LAG_BUFFERS - 1));
|
||||
}
|
||||
|
||||
RANGE_CHECK_HI(cfg, rc_resize_mode, RESIZE_DYNAMIC);
|
||||
RANGE_CHECK(cfg, rc_resize_numerator, SCALE_DENOMINATOR / 2,
|
||||
SCALE_DENOMINATOR);
|
||||
RANGE_CHECK(cfg, rc_resize_kf_numerator, SCALE_DENOMINATOR / 2,
|
||||
SCALE_DENOMINATOR);
|
||||
RANGE_CHECK_HI(cfg, rc_resize_mode, RESIZE_MODES - 1);
|
||||
RANGE_CHECK(cfg, rc_resize_denominator, SCALE_NUMERATOR,
|
||||
SCALE_NUMERATOR << 1);
|
||||
RANGE_CHECK(cfg, rc_resize_kf_denominator, SCALE_NUMERATOR,
|
||||
SCALE_NUMERATOR << 1);
|
||||
#if CONFIG_FRAME_SUPERRES
|
||||
RANGE_CHECK_HI(cfg, rc_superres_mode, SUPERRES_DYNAMIC);
|
||||
RANGE_CHECK(cfg, rc_superres_numerator, SCALE_DENOMINATOR / 2,
|
||||
SCALE_DENOMINATOR);
|
||||
RANGE_CHECK(cfg, rc_superres_kf_numerator, SCALE_DENOMINATOR / 2,
|
||||
SCALE_DENOMINATOR);
|
||||
RANGE_CHECK_HI(cfg, rc_superres_mode, SUPERRES_MODES - 1);
|
||||
RANGE_CHECK(cfg, rc_superres_denominator, SCALE_NUMERATOR,
|
||||
SCALE_NUMERATOR << 1);
|
||||
RANGE_CHECK(cfg, rc_superres_kf_denominator, SCALE_NUMERATOR,
|
||||
SCALE_NUMERATOR << 1);
|
||||
RANGE_CHECK(cfg, rc_superres_qthresh, 1, 63);
|
||||
RANGE_CHECK(cfg, rc_superres_kf_qthresh, 1, 63);
|
||||
#endif // CONFIG_FRAME_SUPERRES
|
||||
|
||||
// AV1 does not support a lower bound on the keyframe interval in
|
||||
|
|
@ -299,8 +305,13 @@ static aom_codec_err_t validate_config(aom_codec_alg_priv_t *ctx,
|
|||
#endif // CONFIG_EXT_PARTITION
|
||||
} else {
|
||||
#endif // CONFIG_EXT_TILE
|
||||
#if CONFIG_MAX_TILE
|
||||
RANGE_CHECK_HI(extra_cfg, tile_columns, 6);
|
||||
RANGE_CHECK_HI(extra_cfg, tile_rows, 2);
|
||||
RANGE_CHECK_HI(extra_cfg, tile_rows, 6);
|
||||
#else // CONFIG_MAX_TILE
|
||||
RANGE_CHECK_HI(extra_cfg, tile_columns, 6);
|
||||
RANGE_CHECK_HI(extra_cfg, tile_rows, 2);
|
||||
#endif // CONFIG_MAX_TILE
|
||||
#if CONFIG_EXT_TILE
|
||||
}
|
||||
#endif // CONFIG_EXT_TILE
|
||||
|
|
@ -323,6 +334,14 @@ static aom_codec_err_t validate_config(aom_codec_alg_priv_t *ctx,
|
|||
if (extra_cfg->tuning == AOM_TUNE_SSIM)
|
||||
ERROR("Option --tune=ssim is not currently supported in AV1.");
|
||||
|
||||
// TODO(anybody) : remove this flag when PVQ supports pallete coding tool
|
||||
#if CONFIG_PVQ
|
||||
if (extra_cfg->content == AOM_CONTENT_SCREEN)
|
||||
ERROR(
|
||||
"Option --tune-content=screen is not currently supported when PVQ is "
|
||||
"enabled.");
|
||||
#endif // CONFIG_PVQ
|
||||
|
||||
if (cfg->g_pass == AOM_RC_LAST_PASS) {
|
||||
#if !CONFIG_XIPHRC
|
||||
const size_t packet_sz = sizeof(FIRSTPASS_STATS);
|
||||
|
|
@ -477,7 +496,12 @@ static aom_codec_err_t set_encoder_config(
|
|||
oxcf->qm_minlevel = extra_cfg->qm_min;
|
||||
oxcf->qm_maxlevel = extra_cfg->qm_max;
|
||||
#endif
|
||||
|
||||
#if CONFIG_DIST_8X8
|
||||
oxcf->using_dist_8x8 = extra_cfg->enable_dist_8x8;
|
||||
if (extra_cfg->tuning == AOM_TUNE_CDEF_DIST ||
|
||||
extra_cfg->tuning == AOM_TUNE_DAALA_DIST)
|
||||
oxcf->using_dist_8x8 = 1;
|
||||
#endif
|
||||
oxcf->num_tile_groups = extra_cfg->num_tg;
|
||||
#if CONFIG_EXT_TILE
|
||||
// In large-scale tile encoding mode, num_tile_groups is always 1.
|
||||
|
|
@ -492,20 +516,31 @@ static aom_codec_err_t set_encoder_config(
|
|||
oxcf->over_shoot_pct = cfg->rc_overshoot_pct;
|
||||
|
||||
oxcf->resize_mode = (RESIZE_MODE)cfg->rc_resize_mode;
|
||||
oxcf->resize_scale_numerator = (uint8_t)cfg->rc_resize_numerator;
|
||||
oxcf->resize_kf_scale_numerator = (uint8_t)cfg->rc_resize_kf_numerator;
|
||||
oxcf->resize_scale_denominator = (uint8_t)cfg->rc_resize_denominator;
|
||||
oxcf->resize_kf_scale_denominator = (uint8_t)cfg->rc_resize_kf_denominator;
|
||||
if (oxcf->resize_mode == RESIZE_FIXED &&
|
||||
oxcf->resize_scale_numerator == SCALE_DENOMINATOR &&
|
||||
oxcf->resize_kf_scale_numerator == SCALE_DENOMINATOR)
|
||||
oxcf->resize_scale_denominator == SCALE_NUMERATOR &&
|
||||
oxcf->resize_kf_scale_denominator == SCALE_NUMERATOR)
|
||||
oxcf->resize_mode = RESIZE_NONE;
|
||||
|
||||
#if CONFIG_FRAME_SUPERRES
|
||||
oxcf->superres_mode = (SUPERRES_MODE)cfg->rc_superres_mode;
|
||||
oxcf->superres_scale_numerator = (uint8_t)cfg->rc_superres_numerator;
|
||||
oxcf->superres_kf_scale_numerator = (uint8_t)cfg->rc_superres_kf_numerator;
|
||||
oxcf->superres_scale_denominator = (uint8_t)cfg->rc_superres_denominator;
|
||||
oxcf->superres_kf_scale_denominator =
|
||||
(uint8_t)cfg->rc_superres_kf_denominator;
|
||||
oxcf->superres_qthresh =
|
||||
extra_cfg->lossless ? 255
|
||||
: av1_quantizer_to_qindex(cfg->rc_superres_qthresh);
|
||||
oxcf->superres_kf_qthresh =
|
||||
extra_cfg->lossless
|
||||
? 255
|
||||
: av1_quantizer_to_qindex(cfg->rc_superres_kf_qthresh);
|
||||
if (oxcf->superres_mode == SUPERRES_FIXED &&
|
||||
oxcf->superres_scale_numerator == SCALE_DENOMINATOR &&
|
||||
oxcf->superres_kf_scale_numerator == SCALE_DENOMINATOR)
|
||||
oxcf->superres_scale_denominator == SCALE_NUMERATOR &&
|
||||
oxcf->superres_kf_scale_denominator == SCALE_NUMERATOR)
|
||||
oxcf->superres_mode = SUPERRES_NONE;
|
||||
if (oxcf->superres_mode == SUPERRES_QTHRESH &&
|
||||
oxcf->superres_qthresh == 255 && oxcf->superres_kf_qthresh == 255)
|
||||
oxcf->superres_mode = SUPERRES_NONE;
|
||||
#endif // CONFIG_FRAME_SUPERRES
|
||||
|
||||
|
|
@ -539,10 +574,17 @@ static aom_codec_err_t set_encoder_config(
|
|||
#endif
|
||||
|
||||
oxcf->color_space = extra_cfg->color_space;
|
||||
|
||||
#if CONFIG_COLORSPACE_HEADERS
|
||||
oxcf->transfer_function = extra_cfg->transfer_function;
|
||||
oxcf->chroma_sample_position = extra_cfg->chroma_sample_position;
|
||||
#else
|
||||
if (extra_cfg->transfer_function != AOM_TF_UNKNOWN)
|
||||
return AOM_CODEC_UNSUP_FEATURE;
|
||||
if (extra_cfg->chroma_sample_position != AOM_CSP_UNKNOWN)
|
||||
return AOM_CODEC_UNSUP_FEATURE;
|
||||
#endif
|
||||
|
||||
oxcf->color_range = extra_cfg->color_range;
|
||||
oxcf->render_width = extra_cfg->render_width;
|
||||
oxcf->render_height = extra_cfg->render_height;
|
||||
|
|
@ -588,6 +630,16 @@ static aom_codec_err_t set_encoder_config(
|
|||
}
|
||||
#endif // CONFIG_EXT_TILE
|
||||
|
||||
#if CONFIG_MAX_TILE
|
||||
oxcf->tile_width_count = AOMMIN(cfg->tile_width_count, MAX_TILE_COLS);
|
||||
oxcf->tile_height_count = AOMMIN(cfg->tile_height_count, MAX_TILE_ROWS);
|
||||
for (int i = 0; i < oxcf->tile_width_count; i++) {
|
||||
oxcf->tile_widths[i] = AOMMAX(cfg->tile_widths[i], 1);
|
||||
}
|
||||
for (int i = 0; i < oxcf->tile_height_count; i++) {
|
||||
oxcf->tile_heights[i] = AOMMAX(cfg->tile_heights[i], 1);
|
||||
}
|
||||
#endif
|
||||
#if CONFIG_DEPENDENT_HORZTILES
|
||||
oxcf->dependent_horz_tiles =
|
||||
#if CONFIG_EXT_TILE
|
||||
|
|
@ -608,39 +660,7 @@ static aom_codec_err_t set_encoder_config(
|
|||
#endif
|
||||
|
||||
oxcf->frame_periodic_boost = extra_cfg->frame_periodic_boost;
|
||||
|
||||
oxcf->motion_vector_unit_test = extra_cfg->motion_vector_unit_test;
|
||||
/*
|
||||
printf("Current AV1 Settings: \n");
|
||||
printf("target_bandwidth: %d\n", oxcf->target_bandwidth);
|
||||
printf("noise_sensitivity: %d\n", oxcf->noise_sensitivity);
|
||||
printf("sharpness: %d\n", oxcf->sharpness);
|
||||
printf("cpu_used: %d\n", oxcf->cpu_used);
|
||||
printf("Mode: %d\n", oxcf->mode);
|
||||
printf("auto_key: %d\n", oxcf->auto_key);
|
||||
printf("key_freq: %d\n", oxcf->key_freq);
|
||||
printf("end_usage: %d\n", oxcf->end_usage);
|
||||
printf("under_shoot_pct: %d\n", oxcf->under_shoot_pct);
|
||||
printf("over_shoot_pct: %d\n", oxcf->over_shoot_pct);
|
||||
printf("starting_buffer_level: %d\n", oxcf->starting_buffer_level);
|
||||
printf("optimal_buffer_level: %d\n", oxcf->optimal_buffer_level);
|
||||
printf("maximum_buffer_size: %d\n", oxcf->maximum_buffer_size);
|
||||
printf("fixed_q: %d\n", oxcf->fixed_q);
|
||||
printf("worst_allowed_q: %d\n", oxcf->worst_allowed_q);
|
||||
printf("best_allowed_q: %d\n", oxcf->best_allowed_q);
|
||||
printf("allow_spatial_resampling: %d\n", oxcf->allow_spatial_resampling);
|
||||
printf("scaled_frame_width: %d\n", oxcf->scaled_frame_width);
|
||||
printf("scaled_frame_height: %d\n", oxcf->scaled_frame_height);
|
||||
printf("two_pass_vbrbias: %d\n", oxcf->two_pass_vbrbias);
|
||||
printf("two_pass_vbrmin_section: %d\n", oxcf->two_pass_vbrmin_section);
|
||||
printf("two_pass_vbrmax_section: %d\n", oxcf->two_pass_vbrmax_section);
|
||||
printf("lag_in_frames: %d\n", oxcf->lag_in_frames);
|
||||
printf("enable_auto_arf: %d\n", oxcf->enable_auto_arf);
|
||||
printf("Version: %d\n", oxcf->Version);
|
||||
printf("error resilient: %d\n", oxcf->error_resilient_mode);
|
||||
printf("frame parallel detokenization: %d\n",
|
||||
oxcf->frame_parallel_decoding_mode);
|
||||
*/
|
||||
return AOM_CODEC_OK;
|
||||
}
|
||||
|
||||
|
|
@ -764,6 +784,7 @@ static aom_codec_err_t ctrl_set_tile_rows(aom_codec_alg_priv_t *ctx,
|
|||
extra_cfg.tile_rows = CAST(AV1E_SET_TILE_ROWS, args);
|
||||
return update_extra_cfg(ctx, &extra_cfg);
|
||||
}
|
||||
|
||||
#if CONFIG_DEPENDENT_HORZTILES
|
||||
static aom_codec_err_t ctrl_set_tile_dependent_rows(aom_codec_alg_priv_t *ctx,
|
||||
va_list args) {
|
||||
|
|
@ -862,7 +883,14 @@ static aom_codec_err_t ctrl_set_qm_max(aom_codec_alg_priv_t *ctx,
|
|||
return update_extra_cfg(ctx, &extra_cfg);
|
||||
}
|
||||
#endif
|
||||
|
||||
#if CONFIG_DIST_8X8
|
||||
static aom_codec_err_t ctrl_set_enable_dist_8x8(aom_codec_alg_priv_t *ctx,
|
||||
va_list args) {
|
||||
struct av1_extracfg extra_cfg = ctx->extra_cfg;
|
||||
extra_cfg.enable_dist_8x8 = CAST(AV1E_SET_ENABLE_DIST_8X8, args);
|
||||
return update_extra_cfg(ctx, &extra_cfg);
|
||||
}
|
||||
#endif
|
||||
static aom_codec_err_t ctrl_set_num_tg(aom_codec_alg_priv_t *ctx,
|
||||
va_list args) {
|
||||
struct av1_extracfg extra_cfg = ctx->extra_cfg;
|
||||
|
|
@ -1044,7 +1072,7 @@ static int write_superframe_index(aom_codec_alg_priv_t *ctx) {
|
|||
// Choose the magnitude
|
||||
int mag;
|
||||
unsigned int mask;
|
||||
for (mag = 0, mask = 0xff; mag < 4; mag++) {
|
||||
for (mag = 0, mask = 0xff; mag < MAG_SIZE; mag++) {
|
||||
if (max_frame_sz <= mask) break;
|
||||
mask <<= 8;
|
||||
mask |= 0xff;
|
||||
|
|
@ -1052,7 +1080,7 @@ static int write_superframe_index(aom_codec_alg_priv_t *ctx) {
|
|||
marker |= mag << 3;
|
||||
|
||||
// Write the index
|
||||
uint8_t buffer[256];
|
||||
uint8_t buffer[MAX_INDEX_SIZE];
|
||||
uint8_t *x = buffer;
|
||||
|
||||
if (TEST_SUPPLEMENTAL_SUPERFRAME_DATA) {
|
||||
|
|
@ -1080,6 +1108,7 @@ static int write_superframe_index(aom_codec_alg_priv_t *ctx) {
|
|||
*x++ = marker;
|
||||
|
||||
const size_t index_sz = x - buffer;
|
||||
assert(index_sz < MAX_INDEX_SIZE);
|
||||
assert(ctx->pending_cx_data_sz + index_sz < ctx->cx_data_sz);
|
||||
|
||||
// move the frame to make room for the index
|
||||
|
|
@ -1229,36 +1258,46 @@ static aom_codec_err_t encoder_encode(aom_codec_alg_priv_t *ctx,
|
|||
}
|
||||
}
|
||||
|
||||
size_t frame_size;
|
||||
size_t frame_size = 0;
|
||||
unsigned int lib_flags = 0;
|
||||
while (cx_data_sz >= ctx->cx_data_sz / 2 &&
|
||||
int is_frame_visible = 0;
|
||||
int index_size = 0;
|
||||
// invisible frames get packed with the next visible frame
|
||||
while (cx_data_sz - index_size >= ctx->cx_data_sz / 2 &&
|
||||
!is_frame_visible &&
|
||||
-1 != av1_get_compressed_data(cpi, &lib_flags, &frame_size, cx_data,
|
||||
&dst_time_stamp, &dst_end_time_stamp,
|
||||
!img)) {
|
||||
#if CONFIG_REFERENCE_BUFFER
|
||||
if (cpi->common.invalid_delta_frame_id_minus1) {
|
||||
ctx->base.err_detail = "Invalid delta_frame_id_minus1";
|
||||
return AOM_CODEC_ERROR;
|
||||
if (cpi->common.seq_params.frame_id_numbers_present_flag) {
|
||||
if (cpi->common.invalid_delta_frame_id_minus1) {
|
||||
ctx->base.err_detail = "Invalid delta_frame_id_minus1";
|
||||
return AOM_CODEC_ERROR;
|
||||
}
|
||||
}
|
||||
#endif
|
||||
if (!frame_size) continue;
|
||||
#endif // CONFIG_REFERENCE_BUFFER
|
||||
if (frame_size) {
|
||||
if (ctx->pending_cx_data == 0) ctx->pending_cx_data = cx_data;
|
||||
|
||||
if (ctx->pending_cx_data == 0) ctx->pending_cx_data = cx_data;
|
||||
ctx->pending_frame_sizes[ctx->pending_frame_count++] = frame_size;
|
||||
ctx->pending_cx_data_sz += frame_size;
|
||||
|
||||
ctx->pending_frame_sizes[ctx->pending_frame_count++] = frame_size;
|
||||
ctx->pending_cx_data_sz += frame_size;
|
||||
cx_data += frame_size;
|
||||
cx_data_sz -= frame_size;
|
||||
|
||||
cx_data += frame_size;
|
||||
cx_data_sz -= frame_size;
|
||||
|
||||
// invisible frames get packed with the next visible frame
|
||||
if (!cpi->common.show_frame) continue;
|
||||
index_size = MAG_SIZE * (ctx->pending_frame_count - 1) + 2;
|
||||
|
||||
is_frame_visible = cpi->common.show_frame;
|
||||
}
|
||||
}
|
||||
if (is_frame_visible) {
|
||||
// insert superframe index if needed
|
||||
if (ctx->pending_frame_count > 1) {
|
||||
const size_t index_size = write_superframe_index(ctx);
|
||||
cx_data += index_size;
|
||||
cx_data_sz -= index_size;
|
||||
#if CONFIG_DEBUG
|
||||
assert(index_size >= write_superframe_index(ctx));
|
||||
#else
|
||||
write_superframe_index(ctx);
|
||||
#endif
|
||||
}
|
||||
|
||||
// Add the frame packet to the list of returned packets.
|
||||
|
|
@ -1294,14 +1333,13 @@ static const aom_codec_cx_pkt_t *encoder_get_cxdata(aom_codec_alg_priv_t *ctx,
|
|||
|
||||
static aom_codec_err_t ctrl_set_reference(aom_codec_alg_priv_t *ctx,
|
||||
va_list args) {
|
||||
aom_ref_frame_t *const frame = va_arg(args, aom_ref_frame_t *);
|
||||
av1_ref_frame_t *const frame = va_arg(args, av1_ref_frame_t *);
|
||||
|
||||
if (frame != NULL) {
|
||||
YV12_BUFFER_CONFIG sd;
|
||||
|
||||
image2yuvconfig(&frame->img, &sd);
|
||||
av1_set_reference_enc(ctx->cpi, ref_frame_to_av1_reframe(frame->frame_type),
|
||||
&sd);
|
||||
av1_set_reference_enc(ctx->cpi, frame->idx, &sd);
|
||||
return AOM_CODEC_OK;
|
||||
} else {
|
||||
return AOM_CODEC_INVALID_PARAM;
|
||||
|
|
@ -1310,14 +1348,13 @@ static aom_codec_err_t ctrl_set_reference(aom_codec_alg_priv_t *ctx,
|
|||
|
||||
static aom_codec_err_t ctrl_copy_reference(aom_codec_alg_priv_t *ctx,
|
||||
va_list args) {
|
||||
aom_ref_frame_t *const frame = va_arg(args, aom_ref_frame_t *);
|
||||
av1_ref_frame_t *const frame = va_arg(args, av1_ref_frame_t *);
|
||||
|
||||
if (frame != NULL) {
|
||||
YV12_BUFFER_CONFIG sd;
|
||||
|
||||
image2yuvconfig(&frame->img, &sd);
|
||||
av1_copy_reference_enc(ctx->cpi,
|
||||
ref_frame_to_av1_reframe(frame->frame_type), &sd);
|
||||
av1_copy_reference_enc(ctx->cpi, frame->idx, &sd);
|
||||
return AOM_CODEC_OK;
|
||||
} else {
|
||||
return AOM_CODEC_INVALID_PARAM;
|
||||
|
|
@ -1450,22 +1487,32 @@ static aom_codec_err_t ctrl_set_color_space(aom_codec_alg_priv_t *ctx,
|
|||
return update_extra_cfg(ctx, &extra_cfg);
|
||||
}
|
||||
|
||||
#if CONFIG_COLORSPACE_HEADERS
|
||||
static aom_codec_err_t ctrl_set_transfer_function(aom_codec_alg_priv_t *ctx,
|
||||
va_list args) {
|
||||
#if CONFIG_COLORSPACE_HEADERS
|
||||
struct av1_extracfg extra_cfg = ctx->extra_cfg;
|
||||
extra_cfg.transfer_function = CAST(AV1E_SET_TRANSFER_FUNCTION, args);
|
||||
return update_extra_cfg(ctx, &extra_cfg);
|
||||
#else
|
||||
(void)ctx;
|
||||
(void)args;
|
||||
return AOM_CODEC_UNSUP_FEATURE;
|
||||
#endif
|
||||
}
|
||||
|
||||
static aom_codec_err_t ctrl_set_chroma_sample_position(
|
||||
aom_codec_alg_priv_t *ctx, va_list args) {
|
||||
#if CONFIG_COLORSPACE_HEADERS
|
||||
struct av1_extracfg extra_cfg = ctx->extra_cfg;
|
||||
extra_cfg.chroma_sample_position =
|
||||
CAST(AV1E_SET_CHROMA_SAMPLE_POSITION, args);
|
||||
return update_extra_cfg(ctx, &extra_cfg);
|
||||
}
|
||||
#else
|
||||
(void)ctx;
|
||||
(void)args;
|
||||
return AOM_CODEC_UNSUP_FEATURE;
|
||||
#endif
|
||||
}
|
||||
|
||||
static aom_codec_err_t ctrl_set_color_range(aom_codec_alg_priv_t *ctx,
|
||||
va_list args) {
|
||||
|
|
@ -1500,11 +1547,11 @@ static aom_codec_err_t ctrl_set_ans_window_size_log2(aom_codec_alg_priv_t *ctx,
|
|||
#endif
|
||||
|
||||
static aom_codec_ctrl_fn_map_t encoder_ctrl_maps[] = {
|
||||
{ AOM_COPY_REFERENCE, ctrl_copy_reference },
|
||||
{ AV1_COPY_REFERENCE, ctrl_copy_reference },
|
||||
{ AOME_USE_REFERENCE, ctrl_use_reference },
|
||||
|
||||
// Setters
|
||||
{ AOM_SET_REFERENCE, ctrl_set_reference },
|
||||
{ AV1_SET_REFERENCE, ctrl_set_reference },
|
||||
{ AOM_SET_POSTPROC, ctrl_set_previewpp },
|
||||
{ AOME_SET_ROI_MAP, ctrl_set_roi_map },
|
||||
{ AOME_SET_ACTIVEMAP, ctrl_set_active_map },
|
||||
|
|
@ -1536,6 +1583,9 @@ static aom_codec_ctrl_fn_map_t encoder_ctrl_maps[] = {
|
|||
{ AV1E_SET_ENABLE_QM, ctrl_set_enable_qm },
|
||||
{ AV1E_SET_QM_MIN, ctrl_set_qm_min },
|
||||
{ AV1E_SET_QM_MAX, ctrl_set_qm_max },
|
||||
#endif
|
||||
#if CONFIG_DIST_8X8
|
||||
{ AV1E_SET_ENABLE_DIST_8X8, ctrl_set_enable_dist_8x8 },
|
||||
#endif
|
||||
{ AV1E_SET_NUM_TG, ctrl_set_num_tg },
|
||||
{ AV1E_SET_MTU, ctrl_set_mtu },
|
||||
|
|
@ -1550,10 +1600,8 @@ static aom_codec_ctrl_fn_map_t encoder_ctrl_maps[] = {
|
|||
{ AV1E_SET_FRAME_PERIODIC_BOOST, ctrl_set_frame_periodic_boost },
|
||||
{ AV1E_SET_TUNE_CONTENT, ctrl_set_tune_content },
|
||||
{ AV1E_SET_COLOR_SPACE, ctrl_set_color_space },
|
||||
#if CONFIG_COLORSPACE_HEADERS
|
||||
{ AV1E_SET_TRANSFER_FUNCTION, ctrl_set_transfer_function },
|
||||
{ AV1E_SET_CHROMA_SAMPLE_POSITION, ctrl_set_chroma_sample_position },
|
||||
#endif
|
||||
{ AV1E_SET_COLOR_RANGE, ctrl_set_color_range },
|
||||
{ AV1E_SET_NOISE_SENSITIVITY, ctrl_set_noise_sensitivity },
|
||||
{ AV1E_SET_MIN_GF_INTERVAL, ctrl_set_min_gf_interval },
|
||||
|
|
@ -1597,16 +1645,18 @@ static aom_codec_enc_cfg_map_t encoder_usage_cfg_map[] = {
|
|||
|
||||
AOM_RC_ONE_PASS, // g_pass
|
||||
|
||||
25, // g_lag_in_frames
|
||||
17, // g_lag_in_frames
|
||||
|
||||
0, // rc_dropframe_thresh
|
||||
RESIZE_NONE, // rc_resize_mode
|
||||
SCALE_DENOMINATOR, // rc_resize_numerator
|
||||
SCALE_DENOMINATOR, // rc_resize_kf_numerator
|
||||
0, // rc_dropframe_thresh
|
||||
RESIZE_NONE, // rc_resize_mode
|
||||
SCALE_NUMERATOR, // rc_resize_denominator
|
||||
SCALE_NUMERATOR, // rc_resize_kf_denominator
|
||||
|
||||
0, // rc_superres_mode
|
||||
SCALE_DENOMINATOR, // rc_superres_numerator
|
||||
SCALE_DENOMINATOR, // rc_superres_kf_numerator
|
||||
0, // rc_superres_mode
|
||||
SCALE_NUMERATOR, // rc_superres_denominator
|
||||
SCALE_NUMERATOR, // rc_superres_kf_denominator
|
||||
63, // rc_superres_qthresh
|
||||
63, // rc_superres_kf_qthresh
|
||||
|
||||
AOM_VBR, // rc_end_usage
|
||||
{ NULL, 0 }, // rc_twopass_stats_in
|
||||
|
|
@ -1630,6 +1680,10 @@ static aom_codec_enc_cfg_map_t encoder_usage_cfg_map[] = {
|
|||
0, // kf_min_dist
|
||||
9999, // kf_max_dist
|
||||
0, // large_scale_tile
|
||||
0, // tile_width_count
|
||||
0, // tile_height_count
|
||||
{ 0 }, // tile_widths
|
||||
{ 0 }, // tile_heights
|
||||
} },
|
||||
};
|
||||
|
||||
|
|
|
|||
6
third_party/aom/av1/av1_dx.mk
vendored
6
third_party/aom/av1/av1_dx.mk
vendored
|
|
@ -32,6 +32,7 @@ AV1_DX_SRCS-yes += decoder/decoder.c
|
|||
AV1_DX_SRCS-yes += decoder/decoder.h
|
||||
AV1_DX_SRCS-yes += decoder/dsubexp.c
|
||||
AV1_DX_SRCS-yes += decoder/dsubexp.h
|
||||
AV1_DX_SRCS-yes += decoder/symbolrate.h
|
||||
|
||||
ifeq ($(CONFIG_ACCOUNTING),yes)
|
||||
AV1_DX_SRCS-yes += decoder/accounting.h
|
||||
|
|
@ -56,11 +57,6 @@ AV1_DX_SRCS-yes += encoder/hybrid_fwd_txfm.h
|
|||
AV1_DX_SRCS-yes += encoder/dct.c
|
||||
AV1_DX_SRCS-$(HAVE_SSE2) += encoder/x86/dct_sse2.asm
|
||||
AV1_DX_SRCS-$(HAVE_SSE2) += encoder/x86/dct_intrin_sse2.c
|
||||
AV1_DX_SRCS-$(HAVE_SSSE3) += encoder/x86/dct_ssse3.c
|
||||
|
||||
ifneq ($(CONFIG_HIGHBITDEPTH),yes)
|
||||
AV1_DX_SRCS-$(HAVE_NEON) += encoder/arm/neon/dct_neon.c
|
||||
endif
|
||||
|
||||
AV1_DX_SRCS-$(HAVE_MSA) += encoder/mips/msa/fdct4x4_msa.c
|
||||
AV1_DX_SRCS-$(HAVE_MSA) += encoder/mips/msa/fdct8x8_msa.c
|
||||
|
|
|
|||
56
third_party/aom/av1/av1_dx_iface.c
vendored
56
third_party/aom/av1/av1_dx_iface.c
vendored
|
|
@ -153,6 +153,7 @@ static aom_codec_err_t decoder_destroy(aom_codec_alg_priv_t *ctx) {
|
|||
return AOM_CODEC_OK;
|
||||
}
|
||||
|
||||
#if !CONFIG_OBU
|
||||
static int parse_bitdepth_colorspace_sampling(BITSTREAM_PROFILE profile,
|
||||
struct aom_read_bit_buffer *rb) {
|
||||
aom_color_space_t color_space;
|
||||
|
|
@ -200,6 +201,7 @@ static int parse_bitdepth_colorspace_sampling(BITSTREAM_PROFILE profile,
|
|||
}
|
||||
return 1;
|
||||
}
|
||||
#endif
|
||||
|
||||
static aom_codec_err_t decoder_peek_si_internal(
|
||||
const uint8_t *data, unsigned int data_sz, aom_codec_stream_info_t *si,
|
||||
|
|
@ -229,9 +231,18 @@ static aom_codec_err_t decoder_peek_si_internal(
|
|||
|
||||
data += index_size;
|
||||
data_sz -= index_size;
|
||||
#if CONFIG_OBU
|
||||
if (data + data_sz <= data) return AOM_CODEC_INVALID_PARAM;
|
||||
#endif
|
||||
}
|
||||
|
||||
{
|
||||
#if CONFIG_OBU
|
||||
// Proper fix needed
|
||||
si->is_kf = 1;
|
||||
intra_only_flag = 1;
|
||||
si->h = 1;
|
||||
#else
|
||||
int show_frame;
|
||||
int error_resilient;
|
||||
struct aom_read_bit_buffer rb = { data, data + data_sz, 0, NULL, NULL };
|
||||
|
|
@ -261,35 +272,35 @@ static aom_codec_err_t decoder_peek_si_internal(
|
|||
|
||||
si->is_kf = !aom_rb_read_bit(&rb);
|
||||
show_frame = aom_rb_read_bit(&rb);
|
||||
if (!si->is_kf) {
|
||||
if (!show_frame) intra_only_flag = show_frame ? 0 : aom_rb_read_bit(&rb);
|
||||
}
|
||||
error_resilient = aom_rb_read_bit(&rb);
|
||||
#if CONFIG_REFERENCE_BUFFER
|
||||
{
|
||||
SequenceHeader seq_params = { 0, 0, 0 };
|
||||
if (si->is_kf) {
|
||||
/* TODO: Move outside frame loop or inside key-frame branch */
|
||||
int frame_id_len;
|
||||
SequenceHeader seq_params;
|
||||
read_sequence_header(&seq_params);
|
||||
read_sequence_header(&seq_params, &rb);
|
||||
#if CONFIG_EXT_TILE
|
||||
if (large_scale_tile) seq_params.frame_id_numbers_present_flag = 0;
|
||||
#endif // CONFIG_EXT_TILE
|
||||
if (seq_params.frame_id_numbers_present_flag) {
|
||||
frame_id_len = seq_params.frame_id_length_minus7 + 7;
|
||||
aom_rb_read_literal(&rb, frame_id_len);
|
||||
}
|
||||
}
|
||||
#endif
|
||||
#endif // CONFIG_REFERENCE_BUFFER
|
||||
#if CONFIG_REFERENCE_BUFFER
|
||||
if (seq_params.frame_id_numbers_present_flag) {
|
||||
int frame_id_len;
|
||||
frame_id_len = seq_params.frame_id_length_minus7 + 7;
|
||||
aom_rb_read_literal(&rb, frame_id_len);
|
||||
}
|
||||
#endif // CONFIG_REFERENCE_BUFFER
|
||||
if (si->is_kf) {
|
||||
if (!av1_read_sync_code(&rb)) return AOM_CODEC_UNSUP_BITSTREAM;
|
||||
|
||||
if (!parse_bitdepth_colorspace_sampling(profile, &rb))
|
||||
return AOM_CODEC_UNSUP_BITSTREAM;
|
||||
av1_read_frame_size(&rb, (int *)&si->w, (int *)&si->h);
|
||||
} else {
|
||||
intra_only_flag = show_frame ? 0 : aom_rb_read_bit(&rb);
|
||||
|
||||
rb.bit_offset += error_resilient ? 0 : 2; // reset_frame_context
|
||||
|
||||
if (intra_only_flag) {
|
||||
if (!av1_read_sync_code(&rb)) return AOM_CODEC_UNSUP_BITSTREAM;
|
||||
if (profile > PROFILE_0) {
|
||||
if (!parse_bitdepth_colorspace_sampling(profile, &rb))
|
||||
return AOM_CODEC_UNSUP_BITSTREAM;
|
||||
|
|
@ -298,6 +309,7 @@ static aom_codec_err_t decoder_peek_si_internal(
|
|||
av1_read_frame_size(&rb, (int *)&si->w, (int *)&si->h);
|
||||
}
|
||||
}
|
||||
#endif // CONFIG_OBU
|
||||
}
|
||||
if (is_intra_only != NULL) *is_intra_only = intra_only_flag;
|
||||
return AOM_CODEC_OK;
|
||||
|
|
@ -876,7 +888,7 @@ static aom_codec_err_t decoder_set_fb_fn(
|
|||
|
||||
static aom_codec_err_t ctrl_set_reference(aom_codec_alg_priv_t *ctx,
|
||||
va_list args) {
|
||||
aom_ref_frame_t *const data = va_arg(args, aom_ref_frame_t *);
|
||||
av1_ref_frame_t *const data = va_arg(args, av1_ref_frame_t *);
|
||||
|
||||
// Only support this function in serial decode.
|
||||
if (ctx->frame_parallel_decode) {
|
||||
|
|
@ -885,13 +897,12 @@ static aom_codec_err_t ctrl_set_reference(aom_codec_alg_priv_t *ctx,
|
|||
}
|
||||
|
||||
if (data) {
|
||||
aom_ref_frame_t *const frame = (aom_ref_frame_t *)data;
|
||||
av1_ref_frame_t *const frame = data;
|
||||
YV12_BUFFER_CONFIG sd;
|
||||
AVxWorker *const worker = ctx->frame_workers;
|
||||
FrameWorkerData *const frame_worker_data = (FrameWorkerData *)worker->data1;
|
||||
image2yuvconfig(&frame->img, &sd);
|
||||
return av1_set_reference_dec(&frame_worker_data->pbi->common,
|
||||
ref_frame_to_av1_reframe(frame->frame_type),
|
||||
return av1_set_reference_dec(&frame_worker_data->pbi->common, frame->idx,
|
||||
&sd);
|
||||
} else {
|
||||
return AOM_CODEC_INVALID_PARAM;
|
||||
|
|
@ -900,7 +911,7 @@ static aom_codec_err_t ctrl_set_reference(aom_codec_alg_priv_t *ctx,
|
|||
|
||||
static aom_codec_err_t ctrl_copy_reference(aom_codec_alg_priv_t *ctx,
|
||||
va_list args) {
|
||||
const aom_ref_frame_t *const frame = va_arg(args, aom_ref_frame_t *);
|
||||
const av1_ref_frame_t *const frame = va_arg(args, av1_ref_frame_t *);
|
||||
|
||||
// Only support this function in serial decode.
|
||||
if (ctx->frame_parallel_decode) {
|
||||
|
|
@ -913,8 +924,7 @@ static aom_codec_err_t ctrl_copy_reference(aom_codec_alg_priv_t *ctx,
|
|||
AVxWorker *const worker = ctx->frame_workers;
|
||||
FrameWorkerData *const frame_worker_data = (FrameWorkerData *)worker->data1;
|
||||
image2yuvconfig(&frame->img, &sd);
|
||||
return av1_copy_reference_dec(frame_worker_data->pbi,
|
||||
(AOM_REFFRAME)frame->frame_type, &sd);
|
||||
return av1_copy_reference_dec(frame_worker_data->pbi, frame->idx, &sd);
|
||||
} else {
|
||||
return AOM_CODEC_INVALID_PARAM;
|
||||
}
|
||||
|
|
@ -1209,10 +1219,10 @@ static aom_codec_err_t ctrl_set_inspection_callback(aom_codec_alg_priv_t *ctx,
|
|||
}
|
||||
|
||||
static aom_codec_ctrl_fn_map_t decoder_ctrl_maps[] = {
|
||||
{ AOM_COPY_REFERENCE, ctrl_copy_reference },
|
||||
{ AV1_COPY_REFERENCE, ctrl_copy_reference },
|
||||
|
||||
// Setters
|
||||
{ AOM_SET_REFERENCE, ctrl_set_reference },
|
||||
{ AV1_SET_REFERENCE, ctrl_set_reference },
|
||||
{ AOM_SET_POSTPROC, ctrl_set_postproc },
|
||||
{ AOM_SET_DBG_COLOR_REF_FRAME, ctrl_set_dbg_options },
|
||||
{ AOM_SET_DBG_COLOR_MB_MODES, ctrl_set_dbg_options },
|
||||
|
|
|
|||
9
third_party/aom/av1/av1_iface_common.h
vendored
9
third_party/aom/av1/av1_iface_common.h
vendored
|
|
@ -142,13 +142,4 @@ static aom_codec_err_t image2yuvconfig(const aom_image_t *img,
|
|||
return AOM_CODEC_OK;
|
||||
}
|
||||
|
||||
static AOM_REFFRAME ref_frame_to_av1_reframe(aom_ref_frame_type_t frame) {
|
||||
switch (frame) {
|
||||
case AOM_LAST_FRAME: return AOM_LAST_FLAG;
|
||||
case AOM_GOLD_FRAME: return AOM_GOLD_FLAG;
|
||||
case AOM_ALTR_FRAME: return AOM_ALT_FLAG;
|
||||
}
|
||||
assert(0 && "Invalid Reference Frame");
|
||||
return AOM_LAST_FLAG;
|
||||
}
|
||||
#endif // AV1_AV1_IFACE_COMMON_H_
|
||||
|
|
|
|||
90
third_party/aom/av1/common/alloccommon.c
vendored
90
third_party/aom/av1/common/alloccommon.c
vendored
|
|
@ -19,9 +19,28 @@
|
|||
#include "av1/common/entropymv.h"
|
||||
#include "av1/common/onyxc_int.h"
|
||||
|
||||
int av1_get_MBs(int width, int height) {
|
||||
const int aligned_width = ALIGN_POWER_OF_TWO(width, 3);
|
||||
const int aligned_height = ALIGN_POWER_OF_TWO(height, 3);
|
||||
const int mi_cols = aligned_width >> MI_SIZE_LOG2;
|
||||
const int mi_rows = aligned_height >> MI_SIZE_LOG2;
|
||||
|
||||
#if CONFIG_CB4X4
|
||||
const int mb_cols = (mi_cols + 2) >> 2;
|
||||
const int mb_rows = (mi_rows + 2) >> 2;
|
||||
#else
|
||||
const int mb_cols = (mi_cols + 1) >> 1;
|
||||
const int mb_rows = (mi_rows + 1) >> 1;
|
||||
#endif
|
||||
return mb_rows * mb_cols;
|
||||
}
|
||||
|
||||
void av1_set_mb_mi(AV1_COMMON *cm, int width, int height) {
|
||||
// TODO(jingning): Fine tune the loop filter operations and bring this
|
||||
// back to integer multiple of 4 for cb4x4.
|
||||
// Ensure that the decoded width and height are both multiples of
|
||||
// 8 luma pixels (note: this may only be a multiple of 4 chroma pixels if
|
||||
// subsampling is used).
|
||||
// This simplifies the implementation of various experiments,
|
||||
// eg. cdef, which operates on units of 8x8 luma pixels.
|
||||
const int aligned_width = ALIGN_POWER_OF_TWO(width, 3);
|
||||
const int aligned_height = ALIGN_POWER_OF_TWO(height, 3);
|
||||
|
||||
|
|
@ -72,6 +91,36 @@ static void free_seg_map(AV1_COMMON *cm) {
|
|||
if (!cm->frame_parallel_decode) {
|
||||
cm->last_frame_seg_map = NULL;
|
||||
}
|
||||
cm->seg_map_alloc_size = 0;
|
||||
}
|
||||
|
||||
static void free_scratch_buffers(AV1_COMMON *cm) {
|
||||
(void)cm;
|
||||
#if CONFIG_NCOBMC && CONFIG_NCOBMC_ADAPT_WEIGHT
|
||||
for (int i = 0; i < 4; ++i) {
|
||||
if (cm->ncobmcaw_buf[i]) {
|
||||
aom_free(cm->ncobmcaw_buf[i]);
|
||||
cm->ncobmcaw_buf[i] = NULL;
|
||||
}
|
||||
}
|
||||
#endif // CONFIG_NCOBMC && CONFIG_NCOBMC_ADAPT_WEIGHT
|
||||
}
|
||||
|
||||
static int alloc_scratch_buffers(AV1_COMMON *cm) {
|
||||
(void)cm;
|
||||
#if CONFIG_NCOBMC && CONFIG_NCOBMC_ADAPT_WEIGHT
|
||||
// If not allocated already, allocate
|
||||
if (!cm->ncobmcaw_buf[0] && !cm->ncobmcaw_buf[1] && !cm->ncobmcaw_buf[2] &&
|
||||
!cm->ncobmcaw_buf[3]) {
|
||||
for (int i = 0; i < 4; ++i) {
|
||||
CHECK_MEM_ERROR(
|
||||
cm, cm->ncobmcaw_buf[i],
|
||||
(uint8_t *)aom_memalign(
|
||||
16, (1 + CONFIG_HIGHBITDEPTH) * MAX_MB_PLANE * MAX_SB_SQUARE));
|
||||
}
|
||||
}
|
||||
#endif // CONFIG_NCOBMC && CONFIG_NCOBMC_ADAPT_WEIGHT
|
||||
return 0;
|
||||
}
|
||||
|
||||
void av1_free_ref_frame_buffers(BufferPool *pool) {
|
||||
|
|
@ -85,7 +134,14 @@ void av1_free_ref_frame_buffers(BufferPool *pool) {
|
|||
}
|
||||
aom_free(pool->frame_bufs[i].mvs);
|
||||
pool->frame_bufs[i].mvs = NULL;
|
||||
#if CONFIG_MFMV
|
||||
aom_free(pool->frame_bufs[i].tpl_mvs);
|
||||
pool->frame_bufs[i].tpl_mvs = NULL;
|
||||
#endif
|
||||
aom_free_frame_buffer(&pool->frame_bufs[i].buf);
|
||||
#if CONFIG_HASH_ME
|
||||
av1_hash_table_destroy(&pool->frame_bufs[i].hash_table);
|
||||
#endif
|
||||
}
|
||||
}
|
||||
|
||||
|
|
@ -108,6 +164,33 @@ void av1_alloc_restoration_buffers(AV1_COMMON *cm) {
|
|||
aom_free(cm->rst_internal.tmpbuf);
|
||||
CHECK_MEM_ERROR(cm, cm->rst_internal.tmpbuf,
|
||||
(int32_t *)aom_memalign(16, RESTORATION_TMPBUF_SIZE));
|
||||
|
||||
#if CONFIG_STRIPED_LOOP_RESTORATION
|
||||
// Allocate internal storage for the loop restoration stripe boundary lines
|
||||
for (p = 0; p < MAX_MB_PLANE; ++p) {
|
||||
int w = p == 0 ? width : ROUND_POWER_OF_TWO(width, cm->subsampling_x);
|
||||
int align_bits = 5; // align for efficiency
|
||||
int stride = ALIGN_POWER_OF_TWO(w, align_bits);
|
||||
int num_stripes = (height + 63) / 64;
|
||||
// for each processing stripe: 2 lines above, 2 below
|
||||
int buf_size = num_stripes * 2 * stride;
|
||||
uint8_t *above_buf, *below_buf;
|
||||
|
||||
aom_free(cm->rst_internal.stripe_boundary_above[p]);
|
||||
aom_free(cm->rst_internal.stripe_boundary_below[p]);
|
||||
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
if (cm->use_highbitdepth) buf_size = buf_size * 2;
|
||||
#endif
|
||||
CHECK_MEM_ERROR(cm, above_buf,
|
||||
(uint8_t *)aom_memalign(1 << align_bits, buf_size));
|
||||
CHECK_MEM_ERROR(cm, below_buf,
|
||||
(uint8_t *)aom_memalign(1 << align_bits, buf_size));
|
||||
cm->rst_internal.stripe_boundary_above[p] = above_buf;
|
||||
cm->rst_internal.stripe_boundary_below[p] = below_buf;
|
||||
cm->rst_internal.stripe_boundary_stride[p] = stride;
|
||||
}
|
||||
#endif // CONFIG_STRIPED_LOOP_RESTORATION
|
||||
}
|
||||
|
||||
void av1_free_restoration_buffers(AV1_COMMON *cm) {
|
||||
|
|
@ -123,12 +206,14 @@ void av1_free_context_buffers(AV1_COMMON *cm) {
|
|||
int i;
|
||||
cm->free_mi(cm);
|
||||
free_seg_map(cm);
|
||||
free_scratch_buffers(cm);
|
||||
for (i = 0; i < MAX_MB_PLANE; i++) {
|
||||
aom_free(cm->above_context[i]);
|
||||
cm->above_context[i] = NULL;
|
||||
}
|
||||
aom_free(cm->above_seg_context);
|
||||
cm->above_seg_context = NULL;
|
||||
cm->above_context_alloc_cols = 0;
|
||||
#if CONFIG_VAR_TX
|
||||
aom_free(cm->above_txfm_context);
|
||||
cm->above_txfm_context = NULL;
|
||||
|
|
@ -155,6 +240,7 @@ int av1_alloc_context_buffers(AV1_COMMON *cm, int width, int height) {
|
|||
free_seg_map(cm);
|
||||
if (alloc_seg_map(cm, cm->mi_rows * cm->mi_cols)) goto fail;
|
||||
}
|
||||
if (alloc_scratch_buffers(cm)) goto fail;
|
||||
|
||||
if (cm->above_context_alloc_cols < cm->mi_cols) {
|
||||
// TODO(geza.lore): These are bigger than they need to be.
|
||||
|
|
|
|||
1
third_party/aom/av1/common/alloccommon.h
vendored
1
third_party/aom/av1/common/alloccommon.h
vendored
|
|
@ -37,6 +37,7 @@ int av1_alloc_state_buffers(struct AV1Common *cm, int width, int height);
|
|||
void av1_free_state_buffers(struct AV1Common *cm);
|
||||
|
||||
void av1_set_mb_mi(struct AV1Common *cm, int width, int height);
|
||||
int av1_get_MBs(int width, int height);
|
||||
|
||||
void av1_swap_current_and_last_seg_map(struct AV1Common *cm);
|
||||
|
||||
|
|
|
|||
|
|
@ -148,13 +148,13 @@ void av1_iht4x4_16_add_neon(const tran_low_t *input, uint8_t *dest,
|
|||
|
||||
TRANSPOSE4X4(&q8s16, &q9s16);
|
||||
|
||||
int tx_type = txfm_param->tx_type;
|
||||
const TX_TYPE tx_type = txfm_param->tx_type;
|
||||
switch (tx_type) {
|
||||
case 0: // idct_idct is not supported. Fall back to C
|
||||
case DCT_DCT: // idct_idct is not supported. Fall back to C
|
||||
av1_iht4x4_16_add_c(input, dest, dest_stride, txfm_param);
|
||||
return;
|
||||
break;
|
||||
case 1: // iadst_idct
|
||||
case ADST_DCT: // iadst_idct
|
||||
// generate constants
|
||||
GENERATE_COSINE_CONSTANTS(&d0s16, &d1s16, &d2s16);
|
||||
GENERATE_SINE_CONSTANTS(&d3s16, &d4s16, &d5s16, &q3s16);
|
||||
|
|
@ -168,7 +168,7 @@ void av1_iht4x4_16_add_neon(const tran_low_t *input, uint8_t *dest,
|
|||
// then transform columns
|
||||
IADST4x4_1D(&d3s16, &d4s16, &d5s16, &q3s16, &q8s16, &q9s16);
|
||||
break;
|
||||
case 2: // idct_iadst
|
||||
case DCT_ADST: // idct_iadst
|
||||
// generate constantsyy
|
||||
GENERATE_COSINE_CONSTANTS(&d0s16, &d1s16, &d2s16);
|
||||
GENERATE_SINE_CONSTANTS(&d3s16, &d4s16, &d5s16, &q3s16);
|
||||
|
|
@ -182,7 +182,7 @@ void av1_iht4x4_16_add_neon(const tran_low_t *input, uint8_t *dest,
|
|||
// then transform columns
|
||||
IDCT4x4_1D(&d0s16, &d1s16, &d2s16, &q8s16, &q9s16);
|
||||
break;
|
||||
case 3: // iadst_iadst
|
||||
case ADST_ADST: // iadst_iadst
|
||||
// generate constants
|
||||
GENERATE_SINE_CONSTANTS(&d3s16, &d4s16, &d5s16, &q3s16);
|
||||
|
||||
|
|
|
|||
|
|
@ -478,13 +478,13 @@ void av1_iht8x8_64_add_neon(const tran_low_t *input, uint8_t *dest,
|
|||
TRANSPOSE8X8(&q8s16, &q9s16, &q10s16, &q11s16, &q12s16, &q13s16, &q14s16,
|
||||
&q15s16);
|
||||
|
||||
int tx_type = txfm_param->tx_type;
|
||||
const TX_TYPE tx_type = txfm_param->tx_type;
|
||||
switch (tx_type) {
|
||||
case 0: // idct_idct is not supported. Fall back to C
|
||||
case DCT_DCT: // idct_idct is not supported. Fall back to C
|
||||
av1_iht8x8_64_add_c(input, dest, dest_stride, txfm_param);
|
||||
return;
|
||||
break;
|
||||
case 1: // iadst_idct
|
||||
case ADST_DCT: // iadst_idct
|
||||
// generate IDCT constants
|
||||
// GENERATE_IDCT_CONSTANTS
|
||||
|
||||
|
|
@ -503,7 +503,7 @@ void av1_iht8x8_64_add_neon(const tran_low_t *input, uint8_t *dest,
|
|||
IADST8X8_1D(&q8s16, &q9s16, &q10s16, &q11s16, &q12s16, &q13s16, &q14s16,
|
||||
&q15s16);
|
||||
break;
|
||||
case 2: // idct_iadst
|
||||
case DCT_ADST: // idct_iadst
|
||||
// generate IADST constants
|
||||
// GENERATE_IADST_CONSTANTS
|
||||
|
||||
|
|
@ -522,7 +522,7 @@ void av1_iht8x8_64_add_neon(const tran_low_t *input, uint8_t *dest,
|
|||
IDCT8x8_1D(&q8s16, &q9s16, &q10s16, &q11s16, &q12s16, &q13s16, &q14s16,
|
||||
&q15s16);
|
||||
break;
|
||||
case 3: // iadst_iadst
|
||||
case ADST_ADST: // iadst_iadst
|
||||
// generate IADST constants
|
||||
// GENERATE_IADST_CONSTANTS
|
||||
|
||||
|
|
|
|||
10
third_party/aom/av1/common/av1_fwd_txfm1d.c
vendored
10
third_party/aom/av1/common/av1_fwd_txfm1d.c
vendored
|
|
@ -1547,6 +1547,16 @@ void av1_fidentity32_c(const int32_t *input, int32_t *output,
|
|||
for (int i = 0; i < 32; ++i) output[i] = input[i] * 4;
|
||||
range_check(0, input, output, 32, stage_range[0]);
|
||||
}
|
||||
|
||||
#if CONFIG_TX64X64
|
||||
void av1_fidentity64_c(const int32_t *input, int32_t *output,
|
||||
const int8_t *cos_bit, const int8_t *stage_range) {
|
||||
(void)cos_bit;
|
||||
for (int i = 0; i < 64; ++i)
|
||||
output[i] = (int32_t)dct_const_round_shift(input[i] * 4 * Sqrt2);
|
||||
range_check(0, input, output, 64, stage_range[0]);
|
||||
}
|
||||
#endif // CONFIG_TX64X64
|
||||
#endif // CONFIG_EXT_TX
|
||||
|
||||
#if CONFIG_TX64X64
|
||||
|
|
|
|||
6
third_party/aom/av1/common/av1_fwd_txfm1d.h
vendored
6
third_party/aom/av1/common/av1_fwd_txfm1d.h
vendored
|
|
@ -26,8 +26,10 @@ void av1_fdct16_new(const int32_t *input, int32_t *output,
|
|||
const int8_t *cos_bit, const int8_t *stage_range);
|
||||
void av1_fdct32_new(const int32_t *input, int32_t *output,
|
||||
const int8_t *cos_bit, const int8_t *stage_range);
|
||||
#if CONFIG_TX64X64
|
||||
void av1_fdct64_new(const int32_t *input, int32_t *output,
|
||||
const int8_t *cos_bit, const int8_t *stage_range);
|
||||
#endif // CONFIG_TX64X64
|
||||
|
||||
void av1_fadst4_new(const int32_t *input, int32_t *output,
|
||||
const int8_t *cos_bit, const int8_t *stage_range);
|
||||
|
|
@ -46,6 +48,10 @@ void av1_fidentity16_c(const int32_t *input, int32_t *output,
|
|||
const int8_t *cos_bit, const int8_t *stage_range);
|
||||
void av1_fidentity32_c(const int32_t *input, int32_t *output,
|
||||
const int8_t *cos_bit, const int8_t *stage_range);
|
||||
#if CONFIG_TX64X64
|
||||
void av1_fidentity64_c(const int32_t *input, int32_t *output,
|
||||
const int8_t *cos_bit, const int8_t *stage_range);
|
||||
#endif // CONFIG_TX64X64
|
||||
#endif // CONFIG_EXT_TX
|
||||
|
||||
#ifdef __cplusplus
|
||||
|
|
|
|||
69
third_party/aom/av1/common/av1_fwd_txfm1d_cfg.h
vendored
69
third_party/aom/av1/common/av1_fwd_txfm1d_cfg.h
vendored
|
|
@ -19,11 +19,11 @@
|
|||
static const int8_t fwd_shift_4[3] = { 2, 0, 0 };
|
||||
|
||||
// stage range
|
||||
static const int8_t fwd_stage_range_col_dct_4[4] = { 15, 16, 17, 17 };
|
||||
static const int8_t fwd_stage_range_row_dct_4[4] = { 17, 18, 18, 18 };
|
||||
static const int8_t fwd_stage_range_col_adst_4[6] = { 15, 15, 16, 17, 17, 17 };
|
||||
static const int8_t fwd_stage_range_row_adst_4[6] = { 17, 17, 17, 18, 18, 18 };
|
||||
static const int8_t fwd_stage_range_idx_4[1] = { 18 };
|
||||
static const int8_t fwd_stage_range_col_dct_4[4] = { 0, 1, 2, 2 };
|
||||
static const int8_t fwd_stage_range_row_dct_4[4] = { 2, 3, 3, 3 };
|
||||
static const int8_t fwd_stage_range_col_adst_4[6] = { 0, 0, 1, 2, 2, 2 };
|
||||
static const int8_t fwd_stage_range_row_adst_4[6] = { 2, 2, 2, 3, 3, 3 };
|
||||
static const int8_t fwd_stage_range_idx_4[1] = { 0 };
|
||||
|
||||
// cos bit
|
||||
static const int8_t fwd_cos_bit_col_dct_4[4] = { 13, 13, 13, 13 };
|
||||
|
|
@ -36,13 +36,11 @@ static const int8_t fwd_cos_bit_row_adst_4[6] = { 13, 13, 13, 13, 13, 13 };
|
|||
static const int8_t fwd_shift_8[3] = { 2, -1, 0 };
|
||||
|
||||
// stage range
|
||||
static const int8_t fwd_stage_range_col_dct_8[6] = { 15, 16, 17, 18, 18, 18 };
|
||||
static const int8_t fwd_stage_range_row_dct_8[6] = { 17, 18, 19, 19, 19, 19 };
|
||||
static const int8_t fwd_stage_range_col_adst_8[8] = { 15, 15, 16, 17,
|
||||
17, 18, 18, 18 };
|
||||
static const int8_t fwd_stage_range_row_adst_8[8] = { 17, 17, 17, 18,
|
||||
18, 19, 19, 19 };
|
||||
static const int8_t fwd_stage_range_idx_8[1] = { 19 };
|
||||
static const int8_t fwd_stage_range_col_dct_8[6] = { 0, 1, 2, 3, 3, 3 };
|
||||
static const int8_t fwd_stage_range_row_dct_8[6] = { 3, 4, 5, 5, 5, 5 };
|
||||
static const int8_t fwd_stage_range_col_adst_8[8] = { 0, 0, 1, 2, 2, 3, 3, 3 };
|
||||
static const int8_t fwd_stage_range_row_adst_8[8] = { 3, 3, 3, 4, 4, 5, 5, 5 };
|
||||
static const int8_t fwd_stage_range_idx_8[1] = { 0 };
|
||||
|
||||
// cos bit
|
||||
static const int8_t fwd_cos_bit_col_dct_8[6] = { 13, 13, 13, 13, 13, 13 };
|
||||
|
|
@ -59,15 +57,14 @@ static const int8_t fwd_cos_bit_row_adst_8[8] = {
|
|||
static const int8_t fwd_shift_16[3] = { 2, -2, 0 };
|
||||
|
||||
// stage range
|
||||
static const int8_t fwd_stage_range_col_dct_16[8] = { 15, 16, 17, 18,
|
||||
19, 19, 19, 19 };
|
||||
static const int8_t fwd_stage_range_row_dct_16[8] = { 17, 18, 19, 20,
|
||||
20, 20, 20, 20 };
|
||||
static const int8_t fwd_stage_range_col_adst_16[10] = { 15, 15, 16, 17, 17,
|
||||
18, 18, 19, 19, 19 };
|
||||
static const int8_t fwd_stage_range_row_adst_16[10] = { 17, 17, 17, 18, 18,
|
||||
19, 19, 20, 20, 20 };
|
||||
static const int8_t fwd_stage_range_idx_16[1] = { 20 };
|
||||
static const int8_t fwd_stage_range_col_dct_16[8] = { 0, 1, 2, 3, 4, 4, 4, 4 };
|
||||
static const int8_t fwd_stage_range_row_dct_16[8] = { 4, 5, 6, 7, 7, 7, 7, 7 };
|
||||
static const int8_t fwd_stage_range_col_adst_16[10] = { 0, 0, 1, 2, 2,
|
||||
3, 3, 4, 4, 4 };
|
||||
static const int8_t fwd_stage_range_row_adst_16[10] = {
|
||||
4, 4, 4, 5, 5, 6, 6, 7, 7, 7,
|
||||
};
|
||||
static const int8_t fwd_stage_range_idx_16[1] = { 0 };
|
||||
|
||||
// cos bit
|
||||
static const int8_t fwd_cos_bit_col_dct_16[8] = {
|
||||
|
|
@ -86,17 +83,15 @@ static const int8_t fwd_cos_bit_row_adst_16[10] = { 12, 12, 12, 12, 12,
|
|||
static const int8_t fwd_shift_32[3] = { 2, -4, 0 };
|
||||
|
||||
// stage range
|
||||
static const int8_t fwd_stage_range_col_dct_32[10] = { 15, 16, 17, 18, 19,
|
||||
20, 20, 20, 20, 20 };
|
||||
static const int8_t fwd_stage_range_row_dct_32[10] = { 16, 17, 18, 19, 20,
|
||||
20, 20, 20, 20, 20 };
|
||||
static const int8_t fwd_stage_range_col_adst_32[12] = {
|
||||
15, 15, 16, 17, 17, 18, 18, 19, 19, 20, 20, 20
|
||||
};
|
||||
static const int8_t fwd_stage_range_row_adst_32[12] = {
|
||||
16, 16, 16, 17, 17, 18, 18, 19, 19, 20, 20, 20
|
||||
};
|
||||
static const int8_t fwd_stage_range_idx_32[1] = { 20 };
|
||||
static const int8_t fwd_stage_range_col_dct_32[10] = { 0, 1, 2, 3, 4,
|
||||
5, 5, 5, 5, 5 };
|
||||
static const int8_t fwd_stage_range_row_dct_32[10] = { 5, 6, 7, 8, 9,
|
||||
9, 9, 9, 9, 9 };
|
||||
static const int8_t fwd_stage_range_col_adst_32[12] = { 0, 0, 1, 2, 2, 3,
|
||||
3, 4, 4, 5, 5, 5 };
|
||||
static const int8_t fwd_stage_range_row_adst_32[12] = { 5, 5, 5, 6, 6, 7,
|
||||
7, 8, 8, 9, 9, 9 };
|
||||
static const int8_t fwd_stage_range_idx_32[1] = { 0 };
|
||||
|
||||
// cos bit
|
||||
static const int8_t fwd_cos_bit_col_dct_32[10] = { 12, 12, 12, 12, 12,
|
||||
|
|
@ -113,11 +108,11 @@ static const int8_t fwd_cos_bit_row_adst_32[12] = { 12, 12, 12, 12, 12, 12,
|
|||
static const int8_t fwd_shift_64[3] = { 0, -2, -2 };
|
||||
|
||||
// stage range
|
||||
static const int8_t fwd_stage_range_col_dct_64[12] = { 13, 14, 15, 16, 17, 18,
|
||||
19, 19, 19, 19, 19, 19 };
|
||||
static const int8_t fwd_stage_range_row_dct_64[12] = { 17, 18, 19, 20, 21, 22,
|
||||
22, 22, 22, 22, 22, 22 };
|
||||
static const int8_t fwd_stage_range_idx_64[1] = { 22 };
|
||||
static const int8_t fwd_stage_range_col_dct_64[12] = { 0, 1, 2, 3, 4, 5,
|
||||
6, 6, 6, 6, 6, 6 };
|
||||
static const int8_t fwd_stage_range_row_dct_64[12] = { 6, 7, 8, 9, 10, 11,
|
||||
11, 11, 11, 11, 11, 11 };
|
||||
static const int8_t fwd_stage_range_idx_64[1] = { 0 };
|
||||
|
||||
// cos bit
|
||||
static const int8_t fwd_cos_bit_col_dct_64[12] = { 15, 15, 15, 15, 15, 14,
|
||||
|
|
|
|||
206
third_party/aom/av1/common/av1_fwd_txfm2d.c
vendored
206
third_party/aom/av1/common/av1_fwd_txfm2d.c
vendored
|
|
@ -24,6 +24,9 @@ static INLINE TxfmFunc fwd_txfm_type_to_func(TXFM_TYPE txfm_type) {
|
|||
case TXFM_TYPE_DCT8: return av1_fdct8_new;
|
||||
case TXFM_TYPE_DCT16: return av1_fdct16_new;
|
||||
case TXFM_TYPE_DCT32: return av1_fdct32_new;
|
||||
#if CONFIG_TX64X64
|
||||
case TXFM_TYPE_DCT64: return av1_fdct64_new;
|
||||
#endif // CONFIG_TX64X64
|
||||
case TXFM_TYPE_ADST4: return av1_fadst4_new;
|
||||
case TXFM_TYPE_ADST8: return av1_fadst8_new;
|
||||
case TXFM_TYPE_ADST16: return av1_fadst16_new;
|
||||
|
|
@ -33,14 +36,42 @@ static INLINE TxfmFunc fwd_txfm_type_to_func(TXFM_TYPE txfm_type) {
|
|||
case TXFM_TYPE_IDENTITY8: return av1_fidentity8_c;
|
||||
case TXFM_TYPE_IDENTITY16: return av1_fidentity16_c;
|
||||
case TXFM_TYPE_IDENTITY32: return av1_fidentity32_c;
|
||||
#if CONFIG_TX64X64
|
||||
case TXFM_TYPE_IDENTITY64: return av1_fidentity64_c;
|
||||
#endif // CONFIG_TX64X64
|
||||
#endif // CONFIG_EXT_TX
|
||||
default: assert(0); return NULL;
|
||||
}
|
||||
}
|
||||
|
||||
void av1_gen_fwd_stage_range(int8_t *stage_range_col, int8_t *stage_range_row,
|
||||
const TXFM_2D_FLIP_CFG *cfg, int bd) {
|
||||
// Note when assigning txfm_size_col, we use the txfm_size from the
|
||||
// row configuration and vice versa. This is intentionally done to
|
||||
// accurately perform rectangular transforms. When the transform is
|
||||
// rectangular, the number of columns will be the same as the
|
||||
// txfm_size stored in the row cfg struct. It will make no difference
|
||||
// for square transforms.
|
||||
const int txfm_size_col = cfg->row_cfg->txfm_size;
|
||||
const int txfm_size_row = cfg->col_cfg->txfm_size;
|
||||
// Take the shift from the larger dimension in the rectangular case.
|
||||
const int8_t *shift = (txfm_size_col > txfm_size_row) ? cfg->row_cfg->shift
|
||||
: cfg->col_cfg->shift;
|
||||
// i < MAX_TXFM_STAGE_NUM will mute above array bounds warning
|
||||
for (int i = 0; i < cfg->col_cfg->stage_num && i < MAX_TXFM_STAGE_NUM; ++i) {
|
||||
stage_range_col[i] = cfg->col_cfg->stage_range[i] + shift[0] + bd + 1;
|
||||
}
|
||||
|
||||
// i < MAX_TXFM_STAGE_NUM will mute above array bounds warning
|
||||
for (int i = 0; i < cfg->row_cfg->stage_num && i < MAX_TXFM_STAGE_NUM; ++i) {
|
||||
stage_range_row[i] =
|
||||
cfg->row_cfg->stage_range[i] + shift[0] + shift[1] + bd + 1;
|
||||
}
|
||||
}
|
||||
|
||||
static INLINE void fwd_txfm2d_c(const int16_t *input, int32_t *output,
|
||||
const int stride, const TXFM_2D_FLIP_CFG *cfg,
|
||||
int32_t *buf) {
|
||||
int32_t *buf, int bd) {
|
||||
int c, r;
|
||||
// Note when assigning txfm_size_col, we use the txfm_size from the
|
||||
// row configuration and vice versa. This is intentionally done to
|
||||
|
|
@ -53,8 +84,12 @@ static INLINE void fwd_txfm2d_c(const int16_t *input, int32_t *output,
|
|||
// Take the shift from the larger dimension in the rectangular case.
|
||||
const int8_t *shift = (txfm_size_col > txfm_size_row) ? cfg->row_cfg->shift
|
||||
: cfg->col_cfg->shift;
|
||||
const int8_t *stage_range_col = cfg->col_cfg->stage_range;
|
||||
const int8_t *stage_range_row = cfg->row_cfg->stage_range;
|
||||
int8_t stage_range_col[MAX_TXFM_STAGE_NUM];
|
||||
int8_t stage_range_row[MAX_TXFM_STAGE_NUM];
|
||||
assert(cfg->col_cfg->stage_num <= MAX_TXFM_STAGE_NUM);
|
||||
assert(cfg->row_cfg->stage_num <= MAX_TXFM_STAGE_NUM);
|
||||
av1_gen_fwd_stage_range(stage_range_col, stage_range_row, cfg, bd);
|
||||
|
||||
const int8_t *cos_bit_col = cfg->col_cfg->cos_bit;
|
||||
const int8_t *cos_bit_row = cfg->row_cfg->cos_bit;
|
||||
const TxfmFunc txfm_func_col = fwd_txfm_type_to_func(cfg->col_cfg->txfm_type);
|
||||
|
|
@ -108,93 +143,146 @@ static INLINE void fwd_txfm2d_c(const int16_t *input, int32_t *output,
|
|||
}
|
||||
|
||||
void av1_fwd_txfm2d_4x8_c(const int16_t *input, int32_t *output, int stride,
|
||||
int tx_type, int bd) {
|
||||
TX_TYPE tx_type, int bd) {
|
||||
#if CONFIG_TXMG
|
||||
int32_t txfm_buf[4 * 8];
|
||||
int16_t rinput[4 * 8];
|
||||
TX_SIZE tx_size = TX_4X8;
|
||||
TX_SIZE rtx_size = av1_rotate_tx_size(tx_size);
|
||||
TX_TYPE rtx_type = av1_rotate_tx_type(tx_type);
|
||||
int w = tx_size_wide[tx_size];
|
||||
int h = tx_size_high[tx_size];
|
||||
int rw = h;
|
||||
int rh = w;
|
||||
transpose_int16(rinput, rw, input, stride, w, h);
|
||||
TXFM_2D_FLIP_CFG cfg = av1_get_fwd_txfm_cfg(rtx_type, rtx_size);
|
||||
fwd_txfm2d_c(rinput, txfm_buf, rw, &cfg, output, bd);
|
||||
transpose_int32(output, w, txfm_buf, rw, rw, rh);
|
||||
#else
|
||||
int32_t txfm_buf[4 * 8];
|
||||
TXFM_2D_FLIP_CFG cfg = av1_get_fwd_txfm_cfg(tx_type, TX_4X8);
|
||||
(void)bd;
|
||||
fwd_txfm2d_c(input, output, stride, &cfg, txfm_buf);
|
||||
fwd_txfm2d_c(input, output, stride, &cfg, txfm_buf, bd);
|
||||
#endif
|
||||
}
|
||||
|
||||
void av1_fwd_txfm2d_8x4_c(const int16_t *input, int32_t *output, int stride,
|
||||
int tx_type, int bd) {
|
||||
TX_TYPE tx_type, int bd) {
|
||||
int32_t txfm_buf[8 * 4];
|
||||
TXFM_2D_FLIP_CFG cfg = av1_get_fwd_txfm_cfg(tx_type, TX_8X4);
|
||||
(void)bd;
|
||||
fwd_txfm2d_c(input, output, stride, &cfg, txfm_buf);
|
||||
fwd_txfm2d_c(input, output, stride, &cfg, txfm_buf, bd);
|
||||
}
|
||||
|
||||
void av1_fwd_txfm2d_8x16_c(const int16_t *input, int32_t *output, int stride,
|
||||
int tx_type, int bd) {
|
||||
TX_TYPE tx_type, int bd) {
|
||||
#if CONFIG_TXMG
|
||||
int32_t txfm_buf[8 * 16];
|
||||
int16_t rinput[8 * 16];
|
||||
TX_SIZE tx_size = TX_8X16;
|
||||
TX_SIZE rtx_size = av1_rotate_tx_size(tx_size);
|
||||
TX_TYPE rtx_type = av1_rotate_tx_type(tx_type);
|
||||
int w = tx_size_wide[tx_size];
|
||||
int h = tx_size_high[tx_size];
|
||||
int rw = h;
|
||||
int rh = w;
|
||||
transpose_int16(rinput, rw, input, stride, w, h);
|
||||
TXFM_2D_FLIP_CFG cfg = av1_get_fwd_txfm_cfg(rtx_type, rtx_size);
|
||||
fwd_txfm2d_c(rinput, txfm_buf, rw, &cfg, output, bd);
|
||||
transpose_int32(output, w, txfm_buf, rw, rw, rh);
|
||||
#else
|
||||
int32_t txfm_buf[8 * 16];
|
||||
TXFM_2D_FLIP_CFG cfg = av1_get_fwd_txfm_cfg(tx_type, TX_8X16);
|
||||
(void)bd;
|
||||
fwd_txfm2d_c(input, output, stride, &cfg, txfm_buf);
|
||||
fwd_txfm2d_c(input, output, stride, &cfg, txfm_buf, bd);
|
||||
#endif
|
||||
}
|
||||
|
||||
void av1_fwd_txfm2d_16x8_c(const int16_t *input, int32_t *output, int stride,
|
||||
int tx_type, int bd) {
|
||||
TX_TYPE tx_type, int bd) {
|
||||
int32_t txfm_buf[16 * 8];
|
||||
TXFM_2D_FLIP_CFG cfg = av1_get_fwd_txfm_cfg(tx_type, TX_16X8);
|
||||
(void)bd;
|
||||
fwd_txfm2d_c(input, output, stride, &cfg, txfm_buf);
|
||||
fwd_txfm2d_c(input, output, stride, &cfg, txfm_buf, bd);
|
||||
}
|
||||
|
||||
void av1_fwd_txfm2d_16x32_c(const int16_t *input, int32_t *output, int stride,
|
||||
int tx_type, int bd) {
|
||||
TX_TYPE tx_type, int bd) {
|
||||
#if CONFIG_TXMG
|
||||
int32_t txfm_buf[16 * 32];
|
||||
int16_t rinput[16 * 32];
|
||||
TX_SIZE tx_size = TX_16X32;
|
||||
TX_SIZE rtx_size = av1_rotate_tx_size(tx_size);
|
||||
TX_TYPE rtx_type = av1_rotate_tx_type(tx_type);
|
||||
int w = tx_size_wide[tx_size];
|
||||
int h = tx_size_high[tx_size];
|
||||
int rw = h;
|
||||
int rh = w;
|
||||
transpose_int16(rinput, rw, input, stride, w, h);
|
||||
TXFM_2D_FLIP_CFG cfg = av1_get_fwd_txfm_cfg(rtx_type, rtx_size);
|
||||
fwd_txfm2d_c(rinput, txfm_buf, rw, &cfg, output, bd);
|
||||
transpose_int32(output, w, txfm_buf, rw, rw, rh);
|
||||
#else
|
||||
int32_t txfm_buf[16 * 32];
|
||||
TXFM_2D_FLIP_CFG cfg = av1_get_fwd_txfm_cfg(tx_type, TX_16X32);
|
||||
(void)bd;
|
||||
fwd_txfm2d_c(input, output, stride, &cfg, txfm_buf);
|
||||
fwd_txfm2d_c(input, output, stride, &cfg, txfm_buf, bd);
|
||||
#endif
|
||||
}
|
||||
|
||||
void av1_fwd_txfm2d_32x16_c(const int16_t *input, int32_t *output, int stride,
|
||||
int tx_type, int bd) {
|
||||
TX_TYPE tx_type, int bd) {
|
||||
int32_t txfm_buf[32 * 16];
|
||||
TXFM_2D_FLIP_CFG cfg = av1_get_fwd_txfm_cfg(tx_type, TX_32X16);
|
||||
(void)bd;
|
||||
fwd_txfm2d_c(input, output, stride, &cfg, txfm_buf);
|
||||
fwd_txfm2d_c(input, output, stride, &cfg, txfm_buf, bd);
|
||||
}
|
||||
|
||||
void av1_fwd_txfm2d_4x4_c(const int16_t *input, int32_t *output, int stride,
|
||||
int tx_type, int bd) {
|
||||
TX_TYPE tx_type, int bd) {
|
||||
int32_t txfm_buf[4 * 4];
|
||||
TXFM_2D_FLIP_CFG cfg = av1_get_fwd_txfm_cfg(tx_type, TX_4X4);
|
||||
(void)bd;
|
||||
fwd_txfm2d_c(input, output, stride, &cfg, txfm_buf);
|
||||
fwd_txfm2d_c(input, output, stride, &cfg, txfm_buf, bd);
|
||||
}
|
||||
|
||||
void av1_fwd_txfm2d_8x8_c(const int16_t *input, int32_t *output, int stride,
|
||||
int tx_type, int bd) {
|
||||
TX_TYPE tx_type, int bd) {
|
||||
int32_t txfm_buf[8 * 8];
|
||||
TXFM_2D_FLIP_CFG cfg = av1_get_fwd_txfm_cfg(tx_type, TX_8X8);
|
||||
(void)bd;
|
||||
fwd_txfm2d_c(input, output, stride, &cfg, txfm_buf);
|
||||
fwd_txfm2d_c(input, output, stride, &cfg, txfm_buf, bd);
|
||||
}
|
||||
|
||||
void av1_fwd_txfm2d_16x16_c(const int16_t *input, int32_t *output, int stride,
|
||||
int tx_type, int bd) {
|
||||
TX_TYPE tx_type, int bd) {
|
||||
int32_t txfm_buf[16 * 16];
|
||||
TXFM_2D_FLIP_CFG cfg = av1_get_fwd_txfm_cfg(tx_type, TX_16X16);
|
||||
(void)bd;
|
||||
fwd_txfm2d_c(input, output, stride, &cfg, txfm_buf);
|
||||
fwd_txfm2d_c(input, output, stride, &cfg, txfm_buf, bd);
|
||||
}
|
||||
|
||||
void av1_fwd_txfm2d_32x32_c(const int16_t *input, int32_t *output, int stride,
|
||||
int tx_type, int bd) {
|
||||
TX_TYPE tx_type, int bd) {
|
||||
int32_t txfm_buf[32 * 32];
|
||||
TXFM_2D_FLIP_CFG cfg = av1_get_fwd_txfm_cfg(tx_type, TX_32X32);
|
||||
(void)bd;
|
||||
fwd_txfm2d_c(input, output, stride, &cfg, txfm_buf);
|
||||
fwd_txfm2d_c(input, output, stride, &cfg, txfm_buf, bd);
|
||||
}
|
||||
|
||||
#if CONFIG_TX64X64
|
||||
void av1_fwd_txfm2d_64x64_c(const int16_t *input, int32_t *output, int stride,
|
||||
int tx_type, int bd) {
|
||||
TX_TYPE tx_type, int bd) {
|
||||
int32_t txfm_buf[64 * 64];
|
||||
TXFM_2D_FLIP_CFG cfg = av1_get_fwd_txfm_64x64_cfg(tx_type);
|
||||
(void)bd;
|
||||
fwd_txfm2d_c(input, output, stride, &cfg, txfm_buf);
|
||||
fwd_txfm2d_c(input, output, stride, &cfg, txfm_buf, bd);
|
||||
}
|
||||
|
||||
void av1_fwd_txfm2d_32x64_c(const int16_t *input, int32_t *output, int stride,
|
||||
TX_TYPE tx_type, int bd) {
|
||||
int32_t txfm_buf[32 * 64];
|
||||
TXFM_2D_FLIP_CFG cfg = av1_get_fwd_txfm_32x64_cfg(tx_type);
|
||||
fwd_txfm2d_c(input, output, stride, &cfg, txfm_buf, bd);
|
||||
}
|
||||
|
||||
void av1_fwd_txfm2d_64x32_c(const int16_t *input, int32_t *output, int stride,
|
||||
TX_TYPE tx_type, int bd) {
|
||||
int32_t txfm_buf[64 * 32];
|
||||
TXFM_2D_FLIP_CFG cfg = av1_get_fwd_txfm_64x32_cfg(tx_type);
|
||||
fwd_txfm2d_c(input, output, stride, &cfg, txfm_buf, bd);
|
||||
}
|
||||
#endif // CONFIG_TX64X64
|
||||
|
||||
static const TXFM_1D_CFG *fwd_txfm_col_cfg_ls[TX_TYPES_1D][TX_SIZES] = {
|
||||
// DCT
|
||||
{
|
||||
|
|
@ -261,19 +349,52 @@ static const TXFM_1D_CFG *fwd_txfm_row_cfg_ls[TX_TYPES_1D][TX_SIZES] = {
|
|||
#endif // CONFIG_EXT_TX
|
||||
};
|
||||
|
||||
TXFM_2D_FLIP_CFG av1_get_fwd_txfm_cfg(int tx_type, int tx_size) {
|
||||
TXFM_2D_FLIP_CFG av1_get_fwd_txfm_cfg(TX_TYPE tx_type, TX_SIZE tx_size) {
|
||||
TXFM_2D_FLIP_CFG cfg;
|
||||
set_flip_cfg(tx_type, &cfg);
|
||||
const int tx_type_col = vtx_tab[tx_type];
|
||||
const int tx_type_row = htx_tab[tx_type];
|
||||
const int tx_size_col = txsize_vert_map[tx_size];
|
||||
const int tx_size_row = txsize_horz_map[tx_size];
|
||||
const TX_TYPE_1D tx_type_col = vtx_tab[tx_type];
|
||||
const TX_TYPE_1D tx_type_row = htx_tab[tx_type];
|
||||
const TX_SIZE tx_size_col = txsize_vert_map[tx_size];
|
||||
const TX_SIZE tx_size_row = txsize_horz_map[tx_size];
|
||||
cfg.col_cfg = fwd_txfm_col_cfg_ls[tx_type_col][tx_size_col];
|
||||
cfg.row_cfg = fwd_txfm_row_cfg_ls[tx_type_row][tx_size_row];
|
||||
return cfg;
|
||||
}
|
||||
|
||||
TXFM_2D_FLIP_CFG av1_get_fwd_txfm_64x64_cfg(int tx_type) {
|
||||
#if CONFIG_TX64X64
|
||||
TXFM_2D_FLIP_CFG av1_get_fwd_txfm_32x64_cfg(TX_TYPE tx_type) {
|
||||
TXFM_2D_FLIP_CFG cfg;
|
||||
const TX_TYPE_1D tx_type_row = htx_tab[tx_type];
|
||||
const TX_SIZE tx_size_row = txsize_horz_map[TX_32X64];
|
||||
switch (tx_type) {
|
||||
case DCT_DCT:
|
||||
cfg.col_cfg = &fwd_txfm_1d_col_cfg_dct_64;
|
||||
cfg.row_cfg = fwd_txfm_row_cfg_ls[tx_type_row][tx_size_row];
|
||||
cfg.ud_flip = 0;
|
||||
cfg.lr_flip = 0;
|
||||
break;
|
||||
default: assert(0);
|
||||
}
|
||||
return cfg;
|
||||
}
|
||||
|
||||
TXFM_2D_FLIP_CFG av1_get_fwd_txfm_64x32_cfg(TX_TYPE tx_type) {
|
||||
TXFM_2D_FLIP_CFG cfg;
|
||||
const TX_TYPE_1D tx_type_col = vtx_tab[tx_type];
|
||||
const TX_SIZE tx_size_col = txsize_vert_map[TX_64X32];
|
||||
switch (tx_type) {
|
||||
case DCT_DCT:
|
||||
cfg.col_cfg = fwd_txfm_col_cfg_ls[tx_type_col][tx_size_col];
|
||||
cfg.row_cfg = &fwd_txfm_1d_row_cfg_dct_64;
|
||||
cfg.ud_flip = 0;
|
||||
cfg.lr_flip = 0;
|
||||
break;
|
||||
default: assert(0);
|
||||
}
|
||||
return cfg;
|
||||
}
|
||||
|
||||
TXFM_2D_FLIP_CFG av1_get_fwd_txfm_64x64_cfg(TX_TYPE tx_type) {
|
||||
TXFM_2D_FLIP_CFG cfg;
|
||||
switch (tx_type) {
|
||||
case DCT_DCT:
|
||||
|
|
@ -289,3 +410,4 @@ TXFM_2D_FLIP_CFG av1_get_fwd_txfm_64x64_cfg(int tx_type) {
|
|||
}
|
||||
return cfg;
|
||||
}
|
||||
#endif // CONFIG_TX64X64
|
||||
|
|
|
|||
54
third_party/aom/av1/common/av1_inv_txfm1d.c
vendored
54
third_party/aom/av1/common/av1_inv_txfm1d.c
vendored
|
|
@ -19,24 +19,40 @@ void range_check_func(int32_t stage, const int32_t *input, const int32_t *buf,
|
|||
const int64_t maxValue = (1LL << (bit - 1)) - 1;
|
||||
const int64_t minValue = -(1LL << (bit - 1));
|
||||
|
||||
int in_range = 1;
|
||||
|
||||
for (int i = 0; i < size; ++i) {
|
||||
if (buf[i] < minValue || buf[i] > maxValue) {
|
||||
fprintf(stderr, "Error: coeffs contain out-of-range values\n");
|
||||
fprintf(stderr, "stage: %d\n", stage);
|
||||
fprintf(stderr, "node: %d\n", i);
|
||||
fprintf(stderr, "allowed range: [%" PRId64 ";%" PRId64 "]\n", minValue,
|
||||
maxValue);
|
||||
fprintf(stderr, "coeffs: ");
|
||||
|
||||
fprintf(stderr, "[");
|
||||
for (int j = 0; j < size; j++) {
|
||||
if (j > 0) fprintf(stderr, ", ");
|
||||
fprintf(stderr, "%d", input[j]);
|
||||
}
|
||||
fprintf(stderr, "]\n");
|
||||
assert(0);
|
||||
in_range = 0;
|
||||
}
|
||||
}
|
||||
|
||||
if (!in_range) {
|
||||
fprintf(stderr, "Error: coeffs contain out-of-range values\n");
|
||||
fprintf(stderr, "stage: %d\n", stage);
|
||||
fprintf(stderr, "allowed range: [%" PRId64 ";%" PRId64 "]\n", minValue,
|
||||
maxValue);
|
||||
|
||||
fprintf(stderr, "coeffs: ");
|
||||
|
||||
fprintf(stderr, "[");
|
||||
for (int j = 0; j < size; j++) {
|
||||
if (j > 0) fprintf(stderr, ", ");
|
||||
fprintf(stderr, "%d", input[j]);
|
||||
}
|
||||
fprintf(stderr, "]\n");
|
||||
|
||||
fprintf(stderr, " buf: ");
|
||||
|
||||
fprintf(stderr, "[");
|
||||
for (int j = 0; j < size; j++) {
|
||||
if (j > 0) fprintf(stderr, ", ");
|
||||
fprintf(stderr, "%d", buf[j]);
|
||||
}
|
||||
fprintf(stderr, "]\n\n");
|
||||
}
|
||||
|
||||
assert(in_range);
|
||||
}
|
||||
|
||||
#define range_check(stage, input, buf, size, bit) \
|
||||
|
|
@ -1577,6 +1593,16 @@ void av1_iidentity32_c(const int32_t *input, int32_t *output,
|
|||
for (int i = 0; i < 32; ++i) output[i] = input[i] * 4;
|
||||
range_check(0, input, output, 32, stage_range[0]);
|
||||
}
|
||||
|
||||
#if CONFIG_TX64X64
|
||||
void av1_iidentity64_c(const int32_t *input, int32_t *output,
|
||||
const int8_t *cos_bit, const int8_t *stage_range) {
|
||||
(void)cos_bit;
|
||||
for (int i = 0; i < 64; ++i)
|
||||
output[i] = (int32_t)dct_const_round_shift(input[i] * 4 * Sqrt2);
|
||||
range_check(0, input, output, 64, stage_range[0]);
|
||||
}
|
||||
#endif // CONFIG_TX64X64
|
||||
#endif // CONFIG_EXT_TX
|
||||
|
||||
#if CONFIG_TX64X64
|
||||
|
|
|
|||
6
third_party/aom/av1/common/av1_inv_txfm1d.h
vendored
6
third_party/aom/av1/common/av1_inv_txfm1d.h
vendored
|
|
@ -26,8 +26,10 @@ void av1_idct16_new(const int32_t *input, int32_t *output,
|
|||
const int8_t *cos_bit, const int8_t *stage_range);
|
||||
void av1_idct32_new(const int32_t *input, int32_t *output,
|
||||
const int8_t *cos_bit, const int8_t *stage_range);
|
||||
#if CONFIG_TX64X64
|
||||
void av1_idct64_new(const int32_t *input, int32_t *output,
|
||||
const int8_t *cos_bit, const int8_t *stage_range);
|
||||
#endif // CONFIG_TX64X64
|
||||
|
||||
void av1_iadst4_new(const int32_t *input, int32_t *output,
|
||||
const int8_t *cos_bit, const int8_t *stage_range);
|
||||
|
|
@ -46,6 +48,10 @@ void av1_iidentity16_c(const int32_t *input, int32_t *output,
|
|||
const int8_t *cos_bit, const int8_t *stage_range);
|
||||
void av1_iidentity32_c(const int32_t *input, int32_t *output,
|
||||
const int8_t *cos_bit, const int8_t *stage_range);
|
||||
#if CONFIG_TX64X64
|
||||
void av1_iidentity64_c(const int32_t *input, int32_t *output,
|
||||
const int8_t *cos_bit, const int8_t *stage_range);
|
||||
#endif // CONFIG_TX64X64
|
||||
#endif // CONFIG_EXT_TX
|
||||
|
||||
#ifdef __cplusplus
|
||||
|
|
|
|||
200
third_party/aom/av1/common/av1_inv_txfm1d_cfg.h
vendored
200
third_party/aom/av1/common/av1_inv_txfm1d_cfg.h
vendored
|
|
@ -13,16 +13,31 @@
|
|||
#define AV1_INV_TXFM2D_CFG_H_
|
||||
#include "av1/common/av1_inv_txfm1d.h"
|
||||
|
||||
// sum of fwd_shift_##
|
||||
#if CONFIG_CHROMA_2X2
|
||||
#if CONFIG_TX64X64
|
||||
static const int8_t fwd_shift_sum[TX_SIZES] = { 3, 2, 1, 0, -2, -4 };
|
||||
#else // CONFIG_TX64X64
|
||||
static const int8_t fwd_shift_sum[TX_SIZES] = { 3, 2, 1, 0, -2 };
|
||||
#endif // CONFIG_TX64X64
|
||||
#else // CONFIG_CHROMA_2X2
|
||||
#if CONFIG_TX64X64
|
||||
static const int8_t fwd_shift_sum[TX_SIZES] = { 2, 1, 0, -2, -4 };
|
||||
#else // CONFIG_TX64X64
|
||||
static const int8_t fwd_shift_sum[TX_SIZES] = { 2, 1, 0, -2 };
|
||||
#endif // CONFIG_TX64X64
|
||||
#endif // CONFIG_CHROMA_2X2
|
||||
|
||||
// ---------------- 4x4 1D config -----------------------
|
||||
// shift
|
||||
static const int8_t inv_shift_4[2] = { 0, -4 };
|
||||
|
||||
// stage range
|
||||
static const int8_t inv_stage_range_col_dct_4[4] = { 18, 18, 17, 17 };
|
||||
static const int8_t inv_stage_range_row_dct_4[4] = { 18, 18, 18, 18 };
|
||||
static const int8_t inv_stage_range_col_adst_4[6] = { 18, 18, 18, 18, 17, 17 };
|
||||
static const int8_t inv_stage_range_row_adst_4[6] = { 18, 18, 18, 18, 18, 18 };
|
||||
static const int8_t inv_stage_range_idx_4[1] = { 18 };
|
||||
static const int8_t inv_stage_range_col_dct_4[4] = { 3, 3, 2, 2 };
|
||||
static const int8_t inv_stage_range_row_dct_4[4] = { 3, 3, 3, 3 };
|
||||
static const int8_t inv_stage_range_col_adst_4[6] = { 3, 3, 3, 3, 2, 2 };
|
||||
static const int8_t inv_stage_range_row_adst_4[6] = { 3, 3, 3, 3, 3, 3 };
|
||||
static const int8_t inv_stage_range_idx_4[1] = { 0 };
|
||||
|
||||
// cos bit
|
||||
static const int8_t inv_cos_bit_col_dct_4[4] = { 13, 13, 13, 13 };
|
||||
|
|
@ -35,13 +50,11 @@ static const int8_t inv_cos_bit_row_adst_4[6] = { 13, 13, 13, 13, 13, 13 };
|
|||
static const int8_t inv_shift_8[2] = { 0, -5 };
|
||||
|
||||
// stage range
|
||||
static const int8_t inv_stage_range_col_dct_8[6] = { 19, 19, 19, 19, 18, 18 };
|
||||
static const int8_t inv_stage_range_row_dct_8[6] = { 19, 19, 19, 19, 19, 19 };
|
||||
static const int8_t inv_stage_range_col_adst_8[8] = { 19, 19, 19, 19,
|
||||
19, 19, 18, 18 };
|
||||
static const int8_t inv_stage_range_row_adst_8[8] = { 19, 19, 19, 19,
|
||||
19, 19, 19, 19 };
|
||||
static const int8_t inv_stage_range_idx_8[1] = { 19 };
|
||||
static const int8_t inv_stage_range_col_dct_8[6] = { 5, 5, 5, 5, 4, 4 };
|
||||
static const int8_t inv_stage_range_row_dct_8[6] = { 5, 5, 5, 5, 5, 5 };
|
||||
static const int8_t inv_stage_range_col_adst_8[8] = { 5, 5, 5, 5, 5, 5, 4, 4 };
|
||||
static const int8_t inv_stage_range_row_adst_8[8] = { 5, 5, 5, 5, 5, 5, 5, 5 };
|
||||
static const int8_t inv_stage_range_idx_8[1] = { 0 };
|
||||
|
||||
// cos bit
|
||||
static const int8_t inv_cos_bit_col_dct_8[6] = { 13, 13, 13, 13, 13, 13 };
|
||||
|
|
@ -58,15 +71,13 @@ static const int8_t inv_cos_bit_row_adst_8[8] = {
|
|||
static const int8_t inv_shift_16[2] = { -1, -5 };
|
||||
|
||||
// stage range
|
||||
static const int8_t inv_stage_range_col_dct_16[8] = { 19, 19, 19, 19,
|
||||
19, 19, 18, 18 };
|
||||
static const int8_t inv_stage_range_row_dct_16[8] = { 20, 20, 20, 20,
|
||||
20, 20, 20, 20 };
|
||||
static const int8_t inv_stage_range_col_adst_16[10] = { 19, 19, 19, 19, 19,
|
||||
19, 19, 19, 18, 18 };
|
||||
static const int8_t inv_stage_range_row_adst_16[10] = { 20, 20, 20, 20, 20,
|
||||
20, 20, 20, 20, 20 };
|
||||
static const int8_t inv_stage_range_idx_16[1] = { 20 };
|
||||
static const int8_t inv_stage_range_col_dct_16[8] = { 7, 7, 7, 7, 7, 7, 6, 6 };
|
||||
static const int8_t inv_stage_range_row_dct_16[8] = { 7, 7, 7, 7, 7, 7, 7, 7 };
|
||||
static const int8_t inv_stage_range_col_adst_16[10] = { 7, 7, 7, 7, 7,
|
||||
7, 7, 7, 6, 6 };
|
||||
static const int8_t inv_stage_range_row_adst_16[10] = { 7, 7, 7, 7, 7,
|
||||
7, 7, 7, 7, 7 };
|
||||
static const int8_t inv_stage_range_idx_16[1] = { 0 };
|
||||
|
||||
// cos bit
|
||||
static const int8_t inv_cos_bit_col_dct_16[8] = {
|
||||
|
|
@ -85,17 +96,15 @@ static const int8_t inv_cos_bit_row_adst_16[10] = { 12, 12, 12, 12, 12,
|
|||
static const int8_t inv_shift_32[2] = { -1, -5 };
|
||||
|
||||
// stage range
|
||||
static const int8_t inv_stage_range_col_dct_32[10] = { 19, 19, 19, 19, 19,
|
||||
19, 19, 19, 18, 18 };
|
||||
static const int8_t inv_stage_range_row_dct_32[10] = { 20, 20, 20, 20, 20,
|
||||
20, 20, 20, 20, 20 };
|
||||
static const int8_t inv_stage_range_col_adst_32[12] = {
|
||||
19, 19, 19, 19, 19, 19, 19, 19, 19, 19, 18, 18
|
||||
};
|
||||
static const int8_t inv_stage_range_row_adst_32[12] = {
|
||||
20, 20, 20, 20, 20, 20, 20, 20, 20, 20, 20, 20
|
||||
};
|
||||
static const int8_t inv_stage_range_idx_32[1] = { 20 };
|
||||
static const int8_t inv_stage_range_col_dct_32[10] = { 9, 9, 9, 9, 9,
|
||||
9, 9, 9, 8, 8 };
|
||||
static const int8_t inv_stage_range_row_dct_32[10] = { 9, 9, 9, 9, 9,
|
||||
9, 9, 9, 9, 9 };
|
||||
static const int8_t inv_stage_range_col_adst_32[12] = { 9, 9, 9, 9, 9, 9,
|
||||
9, 9, 9, 9, 8, 8 };
|
||||
static const int8_t inv_stage_range_row_adst_32[12] = { 9, 9, 9, 9, 9, 9,
|
||||
9, 9, 9, 9, 9, 9 };
|
||||
static const int8_t inv_stage_range_idx_32[1] = { 0 };
|
||||
|
||||
// cos bit
|
||||
static const int8_t inv_cos_bit_col_dct_32[10] = { 13, 13, 13, 13, 13,
|
||||
|
|
@ -109,14 +118,15 @@ static const int8_t inv_cos_bit_row_adst_32[12] = { 12, 12, 12, 12, 12, 12,
|
|||
|
||||
// ---------------- 64x64 1D constants -----------------------
|
||||
// shift
|
||||
static const int8_t inv_shift_64[2] = { -1, -7 };
|
||||
static const int8_t inv_shift_64[2] = { -1, -5 };
|
||||
|
||||
// stage range
|
||||
static const int8_t inv_stage_range_col_dct_64[12] = { 19, 19, 19, 19, 19, 19,
|
||||
19, 19, 19, 19, 18, 18 };
|
||||
static const int8_t inv_stage_range_row_dct_64[12] = { 20, 20, 20, 20, 20, 20,
|
||||
20, 20, 20, 20, 20, 20 };
|
||||
static const int8_t inv_stage_range_idx_64[1] = { 20 };
|
||||
static const int8_t inv_stage_range_col_dct_64[12] = { 11, 11, 11, 11, 11, 11,
|
||||
11, 11, 11, 11, 10, 10 };
|
||||
static const int8_t inv_stage_range_row_dct_64[12] = { 11, 11, 11, 11, 11, 11,
|
||||
11, 11, 11, 11, 11, 11 };
|
||||
|
||||
static const int8_t inv_stage_range_idx_64[1] = { 0 };
|
||||
|
||||
// cos bit
|
||||
static const int8_t inv_cos_bit_col_dct_64[12] = { 13, 13, 13, 13, 13, 13,
|
||||
|
|
@ -126,9 +136,8 @@ static const int8_t inv_cos_bit_row_dct_64[12] = { 12, 12, 12, 12, 12, 12,
|
|||
|
||||
// ---------------- row config inv_dct_4 ----------------
|
||||
static const TXFM_1D_CFG inv_txfm_1d_row_cfg_dct_4 = {
|
||||
4, // .txfm_size
|
||||
4, // .stage_num
|
||||
// 0, // .log_scale
|
||||
4, // .txfm_size
|
||||
4, // .stage_num
|
||||
inv_shift_4, // .shift
|
||||
inv_stage_range_row_dct_4, // .stage_range
|
||||
inv_cos_bit_row_dct_4, // .cos_bit
|
||||
|
|
@ -137,9 +146,8 @@ static const TXFM_1D_CFG inv_txfm_1d_row_cfg_dct_4 = {
|
|||
|
||||
// ---------------- row config inv_dct_8 ----------------
|
||||
static const TXFM_1D_CFG inv_txfm_1d_row_cfg_dct_8 = {
|
||||
8, // .txfm_size
|
||||
6, // .stage_num
|
||||
// 0, // .log_scale
|
||||
8, // .txfm_size
|
||||
6, // .stage_num
|
||||
inv_shift_8, // .shift
|
||||
inv_stage_range_row_dct_8, // .stage_range
|
||||
inv_cos_bit_row_dct_8, // .cos_bit_
|
||||
|
|
@ -147,9 +155,8 @@ static const TXFM_1D_CFG inv_txfm_1d_row_cfg_dct_8 = {
|
|||
};
|
||||
// ---------------- row config inv_dct_16 ----------------
|
||||
static const TXFM_1D_CFG inv_txfm_1d_row_cfg_dct_16 = {
|
||||
16, // .txfm_size
|
||||
8, // .stage_num
|
||||
// 0, // .log_scale
|
||||
16, // .txfm_size
|
||||
8, // .stage_num
|
||||
inv_shift_16, // .shift
|
||||
inv_stage_range_row_dct_16, // .stage_range
|
||||
inv_cos_bit_row_dct_16, // .cos_bit
|
||||
|
|
@ -158,15 +165,15 @@ static const TXFM_1D_CFG inv_txfm_1d_row_cfg_dct_16 = {
|
|||
|
||||
// ---------------- row config inv_dct_32 ----------------
|
||||
static const TXFM_1D_CFG inv_txfm_1d_row_cfg_dct_32 = {
|
||||
32, // .txfm_size
|
||||
10, // .stage_num
|
||||
// 1, // .log_scale
|
||||
32, // .txfm_size
|
||||
10, // .stage_num
|
||||
inv_shift_32, // .shift
|
||||
inv_stage_range_row_dct_32, // .stage_range
|
||||
inv_cos_bit_row_dct_32, // .cos_bit_row
|
||||
TXFM_TYPE_DCT32 // .txfm_type
|
||||
};
|
||||
|
||||
#if CONFIG_TX64X64
|
||||
// ---------------- row config inv_dct_64 ----------------
|
||||
static const TXFM_1D_CFG inv_txfm_1d_row_cfg_dct_64 = {
|
||||
64, // .txfm_size
|
||||
|
|
@ -176,12 +183,12 @@ static const TXFM_1D_CFG inv_txfm_1d_row_cfg_dct_64 = {
|
|||
inv_cos_bit_row_dct_64, // .cos_bit
|
||||
TXFM_TYPE_DCT64, // .txfm_type_col
|
||||
};
|
||||
#endif // CONFIG_TX64X64
|
||||
|
||||
// ---------------- row config inv_adst_4 ----------------
|
||||
static const TXFM_1D_CFG inv_txfm_1d_row_cfg_adst_4 = {
|
||||
4, // .txfm_size
|
||||
6, // .stage_num
|
||||
// 0, // .log_scale
|
||||
4, // .txfm_size
|
||||
6, // .stage_num
|
||||
inv_shift_4, // .shift
|
||||
inv_stage_range_row_adst_4, // .stage_range
|
||||
inv_cos_bit_row_adst_4, // .cos_bit
|
||||
|
|
@ -190,9 +197,8 @@ static const TXFM_1D_CFG inv_txfm_1d_row_cfg_adst_4 = {
|
|||
|
||||
// ---------------- row config inv_adst_8 ----------------
|
||||
static const TXFM_1D_CFG inv_txfm_1d_row_cfg_adst_8 = {
|
||||
8, // .txfm_size
|
||||
8, // .stage_num
|
||||
// 0, // .log_scale
|
||||
8, // .txfm_size
|
||||
8, // .stage_num
|
||||
inv_shift_8, // .shift
|
||||
inv_stage_range_row_adst_8, // .stage_range
|
||||
inv_cos_bit_row_adst_8, // .cos_bit
|
||||
|
|
@ -201,9 +207,8 @@ static const TXFM_1D_CFG inv_txfm_1d_row_cfg_adst_8 = {
|
|||
|
||||
// ---------------- row config inv_adst_16 ----------------
|
||||
static const TXFM_1D_CFG inv_txfm_1d_row_cfg_adst_16 = {
|
||||
16, // .txfm_size
|
||||
10, // .stage_num
|
||||
// 0, // .log_scale
|
||||
16, // .txfm_size
|
||||
10, // .stage_num
|
||||
inv_shift_16, // .shift
|
||||
inv_stage_range_row_adst_16, // .stage_range
|
||||
inv_cos_bit_row_adst_16, // .cos_bit
|
||||
|
|
@ -212,9 +217,8 @@ static const TXFM_1D_CFG inv_txfm_1d_row_cfg_adst_16 = {
|
|||
|
||||
// ---------------- row config inv_adst_32 ----------------
|
||||
static const TXFM_1D_CFG inv_txfm_1d_row_cfg_adst_32 = {
|
||||
32, // .txfm_size
|
||||
12, // .stage_num
|
||||
// 1, // .log_scale
|
||||
32, // .txfm_size
|
||||
12, // .stage_num
|
||||
inv_shift_32, // .shift
|
||||
inv_stage_range_row_adst_32, // .stage_range
|
||||
inv_cos_bit_row_adst_32, // .cos_bit
|
||||
|
|
@ -223,9 +227,8 @@ static const TXFM_1D_CFG inv_txfm_1d_row_cfg_adst_32 = {
|
|||
|
||||
// ---------------- col config inv_dct_4 ----------------
|
||||
static const TXFM_1D_CFG inv_txfm_1d_col_cfg_dct_4 = {
|
||||
4, // .txfm_size
|
||||
4, // .stage_num
|
||||
// 0, // .log_scale
|
||||
4, // .txfm_size
|
||||
4, // .stage_num
|
||||
inv_shift_4, // .shift
|
||||
inv_stage_range_col_dct_4, // .stage_range
|
||||
inv_cos_bit_col_dct_4, // .cos_bit
|
||||
|
|
@ -234,9 +237,8 @@ static const TXFM_1D_CFG inv_txfm_1d_col_cfg_dct_4 = {
|
|||
|
||||
// ---------------- col config inv_dct_8 ----------------
|
||||
static const TXFM_1D_CFG inv_txfm_1d_col_cfg_dct_8 = {
|
||||
8, // .txfm_size
|
||||
6, // .stage_num
|
||||
// 0, // .log_scale
|
||||
8, // .txfm_size
|
||||
6, // .stage_num
|
||||
inv_shift_8, // .shift
|
||||
inv_stage_range_col_dct_8, // .stage_range
|
||||
inv_cos_bit_col_dct_8, // .cos_bit_
|
||||
|
|
@ -244,9 +246,8 @@ static const TXFM_1D_CFG inv_txfm_1d_col_cfg_dct_8 = {
|
|||
};
|
||||
// ---------------- col config inv_dct_16 ----------------
|
||||
static const TXFM_1D_CFG inv_txfm_1d_col_cfg_dct_16 = {
|
||||
16, // .txfm_size
|
||||
8, // .stage_num
|
||||
// 0, // .log_scale
|
||||
16, // .txfm_size
|
||||
8, // .stage_num
|
||||
inv_shift_16, // .shift
|
||||
inv_stage_range_col_dct_16, // .stage_range
|
||||
inv_cos_bit_col_dct_16, // .cos_bit
|
||||
|
|
@ -255,9 +256,8 @@ static const TXFM_1D_CFG inv_txfm_1d_col_cfg_dct_16 = {
|
|||
|
||||
// ---------------- col config inv_dct_32 ----------------
|
||||
static const TXFM_1D_CFG inv_txfm_1d_col_cfg_dct_32 = {
|
||||
32, // .txfm_size
|
||||
10, // .stage_num
|
||||
// 1, // .log_scale
|
||||
32, // .txfm_size
|
||||
10, // .stage_num
|
||||
inv_shift_32, // .shift
|
||||
inv_stage_range_col_dct_32, // .stage_range
|
||||
inv_cos_bit_col_dct_32, // .cos_bit_col
|
||||
|
|
@ -276,9 +276,8 @@ static const TXFM_1D_CFG inv_txfm_1d_col_cfg_dct_64 = {
|
|||
|
||||
// ---------------- col config inv_adst_4 ----------------
|
||||
static const TXFM_1D_CFG inv_txfm_1d_col_cfg_adst_4 = {
|
||||
4, // .txfm_size
|
||||
6, // .stage_num
|
||||
// 0, // .log_scale
|
||||
4, // .txfm_size
|
||||
6, // .stage_num
|
||||
inv_shift_4, // .shift
|
||||
inv_stage_range_col_adst_4, // .stage_range
|
||||
inv_cos_bit_col_adst_4, // .cos_bit
|
||||
|
|
@ -287,9 +286,8 @@ static const TXFM_1D_CFG inv_txfm_1d_col_cfg_adst_4 = {
|
|||
|
||||
// ---------------- col config inv_adst_8 ----------------
|
||||
static const TXFM_1D_CFG inv_txfm_1d_col_cfg_adst_8 = {
|
||||
8, // .txfm_size
|
||||
8, // .stage_num
|
||||
// 0, // .log_scale
|
||||
8, // .txfm_size
|
||||
8, // .stage_num
|
||||
inv_shift_8, // .shift
|
||||
inv_stage_range_col_adst_8, // .stage_range
|
||||
inv_cos_bit_col_adst_8, // .cos_bit
|
||||
|
|
@ -298,9 +296,8 @@ static const TXFM_1D_CFG inv_txfm_1d_col_cfg_adst_8 = {
|
|||
|
||||
// ---------------- col config inv_adst_16 ----------------
|
||||
static const TXFM_1D_CFG inv_txfm_1d_col_cfg_adst_16 = {
|
||||
16, // .txfm_size
|
||||
10, // .stage_num
|
||||
// 0, // .log_scale
|
||||
16, // .txfm_size
|
||||
10, // .stage_num
|
||||
inv_shift_16, // .shift
|
||||
inv_stage_range_col_adst_16, // .stage_range
|
||||
inv_cos_bit_col_adst_16, // .cos_bit
|
||||
|
|
@ -309,9 +306,8 @@ static const TXFM_1D_CFG inv_txfm_1d_col_cfg_adst_16 = {
|
|||
|
||||
// ---------------- col config inv_adst_32 ----------------
|
||||
static const TXFM_1D_CFG inv_txfm_1d_col_cfg_adst_32 = {
|
||||
32, // .txfm_size
|
||||
12, // .stage_num
|
||||
// 1, // .log_scale
|
||||
32, // .txfm_size
|
||||
12, // .stage_num
|
||||
inv_shift_32, // .shift
|
||||
inv_stage_range_col_adst_32, // .stage_range
|
||||
inv_cos_bit_col_adst_32, // .cos_bit
|
||||
|
|
@ -322,9 +318,8 @@ static const TXFM_1D_CFG inv_txfm_1d_col_cfg_adst_32 = {
|
|||
// identity does not need to differentiate between row and col
|
||||
// ---------------- row/col config inv_identity_4 ----------
|
||||
static const TXFM_1D_CFG inv_txfm_1d_cfg_identity_4 = {
|
||||
4, // .txfm_size
|
||||
1, // .stage_num
|
||||
// 0, // .log_scale
|
||||
4, // .txfm_size
|
||||
1, // .stage_num
|
||||
inv_shift_4, // .shift
|
||||
inv_stage_range_idx_4, // .stage_range
|
||||
NULL, // .cos_bit
|
||||
|
|
@ -333,9 +328,8 @@ static const TXFM_1D_CFG inv_txfm_1d_cfg_identity_4 = {
|
|||
|
||||
// ---------------- row/col config inv_identity_8 ----------------
|
||||
static const TXFM_1D_CFG inv_txfm_1d_cfg_identity_8 = {
|
||||
8, // .txfm_size
|
||||
1, // .stage_num
|
||||
// 0, // .log_scale
|
||||
8, // .txfm_size
|
||||
1, // .stage_num
|
||||
inv_shift_8, // .shift
|
||||
inv_stage_range_idx_8, // .stage_range
|
||||
NULL, // .cos_bit
|
||||
|
|
@ -344,9 +338,8 @@ static const TXFM_1D_CFG inv_txfm_1d_cfg_identity_8 = {
|
|||
|
||||
// ---------------- row/col config inv_identity_16 ----------------
|
||||
static const TXFM_1D_CFG inv_txfm_1d_cfg_identity_16 = {
|
||||
16, // .txfm_size
|
||||
1, // .stage_num
|
||||
// 0, // .log_scale
|
||||
16, // .txfm_size
|
||||
1, // .stage_num
|
||||
inv_shift_16, // .shift
|
||||
inv_stage_range_idx_16, // .stage_range
|
||||
NULL, // .cos_bit
|
||||
|
|
@ -355,13 +348,24 @@ static const TXFM_1D_CFG inv_txfm_1d_cfg_identity_16 = {
|
|||
|
||||
// ---------------- row/col config inv_identity_32 ----------------
|
||||
static const TXFM_1D_CFG inv_txfm_1d_cfg_identity_32 = {
|
||||
32, // .txfm_size
|
||||
1, // .stage_num
|
||||
// 1, // .log_scale
|
||||
32, // .txfm_size
|
||||
1, // .stage_num
|
||||
inv_shift_32, // .shift
|
||||
inv_stage_range_idx_32, // .stage_range
|
||||
NULL, // .cos_bit
|
||||
TXFM_TYPE_IDENTITY32, // .txfm_type
|
||||
};
|
||||
|
||||
#if CONFIG_TX64X64
|
||||
// ---------------- row/col config inv_identity_32 ----------------
|
||||
static const TXFM_1D_CFG inv_txfm_1d_cfg_identity_64 = {
|
||||
64, // .txfm_size
|
||||
1, // .stage_num
|
||||
inv_shift_64, // .shift
|
||||
inv_stage_range_idx_64, // .stage_range
|
||||
NULL, // .cos_bit
|
||||
TXFM_TYPE_IDENTITY64, // .txfm_type
|
||||
};
|
||||
#endif // CONFIG_TX64X64
|
||||
#endif // CONFIG_EXT_TX
|
||||
#endif // AV1_INV_TXFM2D_CFG_H_
|
||||
|
|
|
|||
248
third_party/aom/av1/common/av1_inv_txfm2d.c
vendored
248
third_party/aom/av1/common/av1_inv_txfm2d.c
vendored
|
|
@ -22,6 +22,9 @@ static INLINE TxfmFunc inv_txfm_type_to_func(TXFM_TYPE txfm_type) {
|
|||
case TXFM_TYPE_DCT8: return av1_idct8_new;
|
||||
case TXFM_TYPE_DCT16: return av1_idct16_new;
|
||||
case TXFM_TYPE_DCT32: return av1_idct32_new;
|
||||
#if CONFIG_TX64X64
|
||||
case TXFM_TYPE_DCT64: return av1_idct64_new;
|
||||
#endif // CONFIG_TX64X64
|
||||
case TXFM_TYPE_ADST4: return av1_iadst4_new;
|
||||
case TXFM_TYPE_ADST8: return av1_iadst8_new;
|
||||
case TXFM_TYPE_ADST16: return av1_iadst16_new;
|
||||
|
|
@ -31,6 +34,9 @@ static INLINE TxfmFunc inv_txfm_type_to_func(TXFM_TYPE txfm_type) {
|
|||
case TXFM_TYPE_IDENTITY8: return av1_iidentity8_c;
|
||||
case TXFM_TYPE_IDENTITY16: return av1_iidentity16_c;
|
||||
case TXFM_TYPE_IDENTITY32: return av1_iidentity32_c;
|
||||
#if CONFIG_TX64X64
|
||||
case TXFM_TYPE_IDENTITY64: return av1_iidentity64_c;
|
||||
#endif // CONFIG_TX64X64
|
||||
#endif // CONFIG_EXT_TX
|
||||
default: assert(0); return NULL;
|
||||
}
|
||||
|
|
@ -43,14 +49,22 @@ static const TXFM_1D_CFG *inv_txfm_col_cfg_ls[TX_TYPES_1D][TX_SIZES] = {
|
|||
NULL,
|
||||
#endif
|
||||
&inv_txfm_1d_col_cfg_dct_4, &inv_txfm_1d_col_cfg_dct_8,
|
||||
&inv_txfm_1d_col_cfg_dct_16, &inv_txfm_1d_col_cfg_dct_32 },
|
||||
&inv_txfm_1d_col_cfg_dct_16, &inv_txfm_1d_col_cfg_dct_32,
|
||||
#if CONFIG_TX64X64
|
||||
&inv_txfm_1d_col_cfg_dct_64
|
||||
#endif // CONFIG_TX64X64
|
||||
},
|
||||
// ADST
|
||||
{
|
||||
#if CONFIG_CHROMA_2X2
|
||||
NULL,
|
||||
#endif
|
||||
&inv_txfm_1d_col_cfg_adst_4, &inv_txfm_1d_col_cfg_adst_8,
|
||||
&inv_txfm_1d_col_cfg_adst_16, &inv_txfm_1d_col_cfg_adst_32 },
|
||||
&inv_txfm_1d_col_cfg_adst_16, &inv_txfm_1d_col_cfg_adst_32,
|
||||
#if CONFIG_TX64X64
|
||||
NULL
|
||||
#endif // CONFIG_TX64X64
|
||||
},
|
||||
#if CONFIG_EXT_TX
|
||||
// FLIPADST
|
||||
{
|
||||
|
|
@ -58,14 +72,22 @@ static const TXFM_1D_CFG *inv_txfm_col_cfg_ls[TX_TYPES_1D][TX_SIZES] = {
|
|||
NULL,
|
||||
#endif
|
||||
&inv_txfm_1d_col_cfg_adst_4, &inv_txfm_1d_col_cfg_adst_8,
|
||||
&inv_txfm_1d_col_cfg_adst_16, &inv_txfm_1d_col_cfg_adst_32 },
|
||||
&inv_txfm_1d_col_cfg_adst_16, &inv_txfm_1d_col_cfg_adst_32,
|
||||
#if CONFIG_TX64X64
|
||||
NULL
|
||||
#endif // CONFIG_TX64X64
|
||||
},
|
||||
// IDENTITY
|
||||
{
|
||||
#if CONFIG_CHROMA_2X2
|
||||
NULL,
|
||||
#endif
|
||||
&inv_txfm_1d_cfg_identity_4, &inv_txfm_1d_cfg_identity_8,
|
||||
&inv_txfm_1d_cfg_identity_16, &inv_txfm_1d_cfg_identity_32 },
|
||||
&inv_txfm_1d_cfg_identity_16, &inv_txfm_1d_cfg_identity_32,
|
||||
#if CONFIG_TX64X64
|
||||
&inv_txfm_1d_cfg_identity_64
|
||||
#endif // CONFIG_TX64X64
|
||||
},
|
||||
#endif // CONFIG_EXT_TX
|
||||
};
|
||||
|
||||
|
|
@ -76,14 +98,22 @@ static const TXFM_1D_CFG *inv_txfm_row_cfg_ls[TX_TYPES_1D][TX_SIZES] = {
|
|||
NULL,
|
||||
#endif
|
||||
&inv_txfm_1d_row_cfg_dct_4, &inv_txfm_1d_row_cfg_dct_8,
|
||||
&inv_txfm_1d_row_cfg_dct_16, &inv_txfm_1d_row_cfg_dct_32 },
|
||||
&inv_txfm_1d_row_cfg_dct_16, &inv_txfm_1d_row_cfg_dct_32,
|
||||
#if CONFIG_TX64X64
|
||||
&inv_txfm_1d_row_cfg_dct_64,
|
||||
#endif // CONFIG_TX64X64
|
||||
},
|
||||
// ADST
|
||||
{
|
||||
#if CONFIG_CHROMA_2X2
|
||||
NULL,
|
||||
#endif
|
||||
&inv_txfm_1d_row_cfg_adst_4, &inv_txfm_1d_row_cfg_adst_8,
|
||||
&inv_txfm_1d_row_cfg_adst_16, &inv_txfm_1d_row_cfg_adst_32 },
|
||||
&inv_txfm_1d_row_cfg_adst_16, &inv_txfm_1d_row_cfg_adst_32,
|
||||
#if CONFIG_TX64X64
|
||||
NULL
|
||||
#endif // CONFIG_TX64X64
|
||||
},
|
||||
#if CONFIG_EXT_TX
|
||||
// FLIPADST
|
||||
{
|
||||
|
|
@ -91,30 +121,39 @@ static const TXFM_1D_CFG *inv_txfm_row_cfg_ls[TX_TYPES_1D][TX_SIZES] = {
|
|||
NULL,
|
||||
#endif
|
||||
&inv_txfm_1d_row_cfg_adst_4, &inv_txfm_1d_row_cfg_adst_8,
|
||||
&inv_txfm_1d_row_cfg_adst_16, &inv_txfm_1d_row_cfg_adst_32 },
|
||||
&inv_txfm_1d_row_cfg_adst_16, &inv_txfm_1d_row_cfg_adst_32,
|
||||
#if CONFIG_TX64X64
|
||||
NULL
|
||||
#endif // CONFIG_TX64X64
|
||||
},
|
||||
// IDENTITY
|
||||
{
|
||||
#if CONFIG_CHROMA_2X2
|
||||
NULL,
|
||||
#endif
|
||||
&inv_txfm_1d_cfg_identity_4, &inv_txfm_1d_cfg_identity_8,
|
||||
&inv_txfm_1d_cfg_identity_16, &inv_txfm_1d_cfg_identity_32 },
|
||||
&inv_txfm_1d_cfg_identity_16, &inv_txfm_1d_cfg_identity_32,
|
||||
#if CONFIG_TX64X64
|
||||
&inv_txfm_1d_cfg_identity_64
|
||||
#endif // CONFIG_TX64X64
|
||||
},
|
||||
#endif // CONFIG_EXT_TX
|
||||
};
|
||||
|
||||
TXFM_2D_FLIP_CFG av1_get_inv_txfm_cfg(int tx_type, int tx_size) {
|
||||
TXFM_2D_FLIP_CFG av1_get_inv_txfm_cfg(TX_TYPE tx_type, TX_SIZE tx_size) {
|
||||
TXFM_2D_FLIP_CFG cfg;
|
||||
set_flip_cfg(tx_type, &cfg);
|
||||
const int tx_type_col = vtx_tab[tx_type];
|
||||
const int tx_type_row = htx_tab[tx_type];
|
||||
const int tx_size_col = txsize_vert_map[tx_size];
|
||||
const int tx_size_row = txsize_horz_map[tx_size];
|
||||
const TX_TYPE_1D tx_type_col = vtx_tab[tx_type];
|
||||
const TX_TYPE_1D tx_type_row = htx_tab[tx_type];
|
||||
const TX_SIZE tx_size_col = txsize_vert_map[tx_size];
|
||||
const TX_SIZE tx_size_row = txsize_horz_map[tx_size];
|
||||
cfg.col_cfg = inv_txfm_col_cfg_ls[tx_type_col][tx_size_col];
|
||||
cfg.row_cfg = inv_txfm_row_cfg_ls[tx_type_row][tx_size_row];
|
||||
return cfg;
|
||||
}
|
||||
|
||||
TXFM_2D_FLIP_CFG av1_get_inv_txfm_64x64_cfg(int tx_type) {
|
||||
#if CONFIG_TX64X64
|
||||
TXFM_2D_FLIP_CFG av1_get_inv_txfm_64x64_cfg(TX_TYPE tx_type) {
|
||||
TXFM_2D_FLIP_CFG cfg = { 0, 0, NULL, NULL };
|
||||
switch (tx_type) {
|
||||
case DCT_DCT:
|
||||
|
|
@ -127,9 +166,36 @@ TXFM_2D_FLIP_CFG av1_get_inv_txfm_64x64_cfg(int tx_type) {
|
|||
return cfg;
|
||||
}
|
||||
|
||||
static INLINE void inv_txfm2d_add_c(const int32_t *input, uint16_t *output,
|
||||
int stride, TXFM_2D_FLIP_CFG *cfg,
|
||||
int32_t *txfm_buf, int bd) {
|
||||
TXFM_2D_FLIP_CFG av1_get_inv_txfm_32x64_cfg(int tx_type) {
|
||||
TXFM_2D_FLIP_CFG cfg = { 0, 0, NULL, NULL };
|
||||
switch (tx_type) {
|
||||
case DCT_DCT:
|
||||
cfg.col_cfg = &inv_txfm_1d_col_cfg_dct_64;
|
||||
cfg.row_cfg = &inv_txfm_1d_row_cfg_dct_32;
|
||||
set_flip_cfg(tx_type, &cfg);
|
||||
break;
|
||||
default: assert(0);
|
||||
}
|
||||
return cfg;
|
||||
}
|
||||
|
||||
TXFM_2D_FLIP_CFG av1_get_inv_txfm_64x32_cfg(int tx_type) {
|
||||
TXFM_2D_FLIP_CFG cfg = { 0, 0, NULL, NULL };
|
||||
switch (tx_type) {
|
||||
case DCT_DCT:
|
||||
cfg.col_cfg = &inv_txfm_1d_col_cfg_dct_32;
|
||||
cfg.row_cfg = &inv_txfm_1d_row_cfg_dct_64;
|
||||
set_flip_cfg(tx_type, &cfg);
|
||||
break;
|
||||
default: assert(0);
|
||||
}
|
||||
return cfg;
|
||||
}
|
||||
#endif // CONFIG_TX64X64
|
||||
|
||||
void av1_gen_inv_stage_range(int8_t *stage_range_col, int8_t *stage_range_row,
|
||||
const TXFM_2D_FLIP_CFG *cfg, int8_t fwd_shift,
|
||||
int bd) {
|
||||
// Note when assigning txfm_size_col, we use the txfm_size from the
|
||||
// row configuration and vice versa. This is intentionally done to
|
||||
// accurately perform rectangular transforms. When the transform is
|
||||
|
|
@ -141,8 +207,38 @@ static INLINE void inv_txfm2d_add_c(const int32_t *input, uint16_t *output,
|
|||
// Take the shift from the larger dimension in the rectangular case.
|
||||
const int8_t *shift = (txfm_size_col > txfm_size_row) ? cfg->row_cfg->shift
|
||||
: cfg->col_cfg->shift;
|
||||
const int8_t *stage_range_col = cfg->col_cfg->stage_range;
|
||||
const int8_t *stage_range_row = cfg->row_cfg->stage_range;
|
||||
// i < MAX_TXFM_STAGE_NUM will mute above array bounds warning
|
||||
for (int i = 0; i < cfg->row_cfg->stage_num && i < MAX_TXFM_STAGE_NUM; ++i) {
|
||||
stage_range_row[i] = cfg->row_cfg->stage_range[i] + fwd_shift + bd + 1;
|
||||
}
|
||||
// i < MAX_TXFM_STAGE_NUM will mute above array bounds warning
|
||||
for (int i = 0; i < cfg->col_cfg->stage_num && i < MAX_TXFM_STAGE_NUM; ++i) {
|
||||
stage_range_col[i] =
|
||||
cfg->col_cfg->stage_range[i] + fwd_shift + shift[0] + bd + 1;
|
||||
}
|
||||
}
|
||||
|
||||
static INLINE void inv_txfm2d_add_c(const int32_t *input, uint16_t *output,
|
||||
int stride, TXFM_2D_FLIP_CFG *cfg,
|
||||
int32_t *txfm_buf, int8_t fwd_shift,
|
||||
int bd) {
|
||||
// Note when assigning txfm_size_col, we use the txfm_size from the
|
||||
// row configuration and vice versa. This is intentionally done to
|
||||
// accurately perform rectangular transforms. When the transform is
|
||||
// rectangular, the number of columns will be the same as the
|
||||
// txfm_size stored in the row cfg struct. It will make no difference
|
||||
// for square transforms.
|
||||
const int txfm_size_col = cfg->row_cfg->txfm_size;
|
||||
const int txfm_size_row = cfg->col_cfg->txfm_size;
|
||||
// Take the shift from the larger dimension in the rectangular case.
|
||||
const int8_t *shift = (txfm_size_col > txfm_size_row) ? cfg->row_cfg->shift
|
||||
: cfg->col_cfg->shift;
|
||||
int8_t stage_range_row[MAX_TXFM_STAGE_NUM];
|
||||
int8_t stage_range_col[MAX_TXFM_STAGE_NUM];
|
||||
assert(cfg->row_cfg->stage_num <= MAX_TXFM_STAGE_NUM);
|
||||
assert(cfg->col_cfg->stage_num <= MAX_TXFM_STAGE_NUM);
|
||||
av1_gen_inv_stage_range(stage_range_col, stage_range_row, cfg, fwd_shift, bd);
|
||||
|
||||
const int8_t *cos_bit_col = cfg->col_cfg->cos_bit;
|
||||
const int8_t *cos_bit_row = cfg->row_cfg->cos_bit;
|
||||
const TxfmFunc txfm_func_col = inv_txfm_type_to_func(cfg->col_cfg->txfm_type);
|
||||
|
|
@ -198,74 +294,158 @@ static INLINE void inv_txfm2d_add_c(const int32_t *input, uint16_t *output,
|
|||
|
||||
static INLINE void inv_txfm2d_add_facade(const int32_t *input, uint16_t *output,
|
||||
int stride, int32_t *txfm_buf,
|
||||
int tx_type, int tx_size, int bd) {
|
||||
TX_TYPE tx_type, TX_SIZE tx_size,
|
||||
int bd) {
|
||||
TXFM_2D_FLIP_CFG cfg = av1_get_inv_txfm_cfg(tx_type, tx_size);
|
||||
inv_txfm2d_add_c(input, output, stride, &cfg, txfm_buf, bd);
|
||||
TX_SIZE tx_size_sqr = txsize_sqr_map[tx_size];
|
||||
inv_txfm2d_add_c(input, output, stride, &cfg, txfm_buf,
|
||||
fwd_shift_sum[tx_size_sqr], bd);
|
||||
}
|
||||
|
||||
void av1_inv_txfm2d_add_4x8_c(const int32_t *input, uint16_t *output,
|
||||
int stride, int tx_type, int bd) {
|
||||
int stride, TX_TYPE tx_type, int bd) {
|
||||
int txfm_buf[4 * 8 + 8 + 8];
|
||||
inv_txfm2d_add_facade(input, output, stride, txfm_buf, tx_type, TX_4X8, bd);
|
||||
}
|
||||
|
||||
void av1_inv_txfm2d_add_8x4_c(const int32_t *input, uint16_t *output,
|
||||
int stride, int tx_type, int bd) {
|
||||
int stride, TX_TYPE tx_type, int bd) {
|
||||
#if CONFIG_TXMG
|
||||
int txfm_buf[8 * 4 + 8 + 8];
|
||||
int32_t rinput[8 * 4];
|
||||
uint16_t routput[8 * 4];
|
||||
TX_SIZE tx_size = TX_8X4;
|
||||
TX_SIZE rtx_size = av1_rotate_tx_size(tx_size);
|
||||
TX_TYPE rtx_type = av1_rotate_tx_type(tx_type);
|
||||
int w = tx_size_wide[tx_size];
|
||||
int h = tx_size_high[tx_size];
|
||||
int rw = h;
|
||||
int rh = w;
|
||||
transpose_int32(rinput, rw, input, w, w, h);
|
||||
transpose_uint16(routput, rw, output, stride, w, h);
|
||||
inv_txfm2d_add_facade(rinput, routput, rw, txfm_buf, rtx_type, rtx_size, bd);
|
||||
transpose_uint16(output, stride, routput, rw, rw, rh);
|
||||
#else
|
||||
int txfm_buf[8 * 4 + 4 + 4];
|
||||
inv_txfm2d_add_facade(input, output, stride, txfm_buf, tx_type, TX_8X4, bd);
|
||||
#endif
|
||||
}
|
||||
|
||||
void av1_inv_txfm2d_add_8x16_c(const int32_t *input, uint16_t *output,
|
||||
int stride, int tx_type, int bd) {
|
||||
int stride, TX_TYPE tx_type, int bd) {
|
||||
int txfm_buf[8 * 16 + 16 + 16];
|
||||
inv_txfm2d_add_facade(input, output, stride, txfm_buf, tx_type, TX_8X16, bd);
|
||||
}
|
||||
|
||||
void av1_inv_txfm2d_add_16x8_c(const int32_t *input, uint16_t *output,
|
||||
int stride, int tx_type, int bd) {
|
||||
int stride, TX_TYPE tx_type, int bd) {
|
||||
#if CONFIG_TXMG
|
||||
int txfm_buf[16 * 8 + 16 + 16];
|
||||
int32_t rinput[16 * 8];
|
||||
uint16_t routput[16 * 8];
|
||||
TX_SIZE tx_size = TX_16X8;
|
||||
TX_SIZE rtx_size = av1_rotate_tx_size(tx_size);
|
||||
TX_TYPE rtx_type = av1_rotate_tx_type(tx_type);
|
||||
int w = tx_size_wide[tx_size];
|
||||
int h = tx_size_high[tx_size];
|
||||
int rw = h;
|
||||
int rh = w;
|
||||
transpose_int32(rinput, rw, input, w, w, h);
|
||||
transpose_uint16(routput, rw, output, stride, w, h);
|
||||
inv_txfm2d_add_facade(rinput, routput, rw, txfm_buf, rtx_type, rtx_size, bd);
|
||||
transpose_uint16(output, stride, routput, rw, rw, rh);
|
||||
#else
|
||||
int txfm_buf[16 * 8 + 8 + 8];
|
||||
inv_txfm2d_add_facade(input, output, stride, txfm_buf, tx_type, TX_16X8, bd);
|
||||
#endif
|
||||
}
|
||||
|
||||
void av1_inv_txfm2d_add_16x32_c(const int32_t *input, uint16_t *output,
|
||||
int stride, int tx_type, int bd) {
|
||||
int stride, TX_TYPE tx_type, int bd) {
|
||||
int txfm_buf[16 * 32 + 32 + 32];
|
||||
inv_txfm2d_add_facade(input, output, stride, txfm_buf, tx_type, TX_16X32, bd);
|
||||
}
|
||||
|
||||
void av1_inv_txfm2d_add_32x16_c(const int32_t *input, uint16_t *output,
|
||||
int stride, int tx_type, int bd) {
|
||||
int stride, TX_TYPE tx_type, int bd) {
|
||||
#if CONFIG_TXMG
|
||||
int txfm_buf[32 * 16 + 32 + 32];
|
||||
int32_t rinput[32 * 16];
|
||||
uint16_t routput[32 * 16];
|
||||
TX_SIZE tx_size = TX_32X16;
|
||||
TX_SIZE rtx_size = av1_rotate_tx_size(tx_size);
|
||||
TX_TYPE rtx_type = av1_rotate_tx_type(tx_type);
|
||||
int w = tx_size_wide[tx_size];
|
||||
int h = tx_size_high[tx_size];
|
||||
int rw = h;
|
||||
int rh = w;
|
||||
transpose_int32(rinput, rw, input, w, w, h);
|
||||
transpose_uint16(routput, rw, output, stride, w, h);
|
||||
inv_txfm2d_add_facade(rinput, routput, rw, txfm_buf, rtx_type, rtx_size, bd);
|
||||
transpose_uint16(output, stride, routput, rw, rw, rh);
|
||||
#else
|
||||
int txfm_buf[32 * 16 + 16 + 16];
|
||||
inv_txfm2d_add_facade(input, output, stride, txfm_buf, tx_type, TX_32X16, bd);
|
||||
#endif
|
||||
}
|
||||
|
||||
void av1_inv_txfm2d_add_4x4_c(const int32_t *input, uint16_t *output,
|
||||
int stride, int tx_type, int bd) {
|
||||
int stride, TX_TYPE tx_type, int bd) {
|
||||
int txfm_buf[4 * 4 + 4 + 4];
|
||||
inv_txfm2d_add_facade(input, output, stride, txfm_buf, tx_type, TX_4X4, bd);
|
||||
}
|
||||
|
||||
void av1_inv_txfm2d_add_8x8_c(const int32_t *input, uint16_t *output,
|
||||
int stride, int tx_type, int bd) {
|
||||
int stride, TX_TYPE tx_type, int bd) {
|
||||
int txfm_buf[8 * 8 + 8 + 8];
|
||||
inv_txfm2d_add_facade(input, output, stride, txfm_buf, tx_type, TX_8X8, bd);
|
||||
}
|
||||
|
||||
void av1_inv_txfm2d_add_16x16_c(const int32_t *input, uint16_t *output,
|
||||
int stride, int tx_type, int bd) {
|
||||
int stride, TX_TYPE tx_type, int bd) {
|
||||
int txfm_buf[16 * 16 + 16 + 16];
|
||||
inv_txfm2d_add_facade(input, output, stride, txfm_buf, tx_type, TX_16X16, bd);
|
||||
}
|
||||
|
||||
void av1_inv_txfm2d_add_32x32_c(const int32_t *input, uint16_t *output,
|
||||
int stride, int tx_type, int bd) {
|
||||
int stride, TX_TYPE tx_type, int bd) {
|
||||
int txfm_buf[32 * 32 + 32 + 32];
|
||||
inv_txfm2d_add_facade(input, output, stride, txfm_buf, tx_type, TX_32X32, bd);
|
||||
}
|
||||
|
||||
#if CONFIG_TX64X64
|
||||
void av1_inv_txfm2d_add_64x64_c(const int32_t *input, uint16_t *output,
|
||||
int stride, int tx_type, int bd) {
|
||||
int stride, TX_TYPE tx_type, int bd) {
|
||||
int txfm_buf[64 * 64 + 64 + 64];
|
||||
TXFM_2D_FLIP_CFG cfg = av1_get_inv_txfm_64x64_cfg(tx_type);
|
||||
inv_txfm2d_add_c(input, output, stride, &cfg, txfm_buf, bd);
|
||||
inv_txfm2d_add_facade(input, output, stride, txfm_buf, tx_type, TX_64X64, bd);
|
||||
}
|
||||
|
||||
void av1_inv_txfm2d_add_64x32_c(const int32_t *input, uint16_t *output,
|
||||
int stride, TX_TYPE tx_type, int bd) {
|
||||
#if CONFIG_TXMG
|
||||
int txfm_buf[64 * 32 + 64 + 64];
|
||||
int32_t rinput[64 * 32];
|
||||
uint16_t routput[64 * 32];
|
||||
TX_SIZE tx_size = TX_64X32;
|
||||
TX_SIZE rtx_size = av1_rotate_tx_size(tx_size);
|
||||
TX_TYPE rtx_type = av1_rotate_tx_type(tx_type);
|
||||
int w = tx_size_wide[tx_size];
|
||||
int h = tx_size_high[tx_size];
|
||||
int rw = h;
|
||||
int rh = w;
|
||||
transpose_int32(rinput, rw, input, w, w, h);
|
||||
transpose_uint16(routput, rw, output, stride, w, h);
|
||||
inv_txfm2d_add_facade(rinput, routput, rw, txfm_buf, rtx_type, rtx_size, bd);
|
||||
transpose_uint16(output, stride, routput, rw, rw, rh);
|
||||
#else
|
||||
int txfm_buf[64 * 32 + 64 + 64];
|
||||
inv_txfm2d_add_facade(input, output, stride, txfm_buf, tx_type, TX_64X32, bd);
|
||||
#endif
|
||||
}
|
||||
|
||||
void av1_inv_txfm2d_add_32x64_c(const int32_t *input, uint16_t *output,
|
||||
int stride, TX_TYPE tx_type, int bd) {
|
||||
int txfm_buf[64 * 32 + 64 + 64];
|
||||
inv_txfm2d_add_facade(input, output, stride, txfm_buf, tx_type, TX_32X64, bd);
|
||||
}
|
||||
#endif // CONFIG_TX64X64
|
||||
|
|
|
|||
1058
third_party/aom/av1/common/av1_loopfilter.c
vendored
1058
third_party/aom/av1/common/av1_loopfilter.c
vendored
File diff suppressed because it is too large
Load diff
59
third_party/aom/av1/common/av1_loopfilter.h
vendored
59
third_party/aom/av1/common/av1_loopfilter.h
vendored
|
|
@ -36,10 +36,12 @@ enum lf_path {
|
|||
};
|
||||
|
||||
struct loopfilter {
|
||||
int filter_level;
|
||||
#if CONFIG_UV_LVL
|
||||
#if CONFIG_LOOPFILTER_LEVEL
|
||||
int filter_level[2];
|
||||
int filter_level_u;
|
||||
int filter_level_v;
|
||||
#else
|
||||
int filter_level;
|
||||
#endif
|
||||
|
||||
int sharpness_level;
|
||||
|
|
@ -49,14 +51,13 @@ struct loopfilter {
|
|||
uint8_t mode_ref_delta_update;
|
||||
|
||||
// 0 = Intra, Last, Last2+Last3(CONFIG_EXT_REFS),
|
||||
// GF, BRF(CONFIG_EXT_REFS),
|
||||
// ARF2(CONFIG_EXT_REFS+CONFIG_ALTREF2), ARF
|
||||
signed char ref_deltas[TOTAL_REFS_PER_FRAME];
|
||||
signed char last_ref_deltas[TOTAL_REFS_PER_FRAME];
|
||||
// GF, BRF(CONFIG_EXT_REFS), ARF2(CONFIG_EXT_REFS), ARF
|
||||
int8_t ref_deltas[TOTAL_REFS_PER_FRAME];
|
||||
int8_t last_ref_deltas[TOTAL_REFS_PER_FRAME];
|
||||
|
||||
// 0 = ZERO_MV, MV
|
||||
signed char mode_deltas[MAX_MODE_LF_DELTAS];
|
||||
signed char last_mode_deltas[MAX_MODE_LF_DELTAS];
|
||||
int8_t mode_deltas[MAX_MODE_LF_DELTAS];
|
||||
int8_t last_mode_deltas[MAX_MODE_LF_DELTAS];
|
||||
};
|
||||
|
||||
// Need to align this structure so when it is declared and
|
||||
|
|
@ -69,7 +70,11 @@ typedef struct {
|
|||
|
||||
typedef struct {
|
||||
loop_filter_thresh lfthr[MAX_LOOP_FILTER + 1];
|
||||
#if CONFIG_LOOPFILTER_LEVEL
|
||||
uint8_t lvl[MAX_SEGMENTS][2][TOTAL_REFS_PER_FRAME][MAX_MODE_LF_DELTAS];
|
||||
#else
|
||||
uint8_t lvl[MAX_SEGMENTS][TOTAL_REFS_PER_FRAME][MAX_MODE_LF_DELTAS];
|
||||
#endif
|
||||
} loop_filter_info_n;
|
||||
|
||||
// This structure holds bit masks for all 8x8 blocks in a 64x64 region.
|
||||
|
|
@ -132,17 +137,42 @@ void av1_loop_filter_init(struct AV1Common *cm);
|
|||
// This should be called before av1_loop_filter_rows(),
|
||||
// av1_loop_filter_frame()
|
||||
// calls this function directly.
|
||||
void av1_loop_filter_frame_init(struct AV1Common *cm, int default_filt_lvl);
|
||||
void av1_loop_filter_frame_init(struct AV1Common *cm, int default_filt_lvl,
|
||||
int default_filt_lvl_r
|
||||
#if CONFIG_LOOPFILTER_LEVEL
|
||||
,
|
||||
int plane
|
||||
#endif
|
||||
);
|
||||
|
||||
#if CONFIG_LPF_SB
|
||||
void av1_loop_filter_frame(YV12_BUFFER_CONFIG *frame, struct AV1Common *cm,
|
||||
struct macroblockd *mbd, int filter_level,
|
||||
int y_only, int partial_frame, int mi_row,
|
||||
int mi_col);
|
||||
|
||||
// Apply the loop filter to [start, stop) macro block rows in frame_buffer.
|
||||
void av1_loop_filter_rows(YV12_BUFFER_CONFIG *frame_buffer,
|
||||
struct AV1Common *cm,
|
||||
struct macroblockd_plane *planes, int start, int stop,
|
||||
int col_start, int col_end, int y_only);
|
||||
|
||||
void av1_loop_filter_sb_level_init(struct AV1Common *cm, int mi_row, int mi_col,
|
||||
int lvl);
|
||||
#else
|
||||
void av1_loop_filter_frame(YV12_BUFFER_CONFIG *frame, struct AV1Common *cm,
|
||||
struct macroblockd *mbd, int filter_level,
|
||||
#if CONFIG_LOOPFILTER_LEVEL
|
||||
int filter_level_r,
|
||||
#endif
|
||||
int y_only, int partial_frame);
|
||||
|
||||
// Apply the loop filter to [start, stop) macro block rows in frame_buffer.
|
||||
void av1_loop_filter_rows(YV12_BUFFER_CONFIG *frame_buffer,
|
||||
struct AV1Common *cm,
|
||||
struct macroblockd_plane planes[MAX_MB_PLANE],
|
||||
int start, int stop, int y_only);
|
||||
struct macroblockd_plane *planes, int start, int stop,
|
||||
int y_only);
|
||||
#endif // CONFIG_LPF_SB
|
||||
|
||||
typedef struct LoopFilterWorkerData {
|
||||
YV12_BUFFER_CONFIG *frame_buffer;
|
||||
|
|
@ -154,9 +184,10 @@ typedef struct LoopFilterWorkerData {
|
|||
int y_only;
|
||||
} LFWorkerData;
|
||||
|
||||
void av1_loop_filter_data_reset(
|
||||
LFWorkerData *lf_data, YV12_BUFFER_CONFIG *frame_buffer,
|
||||
struct AV1Common *cm, const struct macroblockd_plane planes[MAX_MB_PLANE]);
|
||||
void av1_loop_filter_data_reset(LFWorkerData *lf_data,
|
||||
YV12_BUFFER_CONFIG *frame_buffer,
|
||||
struct AV1Common *cm,
|
||||
const struct macroblockd_plane *planes);
|
||||
|
||||
// Operates on the rows described by 'lf_data'.
|
||||
int av1_loop_filter_worker(LFWorkerData *const lf_data, void *unused);
|
||||
|
|
|
|||
637
third_party/aom/av1/common/av1_rtcd_defs.pl
vendored
637
third_party/aom/av1/common/av1_rtcd_defs.pl
vendored
|
|
@ -24,7 +24,6 @@ struct search_site_config;
|
|||
struct mv;
|
||||
union int_mv;
|
||||
struct yv12_buffer_config;
|
||||
typedef uint16_t od_dering_in;
|
||||
EOF
|
||||
}
|
||||
forward_decls qw/av1_common_forward_decls/;
|
||||
|
|
@ -64,86 +63,94 @@ if (aom_config("CONFIG_HIGHBITDEPTH") eq "yes") {
|
|||
# Inverse dct
|
||||
#
|
||||
if (aom_config("CONFIG_HIGHBITDEPTH") eq "yes") {
|
||||
{
|
||||
add_proto qw/void av1_iht4x4_16_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, const struct txfm_param *param";
|
||||
specialize qw/av1_iht4x4_16_add sse2/;
|
||||
add_proto qw/void av1_iht4x4_16_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, const struct txfm_param *param";
|
||||
specialize qw/av1_iht4x4_16_add sse2/;
|
||||
|
||||
add_proto qw/void av1_iht4x8_32_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, const struct txfm_param *param";
|
||||
specialize qw/av1_iht4x8_32_add sse2/;
|
||||
add_proto qw/void av1_iht4x8_32_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, const struct txfm_param *param";
|
||||
specialize qw/av1_iht4x8_32_add sse2/;
|
||||
|
||||
add_proto qw/void av1_iht8x4_32_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, const struct txfm_param *param";
|
||||
specialize qw/av1_iht8x4_32_add sse2/;
|
||||
add_proto qw/void av1_iht8x4_32_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, const struct txfm_param *param";
|
||||
specialize qw/av1_iht8x4_32_add sse2/;
|
||||
|
||||
add_proto qw/void av1_iht8x16_128_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, const struct txfm_param *param";
|
||||
specialize qw/av1_iht8x16_128_add sse2/;
|
||||
add_proto qw/void av1_iht8x16_128_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, const struct txfm_param *param";
|
||||
specialize qw/av1_iht8x16_128_add sse2/;
|
||||
|
||||
add_proto qw/void av1_iht16x8_128_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, const struct txfm_param *param";
|
||||
specialize qw/av1_iht16x8_128_add sse2/;
|
||||
add_proto qw/void av1_iht16x8_128_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, const struct txfm_param *param";
|
||||
specialize qw/av1_iht16x8_128_add sse2/;
|
||||
|
||||
add_proto qw/void av1_iht16x32_512_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, const struct txfm_param *param";
|
||||
specialize qw/av1_iht16x32_512_add sse2/;
|
||||
add_proto qw/void av1_iht16x32_512_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, const struct txfm_param *param";
|
||||
specialize qw/av1_iht16x32_512_add sse2/;
|
||||
|
||||
add_proto qw/void av1_iht32x16_512_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, const struct txfm_param *param";
|
||||
specialize qw/av1_iht32x16_512_add sse2/;
|
||||
add_proto qw/void av1_iht32x16_512_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, const struct txfm_param *param";
|
||||
specialize qw/av1_iht32x16_512_add sse2/;
|
||||
|
||||
add_proto qw/void av1_iht4x16_64_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, const struct txfm_param *param";
|
||||
add_proto qw/void av1_iht4x16_64_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, const struct txfm_param *param";
|
||||
|
||||
add_proto qw/void av1_iht16x4_64_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, const struct txfm_param *param";
|
||||
add_proto qw/void av1_iht16x4_64_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, const struct txfm_param *param";
|
||||
|
||||
add_proto qw/void av1_iht8x32_256_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, const struct txfm_param *param";
|
||||
add_proto qw/void av1_iht8x32_256_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, const struct txfm_param *param";
|
||||
|
||||
add_proto qw/void av1_iht32x8_256_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, const struct txfm_param *param";
|
||||
add_proto qw/void av1_iht32x8_256_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, const struct txfm_param *param";
|
||||
|
||||
add_proto qw/void av1_iht8x8_64_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, const struct txfm_param *param";
|
||||
specialize qw/av1_iht8x8_64_add sse2/;
|
||||
add_proto qw/void av1_iht8x8_64_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, const struct txfm_param *param";
|
||||
specialize qw/av1_iht8x8_64_add sse2/;
|
||||
|
||||
add_proto qw/void av1_iht16x16_256_add/, "const tran_low_t *input, uint8_t *output, int pitch, const struct txfm_param *param";
|
||||
specialize qw/av1_iht16x16_256_add sse2 avx2/;
|
||||
add_proto qw/void av1_iht16x16_256_add/, "const tran_low_t *input, uint8_t *output, int pitch, const struct txfm_param *param";
|
||||
specialize qw/av1_iht16x16_256_add sse2 avx2/;
|
||||
|
||||
add_proto qw/void av1_iht32x32_1024_add/, "const tran_low_t *input, uint8_t *output, int pitch, const struct txfm_param *param";
|
||||
}
|
||||
add_proto qw/void av1_iht32x32_1024_add/, "const tran_low_t *input, uint8_t *output, int pitch, const struct txfm_param *param";
|
||||
} else {
|
||||
{
|
||||
add_proto qw/void av1_iht4x4_16_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, const struct txfm_param *param";
|
||||
specialize qw/av1_iht4x4_16_add sse2 neon dspr2/;
|
||||
add_proto qw/void av1_iht4x4_16_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, const struct txfm_param *param";
|
||||
if (aom_config("CONFIG_DAALA_DCT4") ne "yes") {
|
||||
specialize qw/av1_iht4x4_16_add sse2 neon/;
|
||||
}
|
||||
|
||||
add_proto qw/void av1_iht4x8_32_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, const struct txfm_param *param";
|
||||
specialize qw/av1_iht4x8_32_add sse2/;
|
||||
add_proto qw/void av1_iht4x8_32_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, const struct txfm_param *param";
|
||||
specialize qw/av1_iht4x8_32_add sse2/;
|
||||
|
||||
add_proto qw/void av1_iht8x4_32_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, const struct txfm_param *param";
|
||||
specialize qw/av1_iht8x4_32_add sse2/;
|
||||
add_proto qw/void av1_iht8x4_32_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, const struct txfm_param *param";
|
||||
specialize qw/av1_iht8x4_32_add sse2/;
|
||||
|
||||
add_proto qw/void av1_iht8x16_128_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, const struct txfm_param *param";
|
||||
specialize qw/av1_iht8x16_128_add sse2/;
|
||||
add_proto qw/void av1_iht8x16_128_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, const struct txfm_param *param";
|
||||
specialize qw/av1_iht8x16_128_add sse2/;
|
||||
|
||||
add_proto qw/void av1_iht16x8_128_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, const struct txfm_param *param";
|
||||
specialize qw/av1_iht16x8_128_add sse2/;
|
||||
add_proto qw/void av1_iht16x8_128_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, const struct txfm_param *param";
|
||||
specialize qw/av1_iht16x8_128_add sse2/;
|
||||
|
||||
add_proto qw/void av1_iht16x32_512_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, const struct txfm_param *param";
|
||||
specialize qw/av1_iht16x32_512_add sse2/;
|
||||
add_proto qw/void av1_iht16x32_512_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, const struct txfm_param *param";
|
||||
specialize qw/av1_iht16x32_512_add sse2/;
|
||||
|
||||
add_proto qw/void av1_iht32x16_512_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, const struct txfm_param *param";
|
||||
specialize qw/av1_iht32x16_512_add sse2/;
|
||||
add_proto qw/void av1_iht32x16_512_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, const struct txfm_param *param";
|
||||
specialize qw/av1_iht32x16_512_add sse2/;
|
||||
|
||||
add_proto qw/void av1_iht4x16_64_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, const struct txfm_param *param";
|
||||
add_proto qw/void av1_iht4x16_64_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, const struct txfm_param *param";
|
||||
|
||||
add_proto qw/void av1_iht16x4_64_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, const struct txfm_param *param";
|
||||
add_proto qw/void av1_iht16x4_64_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, const struct txfm_param *param";
|
||||
|
||||
add_proto qw/void av1_iht8x32_256_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, const struct txfm_param *param";
|
||||
add_proto qw/void av1_iht8x32_256_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, const struct txfm_param *param";
|
||||
|
||||
add_proto qw/void av1_iht32x8_256_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, const struct txfm_param *param";
|
||||
add_proto qw/void av1_iht32x8_256_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, const struct txfm_param *param";
|
||||
|
||||
add_proto qw/void av1_iht8x8_64_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, const struct txfm_param *param";
|
||||
specialize qw/av1_iht8x8_64_add sse2 neon dspr2/;
|
||||
add_proto qw/void av1_iht8x8_64_add/, "const tran_low_t *input, uint8_t *dest, int dest_stride, const struct txfm_param *param";
|
||||
if (aom_config("CONFIG_DAALA_DCT8") ne "yes") {
|
||||
specialize qw/av1_iht8x8_64_add sse2 neon/;
|
||||
}
|
||||
|
||||
add_proto qw/void av1_iht16x16_256_add/, "const tran_low_t *input, uint8_t *output, int pitch, const struct txfm_param *param";
|
||||
specialize qw/av1_iht16x16_256_add sse2 avx2 dspr2/;
|
||||
add_proto qw/void av1_iht16x16_256_add/, "const tran_low_t *input, uint8_t *output, int pitch, const struct txfm_param *param";
|
||||
if (aom_config("CONFIG_DAALA_DCT16") ne "yes") {
|
||||
specialize qw/av1_iht16x16_256_add sse2 avx2/;
|
||||
}
|
||||
|
||||
add_proto qw/void av1_iht32x32_1024_add/, "const tran_low_t *input, uint8_t *output, int pitch, const struct txfm_param *param";
|
||||
add_proto qw/void av1_iht32x32_1024_add/, "const tran_low_t *input, uint8_t *output, int pitch, const struct txfm_param *param";
|
||||
|
||||
if (aom_config("CONFIG_EXT_TX") ne "yes") {
|
||||
if (aom_config("CONFIG_EXT_TX") ne "yes") {
|
||||
if (aom_config("CONFIG_DAALA_DCT4") ne "yes") {
|
||||
specialize qw/av1_iht4x4_16_add msa/;
|
||||
}
|
||||
if (aom_config("CONFIG_DAALA_DCT8") ne "yes") {
|
||||
specialize qw/av1_iht8x8_64_add msa/;
|
||||
}
|
||||
if (aom_config("CONFIG_DAALA_DCT16") ne "yes") {
|
||||
specialize qw/av1_iht16x16_256_add msa/;
|
||||
}
|
||||
}
|
||||
|
|
@ -153,6 +160,8 @@ add_proto qw/void av1_iht32x32_1024_add/, "const tran_low_t *input, uint8_t *out
|
|||
|
||||
if (aom_config("CONFIG_TX64X64") eq "yes") {
|
||||
add_proto qw/void av1_iht64x64_4096_add/, "const tran_low_t *input, uint8_t *output, int pitch, const struct txfm_param *param";
|
||||
add_proto qw/void av1_iht32x64_2048_add/, "const tran_low_t *input, uint8_t *output, int pitch, const struct txfm_param *param";
|
||||
add_proto qw/void av1_iht64x32_2048_add/, "const tran_low_t *input, uint8_t *output, int pitch, const struct txfm_param *param";
|
||||
}
|
||||
|
||||
if (aom_config("CONFIG_NEW_QUANT") eq "yes") {
|
||||
|
|
@ -256,63 +265,41 @@ if (aom_config("CONFIG_HIGHBITDEPTH") eq "yes") {
|
|||
}
|
||||
|
||||
#inv txfm
|
||||
add_proto qw/void av1_inv_txfm2d_add_4x8/, "const int32_t *input, uint16_t *output, int stride, int tx_type, int bd";
|
||||
add_proto qw/void av1_inv_txfm2d_add_8x4/, "const int32_t *input, uint16_t *output, int stride, int tx_type, int bd";
|
||||
add_proto qw/void av1_inv_txfm2d_add_8x16/, "const int32_t *input, uint16_t *output, int stride, int tx_type, int bd";
|
||||
add_proto qw/void av1_inv_txfm2d_add_16x8/, "const int32_t *input, uint16_t *output, int stride, int tx_type, int bd";
|
||||
add_proto qw/void av1_inv_txfm2d_add_16x32/, "const int32_t *input, uint16_t *output, int stride, int tx_type, int bd";
|
||||
add_proto qw/void av1_inv_txfm2d_add_32x16/, "const int32_t *input, uint16_t *output, int stride, int tx_type, int bd";
|
||||
add_proto qw/void av1_inv_txfm2d_add_4x4/, "const int32_t *input, uint16_t *output, int stride, int tx_type, int bd";
|
||||
specialize qw/av1_inv_txfm2d_add_4x4 sse4_1/;
|
||||
add_proto qw/void av1_inv_txfm2d_add_8x8/, "const int32_t *input, uint16_t *output, int stride, int tx_type, int bd";
|
||||
specialize qw/av1_inv_txfm2d_add_8x8 sse4_1/;
|
||||
add_proto qw/void av1_inv_txfm2d_add_16x16/, "const int32_t *input, uint16_t *output, int stride, int tx_type, int bd";
|
||||
specialize qw/av1_inv_txfm2d_add_16x16 sse4_1/;
|
||||
add_proto qw/void av1_inv_txfm2d_add_32x32/, "const int32_t *input, uint16_t *output, int stride, int tx_type, int bd";
|
||||
specialize qw/av1_inv_txfm2d_add_32x32 avx2/;
|
||||
add_proto qw/void av1_inv_txfm2d_add_64x64/, "const int32_t *input, uint16_t *output, int stride, int tx_type, int bd";
|
||||
add_proto qw/void av1_inv_txfm2d_add_4x8/, "const int32_t *input, uint16_t *output, int stride, TX_TYPE tx_type, int bd";
|
||||
add_proto qw/void av1_inv_txfm2d_add_8x4/, "const int32_t *input, uint16_t *output, int stride, TX_TYPE tx_type, int bd";
|
||||
add_proto qw/void av1_inv_txfm2d_add_8x16/, "const int32_t *input, uint16_t *output, int stride, TX_TYPE tx_type, int bd";
|
||||
add_proto qw/void av1_inv_txfm2d_add_16x8/, "const int32_t *input, uint16_t *output, int stride, TX_TYPE tx_type, int bd";
|
||||
add_proto qw/void av1_inv_txfm2d_add_16x32/, "const int32_t *input, uint16_t *output, int stride, TX_TYPE tx_type, int bd";
|
||||
add_proto qw/void av1_inv_txfm2d_add_32x16/, "const int32_t *input, uint16_t *output, int stride, TX_TYPE tx_type, int bd";
|
||||
add_proto qw/void av1_inv_txfm2d_add_4x4/, "const int32_t *input, uint16_t *output, int stride, TX_TYPE tx_type, int bd";
|
||||
if (aom_config("CONFIG_DAALA_DCT4") ne "yes") {
|
||||
specialize qw/av1_inv_txfm2d_add_4x4 sse4_1/;
|
||||
}
|
||||
add_proto qw/void av1_inv_txfm2d_add_8x8/, "const int32_t *input, uint16_t *output, int stride, TX_TYPE tx_type, int bd";
|
||||
if (aom_config("CONFIG_DAALA_DCT8") ne "yes") {
|
||||
specialize qw/av1_inv_txfm2d_add_8x8 sse4_1/;
|
||||
}
|
||||
add_proto qw/void av1_inv_txfm2d_add_16x16/, "const int32_t *input, uint16_t *output, int stride, TX_TYPE tx_type, int bd";
|
||||
if (aom_config("CONFIG_DAALA_DCT16") ne "yes") {
|
||||
specialize qw/av1_inv_txfm2d_add_16x16 sse4_1/;
|
||||
}
|
||||
add_proto qw/void av1_inv_txfm2d_add_32x32/, "const int32_t *input, uint16_t *output, int stride, TX_TYPE tx_type, int bd";
|
||||
if (aom_config("CONFIG_DAALA_DCT32") ne "yes") {
|
||||
specialize qw/av1_inv_txfm2d_add_32x32 avx2/;
|
||||
}
|
||||
if (aom_config("CONFIG_TX64X64") eq "yes") {
|
||||
add_proto qw/void av1_inv_txfm2d_add_64x64/, "const int32_t *input, uint16_t *output, int stride, TX_TYPE tx_type, int bd";
|
||||
add_proto qw/void av1_inv_txfm2d_add_64x32/, "const int32_t *input, uint16_t *output, int stride, TX_TYPE tx_type, int bd";
|
||||
add_proto qw/void av1_inv_txfm2d_add_32x64/, "const int32_t *input, uint16_t *output, int stride, TX_TYPE tx_type, int bd";
|
||||
}
|
||||
|
||||
#
|
||||
# Encoder functions below this point.
|
||||
#
|
||||
if (aom_config("CONFIG_AV1_ENCODER") eq "yes") {
|
||||
|
||||
# ENCODEMB INVOKE
|
||||
# ENCODEMB INVOKE
|
||||
|
||||
if (aom_config("CONFIG_AOM_QM") eq "yes") {
|
||||
if (aom_config("CONFIG_HIGHBITDEPTH") eq "yes") {
|
||||
# the transform coefficients are held in 32-bit
|
||||
# values, so the assembler code for av1_block_error can no longer be used.
|
||||
add_proto qw/int64_t av1_block_error/, "const tran_low_t *coeff, const tran_low_t *dqcoeff, intptr_t block_size, int64_t *ssz";
|
||||
specialize qw/av1_block_error avx2/;
|
||||
|
||||
add_proto qw/void av1_quantize_fp/, "const tran_low_t *coeff_ptr, intptr_t n_coeffs, int skip_block, const int16_t *zbin_ptr, const int16_t *round_ptr, const int16_t *quant_ptr, const int16_t *quant_shift_ptr, tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr, const int16_t *dequant_ptr, uint16_t *eob_ptr, const int16_t *scan, const int16_t *iscan, const qm_val_t * qm_ptr, const qm_val_t *iqm_ptr";
|
||||
|
||||
add_proto qw/void av1_quantize_fp_32x32/, "const tran_low_t *coeff_ptr, intptr_t n_coeffs, int skip_block, const int16_t *zbin_ptr, const int16_t *round_ptr, const int16_t *quant_ptr, const int16_t *quant_shift_ptr, tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr, const int16_t *dequant_ptr, uint16_t *eob_ptr, const int16_t *scan, const int16_t *iscan, const qm_val_t * qm_ptr, const qm_val_t *iqm_ptr";
|
||||
|
||||
if (aom_config("CONFIG_TX64X64") eq "yes") {
|
||||
add_proto qw/void av1_quantize_fp_64x64/, "const tran_low_t *coeff_ptr, intptr_t n_coeffs, int skip_block, const int16_t *zbin_ptr, const int16_t *round_ptr, const int16_t *quant_ptr, const int16_t *quant_shift_ptr, tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr, const int16_t *dequant_ptr, uint16_t *eob_ptr, const int16_t *scan, const int16_t *iscan, const qm_val_t * qm_ptr, const qm_val_t *iqm_ptr";
|
||||
}
|
||||
|
||||
add_proto qw/void av1_fdct8x8_quant/, "const int16_t *input, int stride, tran_low_t *coeff_ptr, intptr_t n_coeffs, int skip_block, const int16_t *zbin_ptr, const int16_t *round_ptr, const int16_t *quant_ptr, const int16_t *quant_shift_ptr, tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr, const int16_t *dequant_ptr, uint16_t *eob_ptr, const int16_t *scan, const int16_t *iscan, const qm_val_t * qm_ptr, const qm_val_t *iqm_ptr";
|
||||
} else {
|
||||
add_proto qw/int64_t av1_block_error/, "const tran_low_t *coeff, const tran_low_t *dqcoeff, intptr_t block_size, int64_t *ssz";
|
||||
specialize qw/av1_block_error avx2 msa/, "$sse2_x86inc";
|
||||
|
||||
add_proto qw/int64_t av1_block_error_fp/, "const int16_t *coeff, const int16_t *dqcoeff, int block_size";
|
||||
specialize qw/av1_block_error_fp neon/, "$sse2_x86inc";
|
||||
|
||||
add_proto qw/void av1_quantize_fp/, "const tran_low_t *coeff_ptr, intptr_t n_coeffs, int skip_block, const int16_t *zbin_ptr, const int16_t *round_ptr, const int16_t *quant_ptr, const int16_t *quant_shift_ptr, tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr, const int16_t *dequant_ptr, uint16_t *eob_ptr, const int16_t *scan, const int16_t *iscan, const qm_val_t * qm_ptr, const qm_val_t *iqm_ptr";
|
||||
|
||||
add_proto qw/void av1_quantize_fp_32x32/, "const tran_low_t *coeff_ptr, intptr_t n_coeffs, int skip_block, const int16_t *zbin_ptr, const int16_t *round_ptr, const int16_t *quant_ptr, const int16_t *quant_shift_ptr, tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr, const int16_t *dequant_ptr, uint16_t *eob_ptr, const int16_t *scan, const int16_t *iscan, const qm_val_t * qm_ptr, const qm_val_t *iqm_ptr";
|
||||
|
||||
if (aom_config("CONFIG_TX64X64") eq "yes") {
|
||||
add_proto qw/void av1_quantize_fp_64x64/, "const tran_low_t *coeff_ptr, intptr_t n_coeffs, int skip_block, const int16_t *zbin_ptr, const int16_t *round_ptr, const int16_t *quant_ptr, const int16_t *quant_shift_ptr, tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr, const int16_t *dequant_ptr, uint16_t *eob_ptr, const int16_t *scan, const int16_t *iscan, const qm_val_t * qm_ptr, const qm_val_t *iqm_ptr";
|
||||
}
|
||||
|
||||
add_proto qw/void av1_fdct8x8_quant/, "const int16_t *input, int stride, tran_low_t *coeff_ptr, intptr_t n_coeffs, int skip_block, const int16_t *zbin_ptr, const int16_t *round_ptr, const int16_t *quant_ptr, const int16_t *quant_shift_ptr, tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr, const int16_t *dequant_ptr, uint16_t *eob_ptr, const int16_t *scan, const int16_t *iscan, const qm_val_t * qm_ptr, const qm_val_t *iqm_ptr";
|
||||
}
|
||||
} else {
|
||||
if (aom_config("CONFIG_HIGHBITDEPTH") eq "yes") {
|
||||
# the transform coefficients are held in 32-bit
|
||||
# values, so the assembler code for av1_block_error can no longer be used.
|
||||
|
|
@ -328,8 +315,6 @@ if (aom_config("CONFIG_AOM_QM") eq "yes") {
|
|||
if (aom_config("CONFIG_TX64X64") eq "yes") {
|
||||
add_proto qw/void av1_quantize_fp_64x64/, "const tran_low_t *coeff_ptr, intptr_t n_coeffs, int skip_block, const int16_t *zbin_ptr, const int16_t *round_ptr, const int16_t *quant_ptr, const int16_t *quant_shift_ptr, tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr, const int16_t *dequant_ptr, uint16_t *eob_ptr, const int16_t *scan, const int16_t *iscan";
|
||||
}
|
||||
|
||||
add_proto qw/void av1_fdct8x8_quant/, "const int16_t *input, int stride, tran_low_t *coeff_ptr, intptr_t n_coeffs, int skip_block, const int16_t *zbin_ptr, const int16_t *round_ptr, const int16_t *quant_ptr, const int16_t *quant_shift_ptr, tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr, const int16_t *dequant_ptr, uint16_t *eob_ptr, const int16_t *scan, const int16_t *iscan";
|
||||
} else {
|
||||
add_proto qw/int64_t av1_block_error/, "const tran_low_t *coeff, const tran_low_t *dqcoeff, intptr_t block_size, int64_t *ssz";
|
||||
specialize qw/av1_block_error sse2 avx2 msa/;
|
||||
|
|
@ -347,249 +332,257 @@ if (aom_config("CONFIG_AOM_QM") eq "yes") {
|
|||
add_proto qw/void av1_quantize_fp_64x64/, "const tran_low_t *coeff_ptr, intptr_t n_coeffs, int skip_block, const int16_t *zbin_ptr, const int16_t *round_ptr, const int16_t *quant_ptr, const int16_t *quant_shift_ptr, tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr, const int16_t *dequant_ptr, uint16_t *eob_ptr, const int16_t *scan, const int16_t *iscan";
|
||||
}
|
||||
|
||||
add_proto qw/void av1_fdct8x8_quant/, "const int16_t *input, int stride, tran_low_t *coeff_ptr, intptr_t n_coeffs, int skip_block, const int16_t *zbin_ptr, const int16_t *round_ptr, const int16_t *quant_ptr, const int16_t *quant_shift_ptr, tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr, const int16_t *dequant_ptr, uint16_t *eob_ptr, const int16_t *scan, const int16_t *iscan";
|
||||
specialize qw/av1_fdct8x8_quant sse2 ssse3 neon/;
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
# fdct functions
|
||||
|
||||
add_proto qw/void av1_fht4x4/, "const int16_t *input, tran_low_t *output, int stride, struct txfm_param *param";
|
||||
specialize qw/av1_fht4x4 sse2/;
|
||||
|
||||
add_proto qw/void av1_fwht4x4/, "const int16_t *input, tran_low_t *output, int stride";
|
||||
|
||||
add_proto qw/void av1_fht8x8/, "const int16_t *input, tran_low_t *output, int stride, struct txfm_param *param";
|
||||
specialize qw/av1_fht8x8 sse2/;
|
||||
|
||||
add_proto qw/void av1_fht16x16/, "const int16_t *input, tran_low_t *output, int stride, struct txfm_param *param";
|
||||
specialize qw/av1_fht16x16 sse2 avx2/;
|
||||
|
||||
add_proto qw/void av1_fht32x32/, "const int16_t *input, tran_low_t *output, int stride, struct txfm_param *param";
|
||||
specialize qw/av1_fht32x32 sse2 avx2/;
|
||||
|
||||
if (aom_config("CONFIG_TX64X64") eq "yes") {
|
||||
add_proto qw/void av1_fht64x64/, "const int16_t *input, tran_low_t *output, int stride, struct txfm_param *param";
|
||||
}
|
||||
|
||||
add_proto qw/void av1_fht4x8/, "const int16_t *input, tran_low_t *output, int stride, struct txfm_param *param";
|
||||
specialize qw/av1_fht4x8 sse2/;
|
||||
|
||||
add_proto qw/void av1_fht8x4/, "const int16_t *input, tran_low_t *output, int stride, struct txfm_param *param";
|
||||
specialize qw/av1_fht8x4 sse2/;
|
||||
|
||||
add_proto qw/void av1_fht8x16/, "const int16_t *input, tran_low_t *output, int stride, struct txfm_param *param";
|
||||
specialize qw/av1_fht8x16 sse2/;
|
||||
|
||||
add_proto qw/void av1_fht16x8/, "const int16_t *input, tran_low_t *output, int stride, struct txfm_param *param";
|
||||
specialize qw/av1_fht16x8 sse2/;
|
||||
|
||||
add_proto qw/void av1_fht16x32/, "const int16_t *input, tran_low_t *output, int stride, struct txfm_param *param";
|
||||
specialize qw/av1_fht16x32 sse2/;
|
||||
|
||||
add_proto qw/void av1_fht32x16/, "const int16_t *input, tran_low_t *output, int stride, struct txfm_param *param";
|
||||
specialize qw/av1_fht32x16 sse2/;
|
||||
|
||||
add_proto qw/void av1_fht4x16/, "const int16_t *input, tran_low_t *output, int stride, struct txfm_param *param";
|
||||
|
||||
add_proto qw/void av1_fht16x4/, "const int16_t *input, tran_low_t *output, int stride, struct txfm_param *param";
|
||||
|
||||
add_proto qw/void av1_fht8x32/, "const int16_t *input, tran_low_t *output, int stride, struct txfm_param *param";
|
||||
|
||||
add_proto qw/void av1_fht32x8/, "const int16_t *input, tran_low_t *output, int stride, struct txfm_param *param";
|
||||
|
||||
if (aom_config("CONFIG_HIGHBITDEPTH") ne "yes") {
|
||||
if (aom_config("CONFIG_EXT_TX") ne "yes") {
|
||||
specialize qw/av1_fht4x4 msa/;
|
||||
specialize qw/av1_fht8x8 msa/;
|
||||
specialize qw/av1_fht16x16 msa/;
|
||||
}
|
||||
}
|
||||
|
||||
add_proto qw/void av1_fwd_idtx/, "const int16_t *src_diff, tran_low_t *coeff, int stride, int bs, int tx_type";
|
||||
|
||||
if (aom_config("CONFIG_DPCM_INTRA") eq "yes") {
|
||||
@sizes = (4, 8, 16, 32);
|
||||
foreach $size (@sizes) {
|
||||
if (aom_config("CONFIG_HIGHBITDEPTH") eq "yes") {
|
||||
add_proto "void", "av1_hbd_dpcm_ft$size", "const int16_t *input, int stride, TX_TYPE_1D tx_type, tran_low_t *output, int dir";
|
||||
}
|
||||
add_proto "void", "av1_dpcm_ft$size", "const int16_t *input, int stride, TX_TYPE_1D tx_type, tran_low_t *output";
|
||||
}
|
||||
}
|
||||
|
||||
#fwd txfm
|
||||
add_proto qw/void av1_fwd_txfm2d_4x8/, "const int16_t *input, int32_t *output, int stride, int tx_type, int bd";
|
||||
add_proto qw/void av1_fwd_txfm2d_8x4/, "const int16_t *input, int32_t *output, int stride, int tx_type, int bd";
|
||||
add_proto qw/void av1_fwd_txfm2d_8x16/, "const int16_t *input, int32_t *output, int stride, int tx_type, int bd";
|
||||
add_proto qw/void av1_fwd_txfm2d_16x8/, "const int16_t *input, int32_t *output, int stride, int tx_type, int bd";
|
||||
add_proto qw/void av1_fwd_txfm2d_16x32/, "const int16_t *input, int32_t *output, int stride, int tx_type, int bd";
|
||||
add_proto qw/void av1_fwd_txfm2d_32x16/, "const int16_t *input, int32_t *output, int stride, int tx_type, int bd";
|
||||
add_proto qw/void av1_fwd_txfm2d_4x4/, "const int16_t *input, int32_t *output, int stride, int tx_type, int bd";
|
||||
specialize qw/av1_fwd_txfm2d_4x4 sse4_1/;
|
||||
add_proto qw/void av1_fwd_txfm2d_8x8/, "const int16_t *input, int32_t *output, int stride, int tx_type, int bd";
|
||||
specialize qw/av1_fwd_txfm2d_8x8 sse4_1/;
|
||||
add_proto qw/void av1_fwd_txfm2d_16x16/, "const int16_t *input, int32_t *output, int stride, int tx_type, int bd";
|
||||
specialize qw/av1_fwd_txfm2d_16x16 sse4_1/;
|
||||
add_proto qw/void av1_fwd_txfm2d_32x32/, "const int16_t *input, int32_t *output, int stride, int tx_type, int bd";
|
||||
specialize qw/av1_fwd_txfm2d_32x32 sse4_1/;
|
||||
add_proto qw/void av1_fwd_txfm2d_64x64/, "const int16_t *input, int32_t *output, int stride, int tx_type, int bd";
|
||||
specialize qw/av1_fwd_txfm2d_64x64 sse4_1/;
|
||||
|
||||
#
|
||||
# Motion search
|
||||
#
|
||||
add_proto qw/int av1_full_search_sad/, "const struct macroblock *x, const struct mv *ref_mv, int sad_per_bit, int distance, const struct aom_variance_vtable *fn_ptr, const struct mv *center_mv, struct mv *best_mv";
|
||||
specialize qw/av1_full_search_sad sse3 sse4_1/;
|
||||
$av1_full_search_sad_sse3=av1_full_search_sadx3;
|
||||
$av1_full_search_sad_sse4_1=av1_full_search_sadx8;
|
||||
|
||||
add_proto qw/int av1_diamond_search_sad/, "struct macroblock *x, const struct search_site_config *cfg, struct mv *ref_mv, struct mv *best_mv, int search_param, int sad_per_bit, int *num00, const struct aom_variance_vtable *fn_ptr, const struct mv *center_mv";
|
||||
|
||||
add_proto qw/int av1_full_range_search/, "const struct macroblock *x, const struct search_site_config *cfg, struct mv *ref_mv, struct mv *best_mv, int search_param, int sad_per_bit, int *num00, const struct aom_variance_vtable *fn_ptr, const struct mv *center_mv";
|
||||
|
||||
add_proto qw/void av1_temporal_filter_apply/, "uint8_t *frame1, unsigned int stride, uint8_t *frame2, unsigned int block_width, unsigned int block_height, int strength, int filter_weight, unsigned int *accumulator, uint16_t *count";
|
||||
specialize qw/av1_temporal_filter_apply sse2 msa/;
|
||||
|
||||
if (aom_config("CONFIG_AOM_QM") eq "yes") {
|
||||
add_proto qw/void av1_quantize_b/, "const tran_low_t *coeff_ptr, intptr_t n_coeffs, int skip_block, const int16_t *zbin_ptr, const int16_t *round_ptr, const int16_t *quant_ptr, const int16_t *quant_shift_ptr, tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr, const int16_t *dequant_ptr, uint16_t *eob_ptr, const int16_t *scan, const int16_t *iscan, const qm_val_t * qm_ptr, const qm_val_t * iqm_ptr, int log_scale";
|
||||
} else {
|
||||
add_proto qw/void av1_quantize_b/, "const tran_low_t *coeff_ptr, intptr_t n_coeffs, int skip_block, const int16_t *zbin_ptr, const int16_t *round_ptr, const int16_t *quant_ptr, const int16_t *quant_shift_ptr, tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr, const int16_t *dequant_ptr, uint16_t *eob_ptr, const int16_t *scan, const int16_t *iscan, int log_scale";
|
||||
}
|
||||
|
||||
if (aom_config("CONFIG_HIGHBITDEPTH") eq "yes") {
|
||||
|
||||
# ENCODEMB INVOKE
|
||||
if (aom_config("CONFIG_NEW_QUANT") eq "yes") {
|
||||
add_proto qw/void highbd_quantize_nuq/, "const tran_low_t *coeff_ptr, intptr_t n_coeffs, int skip_block, const int16_t *quant_ptr, const int16_t *quant_shift_ptr, const int16_t *dequant_ptr, const cuml_bins_type_nuq *cuml_bins_ptr, const dequant_val_type_nuq *dequant_val, tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr, uint16_t *eob_ptr, const int16_t *scan, const uint8_t *band";
|
||||
|
||||
add_proto qw/void highbd_quantize_fp_nuq/, "const tran_low_t *coeff_ptr, intptr_t n_coeffs, int skip_block, const int16_t *quant_ptr, const int16_t *dequant_ptr, const cuml_bins_type_nuq *cuml_bins_ptr, const dequant_val_type_nuq *dequant_val, tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr, uint16_t *eob_ptr, const int16_t *scan, const uint8_t *band";
|
||||
|
||||
add_proto qw/void highbd_quantize_32x32_nuq/, "const tran_low_t *coeff_ptr, intptr_t n_coeffs, int skip_block, const int16_t *quant_ptr, const int16_t *quant_shift_ptr, const int16_t *dequant_ptr, const cuml_bins_type_nuq *cuml_bins_ptr, const dequant_val_type_nuq *dequant_val, tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr, uint16_t *eob_ptr, const int16_t *scan, const uint8_t *band";
|
||||
|
||||
add_proto qw/void highbd_quantize_32x32_fp_nuq/, "const tran_low_t *coeff_ptr, intptr_t n_coeffs, int skip_block, const int16_t *quant_ptr, const int16_t *dequant_ptr, const cuml_bins_type_nuq *cuml_bins_ptr, const dequant_val_type_nuq *dequant_val, tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr, uint16_t *eob_ptr, const int16_t *scan, const uint8_t *band";
|
||||
|
||||
if (aom_config("CONFIG_TX64X64") eq "yes") {
|
||||
add_proto qw/void highbd_quantize_64x64_nuq/, "const tran_low_t *coeff_ptr, intptr_t n_coeffs, int skip_block, const int16_t *quant_ptr, const int16_t *quant_shift_ptr, const int16_t *dequant_ptr, const cuml_bins_type_nuq *cuml_bins_ptr, const dequant_val_type_nuq *dequant_val, tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr, uint16_t *eob_ptr, const int16_t *scan, const uint8_t *band";
|
||||
|
||||
add_proto qw/void highbd_quantize_64x64_fp_nuq/, "const tran_low_t *coeff_ptr, intptr_t n_coeffs, int skip_block, const int16_t *quant_ptr, const int16_t *dequant_ptr, const cuml_bins_type_nuq *cuml_bins_ptr, const dequant_val_type_nuq *dequant_val, tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr, uint16_t *eob_ptr, const int16_t *scan, const uint8_t *band";
|
||||
}
|
||||
}
|
||||
|
||||
add_proto qw/int64_t av1_highbd_block_error/, "const tran_low_t *coeff, const tran_low_t *dqcoeff, intptr_t block_size, int64_t *ssz, int bd";
|
||||
specialize qw/av1_highbd_block_error sse2/;
|
||||
|
||||
# fdct functions
|
||||
if (aom_config("CONFIG_TX64X64") eq "yes") {
|
||||
add_proto qw/void av1_highbd_fht64x64/, "const int16_t *input, tran_low_t *output, int stride, int tx_type";
|
||||
|
||||
add_proto qw/void av1_fht4x4/, "const int16_t *input, tran_low_t *output, int stride, struct txfm_param *param";
|
||||
if (aom_config("CONFIG_DAALA_DCT4") ne "yes") {
|
||||
specialize qw/av1_fht4x4 sse2/;
|
||||
}
|
||||
|
||||
add_proto qw/void av1_highbd_temporal_filter_apply/, "uint8_t *frame1, unsigned int stride, uint8_t *frame2, unsigned int block_width, unsigned int block_height, int strength, int filter_weight, unsigned int *accumulator, uint16_t *count";
|
||||
add_proto qw/void av1_fwht4x4/, "const int16_t *input, tran_low_t *output, int stride";
|
||||
|
||||
add_proto qw/void av1_fht8x8/, "const int16_t *input, tran_low_t *output, int stride, struct txfm_param *param";
|
||||
if (aom_config("CONFIG_DAALA_DCT8") ne "yes") {
|
||||
specialize qw/av1_fht8x8 sse2/;
|
||||
}
|
||||
|
||||
add_proto qw/void av1_fht16x16/, "const int16_t *input, tran_low_t *output, int stride, struct txfm_param *param";
|
||||
if (aom_config("CONFIG_DAALA_DCT16") ne "yes") {
|
||||
specialize qw/av1_fht16x16 sse2 avx2/;
|
||||
}
|
||||
|
||||
add_proto qw/void av1_fht32x32/, "const int16_t *input, tran_low_t *output, int stride, struct txfm_param *param";
|
||||
if (aom_config("CONFIG_DAALA_DCT32") ne "yes") {
|
||||
specialize qw/av1_fht32x32 sse2 avx2/;
|
||||
}
|
||||
|
||||
if (aom_config("CONFIG_TX64X64") eq "yes") {
|
||||
add_proto qw/void av1_fht64x64/, "const int16_t *input, tran_low_t *output, int stride, struct txfm_param *param";
|
||||
add_proto qw/void av1_fht64x32/, "const int16_t *input, tran_low_t *output, int stride, struct txfm_param *param";
|
||||
add_proto qw/void av1_fht32x64/, "const int16_t *input, tran_low_t *output, int stride, struct txfm_param *param";
|
||||
}
|
||||
|
||||
add_proto qw/void av1_fht4x8/, "const int16_t *input, tran_low_t *output, int stride, struct txfm_param *param";
|
||||
specialize qw/av1_fht4x8 sse2/;
|
||||
|
||||
add_proto qw/void av1_fht8x4/, "const int16_t *input, tran_low_t *output, int stride, struct txfm_param *param";
|
||||
specialize qw/av1_fht8x4 sse2/;
|
||||
|
||||
add_proto qw/void av1_fht8x16/, "const int16_t *input, tran_low_t *output, int stride, struct txfm_param *param";
|
||||
specialize qw/av1_fht8x16 sse2/;
|
||||
|
||||
add_proto qw/void av1_fht16x8/, "const int16_t *input, tran_low_t *output, int stride, struct txfm_param *param";
|
||||
specialize qw/av1_fht16x8 sse2/;
|
||||
|
||||
add_proto qw/void av1_fht16x32/, "const int16_t *input, tran_low_t *output, int stride, struct txfm_param *param";
|
||||
specialize qw/av1_fht16x32 sse2/;
|
||||
|
||||
add_proto qw/void av1_fht32x16/, "const int16_t *input, tran_low_t *output, int stride, struct txfm_param *param";
|
||||
specialize qw/av1_fht32x16 sse2/;
|
||||
|
||||
add_proto qw/void av1_fht4x16/, "const int16_t *input, tran_low_t *output, int stride, struct txfm_param *param";
|
||||
|
||||
add_proto qw/void av1_fht16x4/, "const int16_t *input, tran_low_t *output, int stride, struct txfm_param *param";
|
||||
|
||||
add_proto qw/void av1_fht8x32/, "const int16_t *input, tran_low_t *output, int stride, struct txfm_param *param";
|
||||
|
||||
add_proto qw/void av1_fht32x8/, "const int16_t *input, tran_low_t *output, int stride, struct txfm_param *param";
|
||||
|
||||
if (aom_config("CONFIG_HIGHBITDEPTH") ne "yes") {
|
||||
if (aom_config("CONFIG_EXT_TX") ne "yes") {
|
||||
if (aom_config("CONFIG_DAALA_DCT4") ne "yes") {
|
||||
specialize qw/av1_fht4x4 msa/;
|
||||
}
|
||||
if (aom_config("CONFIG_DAALA_DCT8") ne "yes") {
|
||||
specialize qw/av1_fht8x8 msa/;
|
||||
}
|
||||
if (aom_config("CONFIG_DAALA_DCT16") ne "yes") {
|
||||
specialize qw/av1_fht16x16 msa/;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
add_proto qw/void av1_fwd_idtx/, "const int16_t *src_diff, tran_low_t *coeff, int stride, int bsx, int bsy, TX_TYPE tx_type";
|
||||
|
||||
#fwd txfm
|
||||
add_proto qw/void av1_fwd_txfm2d_4x8/, "const int16_t *input, int32_t *output, int stride, TX_TYPE tx_type, int bd";
|
||||
add_proto qw/void av1_fwd_txfm2d_8x4/, "const int16_t *input, int32_t *output, int stride, TX_TYPE tx_type, int bd";
|
||||
add_proto qw/void av1_fwd_txfm2d_8x16/, "const int16_t *input, int32_t *output, int stride, TX_TYPE tx_type, int bd";
|
||||
add_proto qw/void av1_fwd_txfm2d_16x8/, "const int16_t *input, int32_t *output, int stride, TX_TYPE tx_type, int bd";
|
||||
add_proto qw/void av1_fwd_txfm2d_16x32/, "const int16_t *input, int32_t *output, int stride, TX_TYPE tx_type, int bd";
|
||||
add_proto qw/void av1_fwd_txfm2d_32x16/, "const int16_t *input, int32_t *output, int stride, TX_TYPE tx_type, int bd";
|
||||
add_proto qw/void av1_fwd_txfm2d_4x4/, "const int16_t *input, int32_t *output, int stride, TX_TYPE tx_type, int bd";
|
||||
if (aom_config("CONFIG_DAALA_DCT4") ne "yes") {
|
||||
specialize qw/av1_fwd_txfm2d_4x4 sse4_1/;
|
||||
}
|
||||
add_proto qw/void av1_fwd_txfm2d_8x8/, "const int16_t *input, int32_t *output, int stride, TX_TYPE tx_type, int bd";
|
||||
if (aom_config("CONFIG_DAALA_DCT8") ne "yes") {
|
||||
specialize qw/av1_fwd_txfm2d_8x8 sse4_1/;
|
||||
}
|
||||
add_proto qw/void av1_fwd_txfm2d_16x16/, "const int16_t *input, int32_t *output, int stride, TX_TYPE tx_type, int bd";
|
||||
if (aom_config("CONFIG_DAALA_DCT16") ne "yes") {
|
||||
specialize qw/av1_fwd_txfm2d_16x16 sse4_1/;
|
||||
}
|
||||
add_proto qw/void av1_fwd_txfm2d_32x32/, "const int16_t *input, int32_t *output, int stride, TX_TYPE tx_type, int bd";
|
||||
if (aom_config("CONFIG_DAALA_DCT32") ne "yes") {
|
||||
specialize qw/av1_fwd_txfm2d_32x32 sse4_1/;
|
||||
}
|
||||
|
||||
if (aom_config("CONFIG_TX64X64") eq "yes") {
|
||||
add_proto qw/void av1_fwd_txfm2d_32x64/, "const int16_t *input, int32_t *output, int stride, TX_TYPE tx_type, int bd";
|
||||
add_proto qw/void av1_fwd_txfm2d_64x32/, "const int16_t *input, int32_t *output, int stride, TX_TYPE tx_type, int bd";
|
||||
add_proto qw/void av1_fwd_txfm2d_64x64/, "const int16_t *input, int32_t *output, int stride, TX_TYPE tx_type, int bd";
|
||||
}
|
||||
#
|
||||
# Motion search
|
||||
#
|
||||
add_proto qw/int av1_full_search_sad/, "const struct macroblock *x, const struct mv *ref_mv, int sad_per_bit, int distance, const struct aom_variance_vtable *fn_ptr, const struct mv *center_mv, struct mv *best_mv";
|
||||
specialize qw/av1_full_search_sad sse3 sse4_1/;
|
||||
$av1_full_search_sad_sse3=av1_full_search_sadx3;
|
||||
$av1_full_search_sad_sse4_1=av1_full_search_sadx8;
|
||||
|
||||
add_proto qw/int av1_diamond_search_sad/, "struct macroblock *x, const struct search_site_config *cfg, struct mv *ref_mv, struct mv *best_mv, int search_param, int sad_per_bit, int *num00, const struct aom_variance_vtable *fn_ptr, const struct mv *center_mv";
|
||||
|
||||
add_proto qw/int av1_full_range_search/, "const struct macroblock *x, const struct search_site_config *cfg, struct mv *ref_mv, struct mv *best_mv, int search_param, int sad_per_bit, int *num00, const struct aom_variance_vtable *fn_ptr, const struct mv *center_mv";
|
||||
|
||||
add_proto qw/void av1_temporal_filter_apply/, "uint8_t *frame1, unsigned int stride, uint8_t *frame2, unsigned int block_width, unsigned int block_height, int strength, int filter_weight, unsigned int *accumulator, uint16_t *count";
|
||||
specialize qw/av1_temporal_filter_apply sse2 msa/;
|
||||
|
||||
if (aom_config("CONFIG_AOM_QM") eq "yes") {
|
||||
add_proto qw/void av1_quantize_b/, "const tran_low_t *coeff_ptr, intptr_t n_coeffs, int skip_block, const int16_t *zbin_ptr, const int16_t *round_ptr, const int16_t *quant_ptr, const int16_t *quant_shift_ptr, tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr, const int16_t *dequant_ptr, uint16_t *eob_ptr, const int16_t *scan, const int16_t *iscan, const qm_val_t * qm_ptr, const qm_val_t * iqm_ptr, int log_scale";
|
||||
} else {
|
||||
add_proto qw/void av1_quantize_b/, "const tran_low_t *coeff_ptr, intptr_t n_coeffs, int skip_block, const int16_t *zbin_ptr, const int16_t *round_ptr, const int16_t *quant_ptr, const int16_t *quant_shift_ptr, tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr, const int16_t *dequant_ptr, uint16_t *eob_ptr, const int16_t *scan, const int16_t *iscan, int log_scale";
|
||||
}
|
||||
|
||||
if (aom_config("CONFIG_LGT_FROM_PRED") eq "yes") {
|
||||
add_proto qw/void flgt2d_from_pred/, "const int16_t *input, tran_low_t *output, int stride, struct txfm_param *param";
|
||||
}
|
||||
|
||||
if (aom_config("CONFIG_HIGHBITDEPTH") eq "yes") {
|
||||
|
||||
# ENCODEMB INVOKE
|
||||
if (aom_config("CONFIG_NEW_QUANT") eq "yes") {
|
||||
add_proto qw/void highbd_quantize_nuq/, "const tran_low_t *coeff_ptr, intptr_t n_coeffs, int skip_block, const int16_t *quant_ptr, const int16_t *quant_shift_ptr, const int16_t *dequant_ptr, const cuml_bins_type_nuq *cuml_bins_ptr, const dequant_val_type_nuq *dequant_val, tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr, uint16_t *eob_ptr, const int16_t *scan, const uint8_t *band";
|
||||
|
||||
add_proto qw/void highbd_quantize_fp_nuq/, "const tran_low_t *coeff_ptr, intptr_t n_coeffs, int skip_block, const int16_t *quant_ptr, const int16_t *dequant_ptr, const cuml_bins_type_nuq *cuml_bins_ptr, const dequant_val_type_nuq *dequant_val, tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr, uint16_t *eob_ptr, const int16_t *scan, const uint8_t *band";
|
||||
|
||||
add_proto qw/void highbd_quantize_32x32_nuq/, "const tran_low_t *coeff_ptr, intptr_t n_coeffs, int skip_block, const int16_t *quant_ptr, const int16_t *quant_shift_ptr, const int16_t *dequant_ptr, const cuml_bins_type_nuq *cuml_bins_ptr, const dequant_val_type_nuq *dequant_val, tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr, uint16_t *eob_ptr, const int16_t *scan, const uint8_t *band";
|
||||
|
||||
add_proto qw/void highbd_quantize_32x32_fp_nuq/, "const tran_low_t *coeff_ptr, intptr_t n_coeffs, int skip_block, const int16_t *quant_ptr, const int16_t *dequant_ptr, const cuml_bins_type_nuq *cuml_bins_ptr, const dequant_val_type_nuq *dequant_val, tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr, uint16_t *eob_ptr, const int16_t *scan, const uint8_t *band";
|
||||
|
||||
if (aom_config("CONFIG_TX64X64") eq "yes") {
|
||||
add_proto qw/void highbd_quantize_64x64_nuq/, "const tran_low_t *coeff_ptr, intptr_t n_coeffs, int skip_block, const int16_t *quant_ptr, const int16_t *quant_shift_ptr, const int16_t *dequant_ptr, const cuml_bins_type_nuq *cuml_bins_ptr, const dequant_val_type_nuq *dequant_val, tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr, uint16_t *eob_ptr, const int16_t *scan, const uint8_t *band";
|
||||
|
||||
add_proto qw/void highbd_quantize_64x64_fp_nuq/, "const tran_low_t *coeff_ptr, intptr_t n_coeffs, int skip_block, const int16_t *quant_ptr, const int16_t *dequant_ptr, const cuml_bins_type_nuq *cuml_bins_ptr, const dequant_val_type_nuq *dequant_val, tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr, uint16_t *eob_ptr, const int16_t *scan, const uint8_t *band";
|
||||
}
|
||||
}
|
||||
|
||||
add_proto qw/int64_t av1_highbd_block_error/, "const tran_low_t *coeff, const tran_low_t *dqcoeff, intptr_t block_size, int64_t *ssz, int bd";
|
||||
specialize qw/av1_highbd_block_error sse2/;
|
||||
|
||||
add_proto qw/void av1_highbd_temporal_filter_apply/, "uint8_t *frame1, unsigned int stride, uint8_t *frame2, unsigned int block_width, unsigned int block_height, int strength, int filter_weight, unsigned int *accumulator, uint16_t *count";
|
||||
|
||||
}
|
||||
|
||||
if (aom_config("CONFIG_AOM_QM") eq "yes") {
|
||||
add_proto qw/void av1_highbd_quantize_fp/, "const tran_low_t *coeff_ptr, intptr_t n_coeffs, int skip_block, const int16_t *zbin_ptr, const int16_t *round_ptr, const int16_t *quant_ptr, const int16_t *quant_shift_ptr, tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr, const int16_t *dequant_ptr, uint16_t *eob_ptr, const int16_t *scan, const int16_t *iscan, const qm_val_t * qm_ptr, const qm_val_t * iqm_ptr, int log_scale";
|
||||
|
||||
add_proto qw/void av1_highbd_quantize_fp_32x32/, "const tran_low_t *coeff_ptr, intptr_t n_coeffs, int skip_block, const int16_t *zbin_ptr, const int16_t *round_ptr, const int16_t *quant_ptr, const int16_t *quant_shift_ptr, tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr, const int16_t *dequant_ptr, uint16_t *eob_ptr, const int16_t *scan, const int16_t *iscan, const qm_val_t * qm_ptr, const qm_val_t * iqm_ptr, int log_scale";
|
||||
|
||||
if (aom_config("CONFIG_TX64X64") eq "yes") {
|
||||
add_proto qw/void av1_highbd_quantize_fp_64x64/, "const tran_low_t *coeff_ptr, intptr_t n_coeffs, int skip_block, const int16_t *zbin_ptr, const int16_t *round_ptr, const int16_t *quant_ptr, const int16_t *quant_shift_ptr, tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr, const int16_t *dequant_ptr, uint16_t *eob_ptr, const int16_t *scan, const int16_t *iscan, const qm_val_t * qm_ptr, const qm_val_t * iqm_ptr, int log_scale";
|
||||
}
|
||||
|
||||
add_proto qw/void av1_highbd_quantize_b/, "const tran_low_t *coeff_ptr, intptr_t n_coeffs, int skip_block, const int16_t *zbin_ptr, const int16_t *round_ptr, const int16_t *quant_ptr, const int16_t *quant_shift_ptr, tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr, const int16_t *dequant_ptr, uint16_t *eob_ptr, const int16_t *scan, const int16_t *iscan, const qm_val_t * qm_ptr, const qm_val_t * iqm_ptr, int log_scale";
|
||||
} else {
|
||||
add_proto qw/void av1_highbd_quantize_fp/, "const tran_low_t *coeff_ptr, intptr_t n_coeffs, int skip_block, const int16_t *zbin_ptr, const int16_t *round_ptr, const int16_t *quant_ptr, const int16_t *quant_shift_ptr, tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr, const int16_t *dequant_ptr, uint16_t *eob_ptr, const int16_t *scan, const int16_t *iscan, int log_scale";
|
||||
specialize qw/av1_highbd_quantize_fp sse4_1 avx2/;
|
||||
|
||||
add_proto qw/void av1_highbd_quantize_b/, "const tran_low_t *coeff_ptr, intptr_t n_coeffs, int skip_block, const int16_t *zbin_ptr, const int16_t *round_ptr, const int16_t *quant_ptr, const int16_t *quant_shift_ptr, tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr, const int16_t *dequant_ptr, uint16_t *eob_ptr, const int16_t *scan, const int16_t *iscan, int log_scale";
|
||||
}
|
||||
add_proto qw/void av1_highbd_fwht4x4/, "const int16_t *input, tran_low_t *output, int stride";
|
||||
|
||||
add_proto qw/void av1_highbd_fwht4x4/, "const int16_t *input, tran_low_t *output, int stride";
|
||||
# End av1_high encoder functions
|
||||
|
||||
# End av1_high encoder functions
|
||||
|
||||
if (aom_config("CONFIG_EXT_INTER") eq "yes") {
|
||||
add_proto qw/uint64_t av1_wedge_sse_from_residuals/, "const int16_t *r1, const int16_t *d, const uint8_t *m, int N";
|
||||
specialize qw/av1_wedge_sse_from_residuals sse2/;
|
||||
add_proto qw/int av1_wedge_sign_from_residuals/, "const int16_t *ds, const uint8_t *m, int N, int64_t limit";
|
||||
specialize qw/av1_wedge_sign_from_residuals sse2/;
|
||||
add_proto qw/void av1_wedge_compute_delta_squares/, "int16_t *d, const int16_t *a, const int16_t *b, int N";
|
||||
specialize qw/av1_wedge_compute_delta_squares sse2/;
|
||||
}
|
||||
|
||||
}
|
||||
# end encoder functions
|
||||
|
||||
# If PVQ is enabled, fwd transforms are required by decoder
|
||||
if (aom_config("CONFIG_PVQ") eq "yes") {
|
||||
# fdct functions
|
||||
# fdct functions
|
||||
|
||||
if (aom_config("CONFIG_HIGHBITDEPTH") eq "yes") {
|
||||
add_proto qw/void av1_fht4x4/, "const int16_t *input, tran_low_t *output, int stride, struct txfm_param *param";
|
||||
specialize qw/av1_fht4x4 sse2/;
|
||||
if (aom_config("CONFIG_HIGHBITDEPTH") eq "yes") {
|
||||
add_proto qw/void av1_fht4x4/, "const int16_t *input, tran_low_t *output, int stride, struct txfm_param *param";
|
||||
specialize qw/av1_fht4x4 sse2/;
|
||||
|
||||
add_proto qw/void av1_fht8x8/, "const int16_t *input, tran_low_t *output, int stride, struct txfm_param *param";
|
||||
specialize qw/av1_fht8x8 sse2/;
|
||||
add_proto qw/void av1_fht8x8/, "const int16_t *input, tran_low_t *output, int stride, struct txfm_param *param";
|
||||
specialize qw/av1_fht8x8 sse2/;
|
||||
|
||||
add_proto qw/void av1_fht16x16/, "const int16_t *input, tran_low_t *output, int stride, struct txfm_param *param";
|
||||
specialize qw/av1_fht16x16 sse2/;
|
||||
add_proto qw/void av1_fht16x16/, "const int16_t *input, tran_low_t *output, int stride, struct txfm_param *param";
|
||||
specialize qw/av1_fht16x16 sse2/;
|
||||
|
||||
add_proto qw/void av1_fwht4x4/, "const int16_t *input, tran_low_t *output, int stride";
|
||||
specialize qw/av1_fwht4x4 sse2/;
|
||||
} else {
|
||||
add_proto qw/void av1_fht4x4/, "const int16_t *input, tran_low_t *output, int stride, struct txfm_param *param";
|
||||
specialize qw/av1_fht4x4 sse2 msa/;
|
||||
add_proto qw/void av1_fwht4x4/, "const int16_t *input, tran_low_t *output, int stride";
|
||||
specialize qw/av1_fwht4x4 sse2/;
|
||||
} else {
|
||||
add_proto qw/void av1_fht4x4/, "const int16_t *input, tran_low_t *output, int stride, struct txfm_param *param";
|
||||
specialize qw/av1_fht4x4 sse2 msa/;
|
||||
|
||||
add_proto qw/void av1_fht8x8/, "const int16_t *input, tran_low_t *output, int stride, struct txfm_param *param";
|
||||
specialize qw/av1_fht8x8 sse2 msa/;
|
||||
add_proto qw/void av1_fht8x8/, "const int16_t *input, tran_low_t *output, int stride, struct txfm_param *param";
|
||||
specialize qw/av1_fht8x8 sse2 msa/;
|
||||
|
||||
add_proto qw/void av1_fht16x16/, "const int16_t *input, tran_low_t *output, int stride, struct txfm_param *param";
|
||||
specialize qw/av1_fht16x16 sse2 msa/;
|
||||
add_proto qw/void av1_fht16x16/, "const int16_t *input, tran_low_t *output, int stride, struct txfm_param *param";
|
||||
specialize qw/av1_fht16x16 sse2 msa/;
|
||||
|
||||
add_proto qw/void av1_fwht4x4/, "const int16_t *input, tran_low_t *output, int stride";
|
||||
specialize qw/av1_fwht4x4 msa sse2/;
|
||||
}
|
||||
add_proto qw/void av1_fwht4x4/, "const int16_t *input, tran_low_t *output, int stride";
|
||||
specialize qw/av1_fwht4x4 msa sse2/;
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
# Deringing Functions
|
||||
|
||||
if (aom_config("CONFIG_CDEF") eq "yes") {
|
||||
add_proto qw/void aom_clpf_block_hbd/, "uint16_t *dst, const uint16_t *src, int dstride, int sstride, int sizex, int sizey, unsigned int strength, unsigned int bd";
|
||||
add_proto qw/void aom_clpf_hblock_hbd/, "uint16_t *dst, const uint16_t *src, int dstride, int sstride, int sizex, int sizey, unsigned int strength, unsigned int bd";
|
||||
add_proto qw/void aom_clpf_block/, "uint8_t *dst, const uint16_t *src, int dstride, int sstride, int sizex, int sizey, unsigned int strength, unsigned int bd";
|
||||
add_proto qw/void aom_clpf_hblock/, "uint8_t *dst, const uint16_t *src, int dstride, int sstride, int sizex, int sizey, unsigned int strength, unsigned int bd";
|
||||
add_proto qw/int od_dir_find8/, "const od_dering_in *img, int stride, int32_t *var, int coeff_shift";
|
||||
add_proto qw/void od_filter_dering_direction_4x4/, "uint16_t *y, int ystride, const uint16_t *in, int threshold, int dir, int damping";
|
||||
add_proto qw/void od_filter_dering_direction_8x8/, "uint16_t *y, int ystride, const uint16_t *in, int threshold, int dir, int damping";
|
||||
add_proto qw/int cdef_find_dir/, "const uint16_t *img, int stride, int32_t *var, int coeff_shift";
|
||||
if (aom_config("CONFIG_CDEF_SINGLEPASS") ne "yes") {
|
||||
add_proto qw/void aom_clpf_block_hbd/, "uint16_t *dst, const uint16_t *src, int dstride, int sstride, int sizex, int sizey, unsigned int strength, unsigned int bd";
|
||||
add_proto qw/void aom_clpf_hblock_hbd/, "uint16_t *dst, const uint16_t *src, int dstride, int sstride, int sizex, int sizey, unsigned int strength, unsigned int bd";
|
||||
add_proto qw/void aom_clpf_block/, "uint8_t *dst, const uint16_t *src, int dstride, int sstride, int sizex, int sizey, unsigned int strength, unsigned int bd";
|
||||
add_proto qw/void aom_clpf_hblock/, "uint8_t *dst, const uint16_t *src, int dstride, int sstride, int sizex, int sizey, unsigned int strength, unsigned int bd";
|
||||
add_proto qw/void cdef_direction_4x4/, "uint16_t *y, int ystride, const uint16_t *in, int threshold, int dir, int damping";
|
||||
add_proto qw/void cdef_direction_8x8/, "uint16_t *y, int ystride, const uint16_t *in, int threshold, int dir, int damping";
|
||||
add_proto qw/void copy_8x8_16bit_to_8bit/, "uint8_t *dst, int dstride, const uint16_t *src, int sstride";
|
||||
add_proto qw/void copy_4x4_16bit_to_8bit/, "uint8_t *dst, int dstride, const uint16_t *src, int sstride";
|
||||
add_proto qw/void copy_8x8_16bit_to_16bit/, "uint16_t *dst, int dstride, const uint16_t *src, int sstride";
|
||||
add_proto qw/void copy_4x4_16bit_to_16bit/, "uint16_t *dst, int dstride, const uint16_t *src, int sstride";
|
||||
} else {
|
||||
add_proto qw/void cdef_filter_block/, "uint8_t *dst8, uint16_t *dst16, int dstride, const uint16_t *in, int pri_strength, int sec_strength, int dir, int pri_damping, int sec_damping, int bsize, int max";
|
||||
}
|
||||
|
||||
add_proto qw/void copy_8x8_16bit_to_8bit/, "uint8_t *dst, int dstride, const uint16_t *src, int sstride";
|
||||
add_proto qw/void copy_4x4_16bit_to_8bit/, "uint8_t *dst, int dstride, const uint16_t *src, int sstride";
|
||||
add_proto qw/void copy_8x8_16bit_to_16bit/, "uint16_t *dst, int dstride, const uint16_t *src, int sstride";
|
||||
add_proto qw/void copy_4x4_16bit_to_16bit/, "uint16_t *dst, int dstride, const uint16_t *src, int sstride";
|
||||
add_proto qw/void copy_rect8_8bit_to_16bit/, "uint16_t *dst, int dstride, const uint8_t *src, int sstride, int v, int h";
|
||||
add_proto qw/void copy_rect8_16bit_to_16bit/, "uint16_t *dst, int dstride, const uint16_t *src, int sstride, int v, int h";
|
||||
|
||||
# VS compiling for 32 bit targets does not support vector types in
|
||||
# VS compiling for 32 bit targets does not support vector types in
|
||||
# structs as arguments, which makes the v256 type of the intrinsics
|
||||
# hard to support, so optimizations for this target are disabled.
|
||||
if ($opts{config} !~ /libs-x86-win32-vs.*/) {
|
||||
specialize qw/aom_clpf_block_hbd sse2 ssse3 sse4_1 neon/;
|
||||
specialize qw/aom_clpf_hblock_hbd sse2 ssse3 sse4_1 neon/;
|
||||
specialize qw/aom_clpf_block sse2 ssse3 sse4_1 neon/;
|
||||
specialize qw/aom_clpf_hblock sse2 ssse3 sse4_1 neon/;
|
||||
specialize qw/od_dir_find8 sse2 ssse3 sse4_1 neon/;
|
||||
specialize qw/od_filter_dering_direction_4x4 sse2 ssse3 sse4_1 neon/;
|
||||
specialize qw/od_filter_dering_direction_8x8 sse2 ssse3 sse4_1 neon/;
|
||||
if (aom_config("CONFIG_CDEF_SINGLEPASS") eq "yes") {
|
||||
specialize qw/cdef_find_dir sse2 ssse3 sse4_1 avx2 neon/;
|
||||
specialize qw/cdef_filter_block sse2 ssse3 sse4_1 avx2 neon/;
|
||||
specialize qw/copy_rect8_8bit_to_16bit sse2 ssse3 sse4_1 avx2 neon/;
|
||||
specialize qw/copy_rect8_16bit_to_16bit sse2 ssse3 sse4_1 avx2 neon/;
|
||||
} else {
|
||||
specialize qw/cdef_find_dir sse2 ssse3 sse4_1 neon/;
|
||||
specialize qw/aom_clpf_block_hbd sse2 ssse3 sse4_1 neon/;
|
||||
specialize qw/aom_clpf_hblock_hbd sse2 ssse3 sse4_1 neon/;
|
||||
specialize qw/aom_clpf_block sse2 ssse3 sse4_1 neon/;
|
||||
specialize qw/aom_clpf_hblock sse2 ssse3 sse4_1 neon/;
|
||||
specialize qw/cdef_find_dir sse2 ssse3 sse4_1 neon/;
|
||||
specialize qw/cdef_direction_4x4 sse2 ssse3 sse4_1 neon/;
|
||||
specialize qw/cdef_direction_8x8 sse2 ssse3 sse4_1 neon/;
|
||||
|
||||
specialize qw/copy_8x8_16bit_to_8bit sse2 ssse3 sse4_1 neon/;
|
||||
specialize qw/copy_4x4_16bit_to_8bit sse2 ssse3 sse4_1 neon/;
|
||||
specialize qw/copy_8x8_16bit_to_16bit sse2 ssse3 sse4_1 neon/;
|
||||
specialize qw/copy_4x4_16bit_to_16bit sse2 ssse3 sse4_1 neon/;
|
||||
specialize qw/copy_rect8_8bit_to_16bit sse2 ssse3 sse4_1 neon/;
|
||||
specialize qw/copy_rect8_16bit_to_16bit sse2 ssse3 sse4_1 neon/;
|
||||
specialize qw/copy_8x8_16bit_to_8bit sse2 ssse3 sse4_1 neon/;
|
||||
specialize qw/copy_4x4_16bit_to_8bit sse2 ssse3 sse4_1 neon/;
|
||||
specialize qw/copy_8x8_16bit_to_16bit sse2 ssse3 sse4_1 neon/;
|
||||
specialize qw/copy_4x4_16bit_to_16bit sse2 ssse3 sse4_1 neon/;
|
||||
specialize qw/copy_rect8_8bit_to_16bit sse2 ssse3 sse4_1 neon/;
|
||||
specialize qw/copy_rect8_16bit_to_16bit sse2 ssse3 sse4_1 neon/;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
|
@ -607,16 +600,9 @@ if ((aom_config("CONFIG_WARPED_MOTION") eq "yes") ||
|
|||
add_proto qw/void av1_warp_affine/, "const int32_t *mat, const uint8_t *ref, int width, int height, int stride, uint8_t *pred, int p_col, int p_row, int p_width, int p_height, int p_stride, int subsampling_x, int subsampling_y, ConvolveParams *conv_params, int16_t alpha, int16_t beta, int16_t gamma, int16_t delta";
|
||||
specialize qw/av1_warp_affine sse2 ssse3/;
|
||||
|
||||
if (aom_config("CONFIG_CONVOLVE_ROUND") eq "yes") {
|
||||
add_proto qw/void av1_warp_affine_post_round/, "const int32_t *mat, const uint8_t *ref, int width, int height, int stride, uint8_t *pred, int p_col, int p_row, int p_width, int p_height, int p_stride, int subsampling_x, int subsampling_y, ConvolveParams *conv_params, int16_t alpha, int16_t beta, int16_t gamma, int16_t delta";
|
||||
}
|
||||
|
||||
if (aom_config("CONFIG_HIGHBITDEPTH") eq "yes") {
|
||||
add_proto qw/void av1_highbd_warp_affine/, "const int32_t *mat, const uint16_t *ref, int width, int height, int stride, uint16_t *pred, int p_col, int p_row, int p_width, int p_height, int p_stride, int subsampling_x, int subsampling_y, int bd, ConvolveParams *conv_params, int16_t alpha, int16_t beta, int16_t gamma, int16_t delta";
|
||||
specialize qw/av1_highbd_warp_affine ssse3/;
|
||||
if (aom_config("CONFIG_CONVOLVE_ROUND") eq "yes") {
|
||||
add_proto qw/void av1_highbd_warp_affine_post_round/, "const int32_t *mat, const uint16_t *ref, int width, int height, int stride, uint16_t *pred, int p_col, int p_row, int p_width, int p_height, int p_stride, int subsampling_x, int subsampling_y, int bd, ConvolveParams *conv_params, int16_t alpha, int16_t beta, int16_t gamma, int16_t delta";
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
|
@ -632,7 +618,7 @@ if (aom_config("CONFIG_LOOP_RESTORATION") eq "yes") {
|
|||
add_proto qw/void apply_selfguided_restoration/, "uint8_t *dat, int width, int height, int stride, int eps, int *xqd, uint8_t *dst, int dst_stride, int32_t *tmpbuf";
|
||||
specialize qw/apply_selfguided_restoration sse4_1/;
|
||||
|
||||
add_proto qw/void av1_selfguided_restoration/, "uint8_t *dgd, int width, int height, int stride, int32_t *dst, int dst_stride, int r, int eps, int32_t *tmpbuf";
|
||||
add_proto qw/void av1_selfguided_restoration/, "uint8_t *dgd, int width, int height, int stride, int32_t *dst, int dst_stride, int r, int eps";
|
||||
specialize qw/av1_selfguided_restoration sse4_1/;
|
||||
|
||||
add_proto qw/void av1_highpass_filter/, "uint8_t *dgd, int width, int height, int stride, int32_t *dst, int dst_stride, int r, int eps";
|
||||
|
|
@ -642,7 +628,7 @@ if (aom_config("CONFIG_LOOP_RESTORATION") eq "yes") {
|
|||
add_proto qw/void apply_selfguided_restoration_highbd/, "uint16_t *dat, int width, int height, int stride, int bit_depth, int eps, int *xqd, uint16_t *dst, int dst_stride, int32_t *tmpbuf";
|
||||
specialize qw/apply_selfguided_restoration_highbd sse4_1/;
|
||||
|
||||
add_proto qw/void av1_selfguided_restoration_highbd/, "uint16_t *dgd, int width, int height, int stride, int32_t *dst, int dst_stride, int bit_depth, int r, int eps, int32_t *tmpbuf";
|
||||
add_proto qw/void av1_selfguided_restoration_highbd/, "uint16_t *dgd, int width, int height, int stride, int32_t *dst, int dst_stride, int bit_depth, int r, int eps";
|
||||
specialize qw/av1_selfguided_restoration_highbd sse4_1/;
|
||||
|
||||
add_proto qw/void av1_highpass_filter_highbd/, "uint16_t *dgd, int width, int height, int stride, int32_t *dst, int dst_stride, int r, int eps";
|
||||
|
|
@ -653,17 +639,40 @@ if (aom_config("CONFIG_LOOP_RESTORATION") eq "yes") {
|
|||
# CONVOLVE_ROUND/COMPOUND_ROUND functions
|
||||
|
||||
if (aom_config("CONFIG_CONVOLVE_ROUND") eq "yes") {
|
||||
add_proto qw/void av1_convolve_2d/, "const uint8_t *src, int src_stride, CONV_BUF_TYPE *dst, int dst_stride, int w, int h, InterpFilterParams *filter_params_x, InterpFilterParams *filter_params_y, const int subpel_x_q4, const int subpel_y_q4, ConvolveParams *conv_params";
|
||||
specialize qw/av1_convolve_2d sse2/;
|
||||
add_proto qw/void av1_convolve_rounding/, "const int32_t *src, int src_stride, uint8_t *dst, int dst_stride, int w, int h, int bits";
|
||||
specialize qw/av1_convolve_rounding avx2/;
|
||||
add_proto qw/void av1_convolve_2d/, "const uint8_t *src, int src_stride, CONV_BUF_TYPE *dst, int dst_stride, int w, int h, InterpFilterParams *filter_params_x, InterpFilterParams *filter_params_y, const int subpel_x_q4, const int subpel_y_q4, ConvolveParams *conv_params";
|
||||
specialize qw/av1_convolve_2d sse2/;
|
||||
add_proto qw/void av1_convolve_rounding/, "const int32_t *src, int src_stride, uint8_t *dst, int dst_stride, int w, int h, int bits";
|
||||
specialize qw/av1_convolve_rounding avx2/;
|
||||
|
||||
add_proto qw/void av1_convolve_2d_scale/, "const uint8_t *src, int src_stride, CONV_BUF_TYPE *dst, int dst_stride, int w, int h, InterpFilterParams *filter_params_x, InterpFilterParams *filter_params_y, const int subpel_x_qn, const int x_step_qn, const int subpel_y_q4, const int y_step_qn, ConvolveParams *conv_params";
|
||||
if (aom_config("CONFIG_COMPOUND_ROUND") ne "yes") {
|
||||
specialize qw/av1_convolve_2d_scale sse4_1/;
|
||||
}
|
||||
|
||||
if (aom_config("CONFIG_HIGHBITDEPTH") eq "yes") {
|
||||
add_proto qw/void av1_highbd_convolve_2d/, "const uint16_t *src, int src_stride, CONV_BUF_TYPE *dst, int dst_stride, int w, int h, InterpFilterParams *filter_params_x, InterpFilterParams *filter_params_y, const int subpel_x_q4, const int subpel_y_q4, ConvolveParams *conv_params, int bd";
|
||||
specialize qw/av1_highbd_convolve_2d ssse3/;
|
||||
add_proto qw/void av1_highbd_convolve_rounding/, "const int32_t *src, int src_stride, uint8_t *dst, int dst_stride, int w, int h, int bits, int bd";
|
||||
specialize qw/av1_highbd_convolve_rounding avx2/;
|
||||
|
||||
add_proto qw/void av1_highbd_convolve_2d_scale/, "const uint16_t *src, int src_stride, CONV_BUF_TYPE *dst, int dst_stride, int w, int h, InterpFilterParams *filter_params_x, InterpFilterParams *filter_params_y, const int subpel_x_q4, const int x_step_qn, const int subpel_y_q4, const int y_step_qn, ConvolveParams *conv_params, int bd";
|
||||
if (aom_config("CONFIG_COMPOUND_ROUND") ne "yes") {
|
||||
specialize qw/av1_highbd_convolve_2d_scale sse4_1/;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
# INTRA_EDGE functions
|
||||
if (aom_config("CONFIG_INTRA_EDGE") eq "yes") {
|
||||
add_proto qw/void av1_filter_intra_edge/, "uint8_t *p, int sz, int strength";
|
||||
specialize qw/av1_filter_intra_edge sse4_1/;
|
||||
add_proto qw/void av1_upsample_intra_edge/, "uint8_t *p, int sz";
|
||||
specialize qw/av1_upsample_intra_edge sse4_1/;
|
||||
if (aom_config("CONFIG_HIGHBITDEPTH") eq "yes") {
|
||||
add_proto qw/void av1_filter_intra_edge_high/, "uint16_t *p, int sz, int strength";
|
||||
specialize qw/av1_filter_intra_edge_high sse4_1/;
|
||||
add_proto qw/void av1_upsample_intra_edge_high/, "uint16_t *p, int sz, int bd";
|
||||
specialize qw/av1_upsample_intra_edge_high sse4_1/;
|
||||
}
|
||||
}
|
||||
|
||||
1;
|
||||
|
|
|
|||
203
third_party/aom/av1/common/av1_txfm.h
vendored
203
third_party/aom/av1/common/av1_txfm.h
vendored
|
|
@ -17,9 +17,16 @@
|
|||
#include <stdio.h>
|
||||
|
||||
#include "av1/common/enums.h"
|
||||
#include "av1/common/blockd.h"
|
||||
#include "aom/aom_integer.h"
|
||||
#include "aom_dsp/aom_dsp_common.h"
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
|
||||
#define MAX_TXFM_STAGE_NUM 12
|
||||
|
||||
static const int cos_bit_min = 10;
|
||||
static const int cos_bit_max = 16;
|
||||
|
||||
|
|
@ -110,27 +117,6 @@ static INLINE int32_t half_btf(int32_t w0, int32_t in0, int32_t w1, int32_t in1,
|
|||
return round_shift(result_32, bit);
|
||||
}
|
||||
|
||||
static INLINE int get_max_bit(int x) {
|
||||
int max_bit = -1;
|
||||
while (x) {
|
||||
x = x >> 1;
|
||||
max_bit++;
|
||||
}
|
||||
return max_bit;
|
||||
}
|
||||
|
||||
// TODO(angiebird): implement SSE
|
||||
static INLINE void clamp_block(int16_t *block, int block_size_row,
|
||||
int block_size_col, int stride, int low,
|
||||
int high) {
|
||||
int i, j;
|
||||
for (i = 0; i < block_size_row; ++i) {
|
||||
for (j = 0; j < block_size_col; ++j) {
|
||||
block[i * stride + j] = clamp(block[i * stride + j], low, high);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
typedef void (*TxfmFunc)(const int32_t *input, int32_t *output,
|
||||
const int8_t *cos_bit, const int8_t *stage_range);
|
||||
|
||||
|
|
@ -148,6 +134,7 @@ typedef enum TXFM_TYPE {
|
|||
TXFM_TYPE_IDENTITY8,
|
||||
TXFM_TYPE_IDENTITY16,
|
||||
TXFM_TYPE_IDENTITY32,
|
||||
TXFM_TYPE_IDENTITY64,
|
||||
} TXFM_TYPE;
|
||||
|
||||
typedef struct TXFM_1D_CFG {
|
||||
|
|
@ -167,7 +154,7 @@ typedef struct TXFM_2D_FLIP_CFG {
|
|||
const TXFM_1D_CFG *row_cfg;
|
||||
} TXFM_2D_FLIP_CFG;
|
||||
|
||||
static INLINE void set_flip_cfg(int tx_type, TXFM_2D_FLIP_CFG *cfg) {
|
||||
static INLINE void set_flip_cfg(TX_TYPE tx_type, TXFM_2D_FLIP_CFG *cfg) {
|
||||
switch (tx_type) {
|
||||
case DCT_DCT:
|
||||
case ADST_DCT:
|
||||
|
|
@ -209,21 +196,171 @@ static INLINE void set_flip_cfg(int tx_type, TXFM_2D_FLIP_CFG *cfg) {
|
|||
}
|
||||
}
|
||||
|
||||
#if CONFIG_MRC_TX
|
||||
static INLINE void get_mrc_mask(const uint8_t *pred, int pred_stride, int *mask,
|
||||
int mask_stride, int width, int height) {
|
||||
for (int i = 0; i < height; ++i) {
|
||||
for (int j = 0; j < width; ++j)
|
||||
mask[i * mask_stride + j] = pred[i * pred_stride + j] > 100 ? 1 : 0;
|
||||
#if CONFIG_TXMG
|
||||
static INLINE TX_SIZE av1_rotate_tx_size(TX_SIZE tx_size) {
|
||||
switch (tx_size) {
|
||||
#if CONFIG_CHROMA_2X2
|
||||
case TX_2X2: return TX_2X2;
|
||||
#endif
|
||||
case TX_4X4: return TX_4X4;
|
||||
case TX_8X8: return TX_8X8;
|
||||
case TX_16X16: return TX_16X16;
|
||||
case TX_32X32: return TX_32X32;
|
||||
#if CONFIG_TX64X64
|
||||
case TX_64X64: return TX_64X64;
|
||||
case TX_32X64: return TX_64X32;
|
||||
case TX_64X32: return TX_32X64;
|
||||
#endif
|
||||
case TX_4X8: return TX_8X4;
|
||||
case TX_8X4: return TX_4X8;
|
||||
case TX_8X16: return TX_16X8;
|
||||
case TX_16X8: return TX_8X16;
|
||||
case TX_16X32: return TX_32X16;
|
||||
case TX_32X16: return TX_16X32;
|
||||
case TX_4X16: return TX_16X4;
|
||||
case TX_16X4: return TX_4X16;
|
||||
case TX_8X32: return TX_32X8;
|
||||
case TX_32X8: return TX_8X32;
|
||||
default: assert(0); return TX_INVALID;
|
||||
}
|
||||
}
|
||||
|
||||
static INLINE TX_TYPE av1_rotate_tx_type(TX_TYPE tx_type) {
|
||||
switch (tx_type) {
|
||||
case DCT_DCT: return DCT_DCT;
|
||||
case ADST_DCT: return DCT_ADST;
|
||||
case DCT_ADST: return ADST_DCT;
|
||||
case ADST_ADST: return ADST_ADST;
|
||||
#if CONFIG_EXT_TX
|
||||
case FLIPADST_DCT: return DCT_FLIPADST;
|
||||
case DCT_FLIPADST: return FLIPADST_DCT;
|
||||
case FLIPADST_FLIPADST: return FLIPADST_FLIPADST;
|
||||
case ADST_FLIPADST: return FLIPADST_ADST;
|
||||
case FLIPADST_ADST: return ADST_FLIPADST;
|
||||
case IDTX: return IDTX;
|
||||
case V_DCT: return H_DCT;
|
||||
case H_DCT: return V_DCT;
|
||||
case V_ADST: return H_ADST;
|
||||
case H_ADST: return V_ADST;
|
||||
case V_FLIPADST: return H_FLIPADST;
|
||||
case H_FLIPADST: return V_FLIPADST;
|
||||
#endif // CONFIG_EXT_TX
|
||||
#if CONFIG_MRC_TX
|
||||
case MRC_DCT: return MRC_DCT;
|
||||
#endif // CONFIG_MRC_TX
|
||||
default: assert(0); return TX_TYPES;
|
||||
}
|
||||
}
|
||||
#endif // CONFIG_TXMG
|
||||
|
||||
#if CONFIG_MRC_TX
|
||||
static INLINE int get_mrc_diff_mask_inter(const int16_t *diff, int diff_stride,
|
||||
uint8_t *mask, int mask_stride,
|
||||
int width, int height) {
|
||||
// placeholder mask generation function
|
||||
assert(SIGNAL_MRC_MASK_INTER);
|
||||
int n_masked_vals = 0;
|
||||
for (int i = 0; i < height; ++i) {
|
||||
for (int j = 0; j < width; ++j) {
|
||||
mask[i * mask_stride + j] = diff[i * diff_stride + j] > 100 ? 1 : 0;
|
||||
n_masked_vals += mask[i * mask_stride + j];
|
||||
}
|
||||
}
|
||||
return n_masked_vals;
|
||||
}
|
||||
|
||||
static INLINE int get_mrc_pred_mask_inter(const uint8_t *pred, int pred_stride,
|
||||
uint8_t *mask, int mask_stride,
|
||||
int width, int height) {
|
||||
// placeholder mask generation function
|
||||
int n_masked_vals = 0;
|
||||
for (int i = 0; i < height; ++i) {
|
||||
for (int j = 0; j < width; ++j) {
|
||||
mask[i * mask_stride + j] = pred[i * pred_stride + j] > 100 ? 1 : 0;
|
||||
n_masked_vals += mask[i * mask_stride + j];
|
||||
}
|
||||
}
|
||||
return n_masked_vals;
|
||||
}
|
||||
|
||||
static INLINE int get_mrc_diff_mask_intra(const int16_t *diff, int diff_stride,
|
||||
uint8_t *mask, int mask_stride,
|
||||
int width, int height) {
|
||||
// placeholder mask generation function
|
||||
assert(SIGNAL_MRC_MASK_INTRA);
|
||||
int n_masked_vals = 0;
|
||||
for (int i = 0; i < height; ++i) {
|
||||
for (int j = 0; j < width; ++j) {
|
||||
mask[i * mask_stride + j] = diff[i * diff_stride + j] > 100 ? 1 : 0;
|
||||
n_masked_vals += mask[i * mask_stride + j];
|
||||
}
|
||||
}
|
||||
return n_masked_vals;
|
||||
}
|
||||
|
||||
static INLINE int get_mrc_pred_mask_intra(const uint8_t *pred, int pred_stride,
|
||||
uint8_t *mask, int mask_stride,
|
||||
int width, int height) {
|
||||
// placeholder mask generation function
|
||||
int n_masked_vals = 0;
|
||||
for (int i = 0; i < height; ++i) {
|
||||
for (int j = 0; j < width; ++j) {
|
||||
mask[i * mask_stride + j] = pred[i * pred_stride + j] > 100 ? 1 : 0;
|
||||
n_masked_vals += mask[i * mask_stride + j];
|
||||
}
|
||||
}
|
||||
return n_masked_vals;
|
||||
}
|
||||
|
||||
static INLINE int get_mrc_diff_mask(const int16_t *diff, int diff_stride,
|
||||
uint8_t *mask, int mask_stride, int width,
|
||||
int height, int is_inter) {
|
||||
if (is_inter) {
|
||||
assert(USE_MRC_INTER && "MRC invalid for inter blocks");
|
||||
assert(SIGNAL_MRC_MASK_INTER);
|
||||
return get_mrc_diff_mask_inter(diff, diff_stride, mask, mask_stride, width,
|
||||
height);
|
||||
} else {
|
||||
assert(USE_MRC_INTRA && "MRC invalid for intra blocks");
|
||||
assert(SIGNAL_MRC_MASK_INTRA);
|
||||
return get_mrc_diff_mask_intra(diff, diff_stride, mask, mask_stride, width,
|
||||
height);
|
||||
}
|
||||
}
|
||||
|
||||
static INLINE int get_mrc_pred_mask(const uint8_t *pred, int pred_stride,
|
||||
uint8_t *mask, int mask_stride, int width,
|
||||
int height, int is_inter) {
|
||||
if (is_inter) {
|
||||
assert(USE_MRC_INTER && "MRC invalid for inter blocks");
|
||||
return get_mrc_pred_mask_inter(pred, pred_stride, mask, mask_stride, width,
|
||||
height);
|
||||
} else {
|
||||
assert(USE_MRC_INTRA && "MRC invalid for intra blocks");
|
||||
return get_mrc_pred_mask_intra(pred, pred_stride, mask, mask_stride, width,
|
||||
height);
|
||||
}
|
||||
}
|
||||
|
||||
static INLINE int is_valid_mrc_mask(int n_masked_vals, int width, int height) {
|
||||
return !(n_masked_vals == 0 || n_masked_vals == (width * height));
|
||||
}
|
||||
#endif // CONFIG_MRC_TX
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
TXFM_2D_FLIP_CFG av1_get_fwd_txfm_cfg(int tx_type, int tx_size);
|
||||
TXFM_2D_FLIP_CFG av1_get_fwd_txfm_64x64_cfg(int tx_type);
|
||||
void av1_gen_fwd_stage_range(int8_t *stage_range_col, int8_t *stage_range_row,
|
||||
const TXFM_2D_FLIP_CFG *cfg, int bd);
|
||||
|
||||
void av1_gen_inv_stage_range(int8_t *stage_range_col, int8_t *stage_range_row,
|
||||
const TXFM_2D_FLIP_CFG *cfg, int8_t fwd_shift,
|
||||
int bd);
|
||||
|
||||
TXFM_2D_FLIP_CFG av1_get_fwd_txfm_cfg(TX_TYPE tx_type, TX_SIZE tx_size);
|
||||
#if CONFIG_TX64X64
|
||||
TXFM_2D_FLIP_CFG av1_get_fwd_txfm_64x64_cfg(TX_TYPE tx_type);
|
||||
TXFM_2D_FLIP_CFG av1_get_fwd_txfm_64x32_cfg(TX_TYPE tx_type);
|
||||
TXFM_2D_FLIP_CFG av1_get_fwd_txfm_32x64_cfg(TX_TYPE tx_type);
|
||||
#endif // CONFIG_TX64X64
|
||||
TXFM_2D_FLIP_CFG av1_get_inv_txfm_cfg(TX_TYPE tx_type, TX_SIZE tx_size);
|
||||
#ifdef __cplusplus
|
||||
}
|
||||
#endif // __cplusplus
|
||||
|
|
|
|||
788
third_party/aom/av1/common/blockd.h
vendored
788
third_party/aom/av1/common/blockd.h
vendored
File diff suppressed because it is too large
Load diff
349
third_party/aom/av1/common/cdef.c
vendored
349
third_party/aom/av1/common/cdef.c
vendored
|
|
@ -16,7 +16,7 @@
|
|||
#include "./aom_scale_rtcd.h"
|
||||
#include "aom/aom_integer.h"
|
||||
#include "av1/common/cdef.h"
|
||||
#include "av1/common/od_dering.h"
|
||||
#include "av1/common/cdef_block.h"
|
||||
#include "av1/common/onyxc_int.h"
|
||||
#include "av1/common/reconinter.h"
|
||||
|
||||
|
|
@ -50,8 +50,8 @@ static int is_8x8_block_skip(MODE_INFO **grid, int mi_row, int mi_col,
|
|||
return is_skip;
|
||||
}
|
||||
|
||||
int sb_compute_dering_list(const AV1_COMMON *const cm, int mi_row, int mi_col,
|
||||
dering_list *dlist, int filter_skip) {
|
||||
int sb_compute_cdef_list(const AV1_COMMON *const cm, int mi_row, int mi_col,
|
||||
cdef_list *dlist, int filter_skip) {
|
||||
int r, c;
|
||||
int maxc, maxr;
|
||||
MODE_INFO **grid;
|
||||
|
|
@ -156,82 +156,82 @@ static INLINE void copy_rect(uint16_t *dst, int dstride, const uint16_t *src,
|
|||
|
||||
void av1_cdef_frame(YV12_BUFFER_CONFIG *frame, AV1_COMMON *cm,
|
||||
MACROBLOCKD *xd) {
|
||||
int sbr, sbc;
|
||||
int nhsb, nvsb;
|
||||
uint16_t src[OD_DERING_INBUF_SIZE];
|
||||
int fbr, fbc;
|
||||
int nhfb, nvfb;
|
||||
uint16_t src[CDEF_INBUF_SIZE];
|
||||
uint16_t *linebuf[3];
|
||||
uint16_t *colbuf[3];
|
||||
dering_list dlist[MI_SIZE_64X64 * MI_SIZE_64X64];
|
||||
unsigned char *row_dering, *prev_row_dering, *curr_row_dering;
|
||||
int dering_count;
|
||||
int dir[OD_DERING_NBLOCKS][OD_DERING_NBLOCKS] = { { 0 } };
|
||||
int var[OD_DERING_NBLOCKS][OD_DERING_NBLOCKS] = { { 0 } };
|
||||
cdef_list dlist[MI_SIZE_64X64 * MI_SIZE_64X64];
|
||||
unsigned char *row_cdef, *prev_row_cdef, *curr_row_cdef;
|
||||
int cdef_count;
|
||||
int dir[CDEF_NBLOCKS][CDEF_NBLOCKS] = { { 0 } };
|
||||
int var[CDEF_NBLOCKS][CDEF_NBLOCKS] = { { 0 } };
|
||||
int stride;
|
||||
int mi_wide_l2[3];
|
||||
int mi_high_l2[3];
|
||||
int xdec[3];
|
||||
int ydec[3];
|
||||
int pli;
|
||||
int dering_left;
|
||||
int cdef_left;
|
||||
int coeff_shift = AOMMAX(cm->bit_depth - 8, 0);
|
||||
int nplanes = 3;
|
||||
int chroma_dering =
|
||||
xd->plane[1].subsampling_x == xd->plane[1].subsampling_y &&
|
||||
xd->plane[2].subsampling_x == xd->plane[2].subsampling_y;
|
||||
nvsb = (cm->mi_rows + MI_SIZE_64X64 - 1) / MI_SIZE_64X64;
|
||||
nhsb = (cm->mi_cols + MI_SIZE_64X64 - 1) / MI_SIZE_64X64;
|
||||
int nplanes = MAX_MB_PLANE;
|
||||
int chroma_cdef = xd->plane[1].subsampling_x == xd->plane[1].subsampling_y &&
|
||||
xd->plane[2].subsampling_x == xd->plane[2].subsampling_y;
|
||||
nvfb = (cm->mi_rows + MI_SIZE_64X64 - 1) / MI_SIZE_64X64;
|
||||
nhfb = (cm->mi_cols + MI_SIZE_64X64 - 1) / MI_SIZE_64X64;
|
||||
av1_setup_dst_planes(xd->plane, cm->sb_size, frame, 0, 0);
|
||||
row_dering = aom_malloc(sizeof(*row_dering) * (nhsb + 2) * 2);
|
||||
memset(row_dering, 1, sizeof(*row_dering) * (nhsb + 2) * 2);
|
||||
prev_row_dering = row_dering + 1;
|
||||
curr_row_dering = prev_row_dering + nhsb + 2;
|
||||
row_cdef = aom_malloc(sizeof(*row_cdef) * (nhfb + 2) * 2);
|
||||
memset(row_cdef, 1, sizeof(*row_cdef) * (nhfb + 2) * 2);
|
||||
prev_row_cdef = row_cdef + 1;
|
||||
curr_row_cdef = prev_row_cdef + nhfb + 2;
|
||||
for (pli = 0; pli < nplanes; pli++) {
|
||||
xdec[pli] = xd->plane[pli].subsampling_x;
|
||||
ydec[pli] = xd->plane[pli].subsampling_y;
|
||||
mi_wide_l2[pli] = MI_SIZE_LOG2 - xd->plane[pli].subsampling_x;
|
||||
mi_high_l2[pli] = MI_SIZE_LOG2 - xd->plane[pli].subsampling_y;
|
||||
if (xdec[pli] != ydec[pli]) nplanes = 1;
|
||||
}
|
||||
stride = (cm->mi_cols << MI_SIZE_LOG2) + 2 * OD_FILT_HBORDER;
|
||||
stride = (cm->mi_cols << MI_SIZE_LOG2) + 2 * CDEF_HBORDER;
|
||||
for (pli = 0; pli < nplanes; pli++) {
|
||||
linebuf[pli] = aom_malloc(sizeof(*linebuf) * OD_FILT_VBORDER * stride);
|
||||
linebuf[pli] = aom_malloc(sizeof(*linebuf) * CDEF_VBORDER * stride);
|
||||
colbuf[pli] =
|
||||
aom_malloc(sizeof(*colbuf) *
|
||||
((MAX_SB_SIZE << mi_high_l2[pli]) + 2 * OD_FILT_VBORDER) *
|
||||
OD_FILT_HBORDER);
|
||||
((CDEF_BLOCKSIZE << mi_high_l2[pli]) + 2 * CDEF_VBORDER) *
|
||||
CDEF_HBORDER);
|
||||
}
|
||||
for (sbr = 0; sbr < nvsb; sbr++) {
|
||||
for (fbr = 0; fbr < nvfb; fbr++) {
|
||||
for (pli = 0; pli < nplanes; pli++) {
|
||||
const int block_height =
|
||||
(MI_SIZE_64X64 << mi_high_l2[pli]) + 2 * OD_FILT_VBORDER;
|
||||
fill_rect(colbuf[pli], OD_FILT_HBORDER, block_height, OD_FILT_HBORDER,
|
||||
OD_DERING_VERY_LARGE);
|
||||
(MI_SIZE_64X64 << mi_high_l2[pli]) + 2 * CDEF_VBORDER;
|
||||
fill_rect(colbuf[pli], CDEF_HBORDER, block_height, CDEF_HBORDER,
|
||||
CDEF_VERY_LARGE);
|
||||
}
|
||||
dering_left = 1;
|
||||
for (sbc = 0; sbc < nhsb; sbc++) {
|
||||
int level, clpf_strength;
|
||||
int uv_level, uv_clpf_strength;
|
||||
cdef_left = 1;
|
||||
for (fbc = 0; fbc < nhfb; fbc++) {
|
||||
int level, sec_strength;
|
||||
int uv_level, uv_sec_strength;
|
||||
int nhb, nvb;
|
||||
int cstart = 0;
|
||||
curr_row_dering[sbc] = 0;
|
||||
if (cm->mi_grid_visible[MI_SIZE_64X64 * sbr * cm->mi_stride +
|
||||
MI_SIZE_64X64 * sbc] == NULL ||
|
||||
cm->mi_grid_visible[MI_SIZE_64X64 * sbr * cm->mi_stride +
|
||||
MI_SIZE_64X64 * sbc]
|
||||
curr_row_cdef[fbc] = 0;
|
||||
if (cm->mi_grid_visible[MI_SIZE_64X64 * fbr * cm->mi_stride +
|
||||
MI_SIZE_64X64 * fbc] == NULL ||
|
||||
cm->mi_grid_visible[MI_SIZE_64X64 * fbr * cm->mi_stride +
|
||||
MI_SIZE_64X64 * fbc]
|
||||
->mbmi.cdef_strength == -1) {
|
||||
dering_left = 0;
|
||||
cdef_left = 0;
|
||||
continue;
|
||||
}
|
||||
if (!dering_left) cstart = -OD_FILT_HBORDER;
|
||||
nhb = AOMMIN(MI_SIZE_64X64, cm->mi_cols - MI_SIZE_64X64 * sbc);
|
||||
nvb = AOMMIN(MI_SIZE_64X64, cm->mi_rows - MI_SIZE_64X64 * sbr);
|
||||
if (!cdef_left) cstart = -CDEF_HBORDER;
|
||||
nhb = AOMMIN(MI_SIZE_64X64, cm->mi_cols - MI_SIZE_64X64 * fbc);
|
||||
nvb = AOMMIN(MI_SIZE_64X64, cm->mi_rows - MI_SIZE_64X64 * fbr);
|
||||
int tile_top, tile_left, tile_bottom, tile_right;
|
||||
int mi_idx = MI_SIZE_64X64 * sbr * cm->mi_stride + MI_SIZE_64X64 * sbc;
|
||||
int mi_idx = MI_SIZE_64X64 * fbr * cm->mi_stride + MI_SIZE_64X64 * fbc;
|
||||
MODE_INFO *const mi_tl = cm->mi + mi_idx;
|
||||
BOUNDARY_TYPE boundary_tl = mi_tl->mbmi.boundary_info;
|
||||
tile_top = boundary_tl & TILE_ABOVE_BOUNDARY;
|
||||
tile_left = boundary_tl & TILE_LEFT_BOUNDARY;
|
||||
|
||||
if (sbr != nvsb - 1 &&
|
||||
if (fbr != nvfb - 1 &&
|
||||
(&cm->mi[mi_idx + (MI_SIZE_64X64 - 1) * cm->mi_stride]))
|
||||
tile_bottom = cm->mi[mi_idx + (MI_SIZE_64X64 - 1) * cm->mi_stride]
|
||||
.mbmi.boundary_info &
|
||||
|
|
@ -239,197 +239,216 @@ void av1_cdef_frame(YV12_BUFFER_CONFIG *frame, AV1_COMMON *cm,
|
|||
else
|
||||
tile_bottom = 1;
|
||||
|
||||
if (sbc != nhsb - 1 && (&cm->mi[mi_idx + MI_SIZE_64X64 - 1]))
|
||||
if (fbc != nhfb - 1 && (&cm->mi[mi_idx + MI_SIZE_64X64 - 1]))
|
||||
tile_right = cm->mi[mi_idx + MI_SIZE_64X64 - 1].mbmi.boundary_info &
|
||||
TILE_RIGHT_BOUNDARY;
|
||||
else
|
||||
tile_right = 1;
|
||||
|
||||
const int mbmi_cdef_strength =
|
||||
cm->mi_grid_visible[MI_SIZE_64X64 * sbr * cm->mi_stride +
|
||||
MI_SIZE_64X64 * sbc]
|
||||
cm->mi_grid_visible[MI_SIZE_64X64 * fbr * cm->mi_stride +
|
||||
MI_SIZE_64X64 * fbc]
|
||||
->mbmi.cdef_strength;
|
||||
level = cm->cdef_strengths[mbmi_cdef_strength] / CLPF_STRENGTHS;
|
||||
clpf_strength = cm->cdef_strengths[mbmi_cdef_strength] % CLPF_STRENGTHS;
|
||||
clpf_strength += clpf_strength == 3;
|
||||
uv_level = cm->cdef_uv_strengths[mbmi_cdef_strength] / CLPF_STRENGTHS;
|
||||
uv_clpf_strength =
|
||||
cm->cdef_uv_strengths[mbmi_cdef_strength] % CLPF_STRENGTHS;
|
||||
uv_clpf_strength += uv_clpf_strength == 3;
|
||||
if ((level == 0 && clpf_strength == 0 && uv_level == 0 &&
|
||||
uv_clpf_strength == 0) ||
|
||||
(dering_count = sb_compute_dering_list(
|
||||
cm, sbr * MI_SIZE_64X64, sbc * MI_SIZE_64X64, dlist,
|
||||
get_filter_skip(level) || get_filter_skip(uv_level))) == 0) {
|
||||
dering_left = 0;
|
||||
level = cm->cdef_strengths[mbmi_cdef_strength] / CDEF_SEC_STRENGTHS;
|
||||
sec_strength =
|
||||
cm->cdef_strengths[mbmi_cdef_strength] % CDEF_SEC_STRENGTHS;
|
||||
sec_strength += sec_strength == 3;
|
||||
uv_level = cm->cdef_uv_strengths[mbmi_cdef_strength] / CDEF_SEC_STRENGTHS;
|
||||
uv_sec_strength =
|
||||
cm->cdef_uv_strengths[mbmi_cdef_strength] % CDEF_SEC_STRENGTHS;
|
||||
uv_sec_strength += uv_sec_strength == 3;
|
||||
if ((level == 0 && sec_strength == 0 && uv_level == 0 &&
|
||||
uv_sec_strength == 0) ||
|
||||
(cdef_count = sb_compute_cdef_list(
|
||||
cm, fbr * MI_SIZE_64X64, fbc * MI_SIZE_64X64, dlist,
|
||||
#if CONFIG_CDEF_SINGLEPASS
|
||||
(level & 1) || (uv_level & 1))) == 0)
|
||||
#else
|
||||
get_filter_skip(level) || get_filter_skip(uv_level))) == 0)
|
||||
#endif
|
||||
{
|
||||
cdef_left = 0;
|
||||
continue;
|
||||
}
|
||||
|
||||
curr_row_dering[sbc] = 1;
|
||||
curr_row_cdef[fbc] = 1;
|
||||
for (pli = 0; pli < nplanes; pli++) {
|
||||
uint16_t dst[MAX_SB_SIZE * MAX_SB_SIZE];
|
||||
#if !CONFIG_CDEF_SINGLEPASS
|
||||
uint16_t dst[CDEF_BLOCKSIZE * CDEF_BLOCKSIZE];
|
||||
#endif
|
||||
int coffset;
|
||||
int rend, cend;
|
||||
int clpf_damping = cm->cdef_clpf_damping;
|
||||
int dering_damping = cm->cdef_dering_damping;
|
||||
int pri_damping = cm->cdef_pri_damping;
|
||||
int sec_damping = cm->cdef_sec_damping;
|
||||
int hsize = nhb << mi_wide_l2[pli];
|
||||
int vsize = nvb << mi_high_l2[pli];
|
||||
|
||||
if (pli) {
|
||||
if (chroma_dering)
|
||||
if (chroma_cdef)
|
||||
level = uv_level;
|
||||
else
|
||||
level = 0;
|
||||
clpf_strength = uv_clpf_strength;
|
||||
sec_strength = uv_sec_strength;
|
||||
}
|
||||
|
||||
if (sbc == nhsb - 1)
|
||||
if (fbc == nhfb - 1)
|
||||
cend = hsize;
|
||||
else
|
||||
cend = hsize + OD_FILT_HBORDER;
|
||||
cend = hsize + CDEF_HBORDER;
|
||||
|
||||
if (sbr == nvsb - 1)
|
||||
if (fbr == nvfb - 1)
|
||||
rend = vsize;
|
||||
else
|
||||
rend = vsize + OD_FILT_VBORDER;
|
||||
rend = vsize + CDEF_VBORDER;
|
||||
|
||||
coffset = sbc * MI_SIZE_64X64 << mi_wide_l2[pli];
|
||||
if (sbc == nhsb - 1) {
|
||||
coffset = fbc * MI_SIZE_64X64 << mi_wide_l2[pli];
|
||||
if (fbc == nhfb - 1) {
|
||||
/* On the last superblock column, fill in the right border with
|
||||
OD_DERING_VERY_LARGE to avoid filtering with the outside. */
|
||||
fill_rect(&src[cend + OD_FILT_HBORDER], OD_FILT_BSTRIDE,
|
||||
rend + OD_FILT_VBORDER, hsize + OD_FILT_HBORDER - cend,
|
||||
OD_DERING_VERY_LARGE);
|
||||
CDEF_VERY_LARGE to avoid filtering with the outside. */
|
||||
fill_rect(&src[cend + CDEF_HBORDER], CDEF_BSTRIDE,
|
||||
rend + CDEF_VBORDER, hsize + CDEF_HBORDER - cend,
|
||||
CDEF_VERY_LARGE);
|
||||
}
|
||||
if (sbr == nvsb - 1) {
|
||||
if (fbr == nvfb - 1) {
|
||||
/* On the last superblock row, fill in the bottom border with
|
||||
OD_DERING_VERY_LARGE to avoid filtering with the outside. */
|
||||
fill_rect(&src[(rend + OD_FILT_VBORDER) * OD_FILT_BSTRIDE],
|
||||
OD_FILT_BSTRIDE, OD_FILT_VBORDER,
|
||||
hsize + 2 * OD_FILT_HBORDER, OD_DERING_VERY_LARGE);
|
||||
CDEF_VERY_LARGE to avoid filtering with the outside. */
|
||||
fill_rect(&src[(rend + CDEF_VBORDER) * CDEF_BSTRIDE], CDEF_BSTRIDE,
|
||||
CDEF_VBORDER, hsize + 2 * CDEF_HBORDER, CDEF_VERY_LARGE);
|
||||
}
|
||||
/* Copy in the pixels we need from the current superblock for
|
||||
deringing.*/
|
||||
copy_sb8_16(
|
||||
cm,
|
||||
&src[OD_FILT_VBORDER * OD_FILT_BSTRIDE + OD_FILT_HBORDER + cstart],
|
||||
OD_FILT_BSTRIDE, xd->plane[pli].dst.buf,
|
||||
(MI_SIZE_64X64 << mi_high_l2[pli]) * sbr, coffset + cstart,
|
||||
xd->plane[pli].dst.stride, rend, cend - cstart);
|
||||
if (!prev_row_dering[sbc]) {
|
||||
copy_sb8_16(
|
||||
cm, &src[OD_FILT_HBORDER], OD_FILT_BSTRIDE,
|
||||
xd->plane[pli].dst.buf,
|
||||
(MI_SIZE_64X64 << mi_high_l2[pli]) * sbr - OD_FILT_VBORDER,
|
||||
coffset, xd->plane[pli].dst.stride, OD_FILT_VBORDER, hsize);
|
||||
} else if (sbr > 0) {
|
||||
copy_rect(&src[OD_FILT_HBORDER], OD_FILT_BSTRIDE,
|
||||
&linebuf[pli][coffset], stride, OD_FILT_VBORDER, hsize);
|
||||
copy_sb8_16(cm,
|
||||
&src[CDEF_VBORDER * CDEF_BSTRIDE + CDEF_HBORDER + cstart],
|
||||
CDEF_BSTRIDE, xd->plane[pli].dst.buf,
|
||||
(MI_SIZE_64X64 << mi_high_l2[pli]) * fbr, coffset + cstart,
|
||||
xd->plane[pli].dst.stride, rend, cend - cstart);
|
||||
if (!prev_row_cdef[fbc]) {
|
||||
copy_sb8_16(cm, &src[CDEF_HBORDER], CDEF_BSTRIDE,
|
||||
xd->plane[pli].dst.buf,
|
||||
(MI_SIZE_64X64 << mi_high_l2[pli]) * fbr - CDEF_VBORDER,
|
||||
coffset, xd->plane[pli].dst.stride, CDEF_VBORDER, hsize);
|
||||
} else if (fbr > 0) {
|
||||
copy_rect(&src[CDEF_HBORDER], CDEF_BSTRIDE, &linebuf[pli][coffset],
|
||||
stride, CDEF_VBORDER, hsize);
|
||||
} else {
|
||||
fill_rect(&src[OD_FILT_HBORDER], OD_FILT_BSTRIDE, OD_FILT_VBORDER,
|
||||
hsize, OD_DERING_VERY_LARGE);
|
||||
fill_rect(&src[CDEF_HBORDER], CDEF_BSTRIDE, CDEF_VBORDER, hsize,
|
||||
CDEF_VERY_LARGE);
|
||||
}
|
||||
if (!prev_row_dering[sbc - 1]) {
|
||||
copy_sb8_16(
|
||||
cm, src, OD_FILT_BSTRIDE, xd->plane[pli].dst.buf,
|
||||
(MI_SIZE_64X64 << mi_high_l2[pli]) * sbr - OD_FILT_VBORDER,
|
||||
coffset - OD_FILT_HBORDER, xd->plane[pli].dst.stride,
|
||||
OD_FILT_VBORDER, OD_FILT_HBORDER);
|
||||
} else if (sbr > 0 && sbc > 0) {
|
||||
copy_rect(src, OD_FILT_BSTRIDE,
|
||||
&linebuf[pli][coffset - OD_FILT_HBORDER], stride,
|
||||
OD_FILT_VBORDER, OD_FILT_HBORDER);
|
||||
if (!prev_row_cdef[fbc - 1]) {
|
||||
copy_sb8_16(cm, src, CDEF_BSTRIDE, xd->plane[pli].dst.buf,
|
||||
(MI_SIZE_64X64 << mi_high_l2[pli]) * fbr - CDEF_VBORDER,
|
||||
coffset - CDEF_HBORDER, xd->plane[pli].dst.stride,
|
||||
CDEF_VBORDER, CDEF_HBORDER);
|
||||
} else if (fbr > 0 && fbc > 0) {
|
||||
copy_rect(src, CDEF_BSTRIDE, &linebuf[pli][coffset - CDEF_HBORDER],
|
||||
stride, CDEF_VBORDER, CDEF_HBORDER);
|
||||
} else {
|
||||
fill_rect(src, OD_FILT_BSTRIDE, OD_FILT_VBORDER, OD_FILT_HBORDER,
|
||||
OD_DERING_VERY_LARGE);
|
||||
fill_rect(src, CDEF_BSTRIDE, CDEF_VBORDER, CDEF_HBORDER,
|
||||
CDEF_VERY_LARGE);
|
||||
}
|
||||
if (!prev_row_dering[sbc + 1]) {
|
||||
copy_sb8_16(
|
||||
cm, &src[OD_FILT_HBORDER + (nhb << mi_wide_l2[pli])],
|
||||
OD_FILT_BSTRIDE, xd->plane[pli].dst.buf,
|
||||
(MI_SIZE_64X64 << mi_high_l2[pli]) * sbr - OD_FILT_VBORDER,
|
||||
coffset + hsize, xd->plane[pli].dst.stride, OD_FILT_VBORDER,
|
||||
OD_FILT_HBORDER);
|
||||
} else if (sbr > 0 && sbc < nhsb - 1) {
|
||||
copy_rect(&src[hsize + OD_FILT_HBORDER], OD_FILT_BSTRIDE,
|
||||
&linebuf[pli][coffset + hsize], stride, OD_FILT_VBORDER,
|
||||
OD_FILT_HBORDER);
|
||||
if (!prev_row_cdef[fbc + 1]) {
|
||||
copy_sb8_16(cm, &src[CDEF_HBORDER + (nhb << mi_wide_l2[pli])],
|
||||
CDEF_BSTRIDE, xd->plane[pli].dst.buf,
|
||||
(MI_SIZE_64X64 << mi_high_l2[pli]) * fbr - CDEF_VBORDER,
|
||||
coffset + hsize, xd->plane[pli].dst.stride, CDEF_VBORDER,
|
||||
CDEF_HBORDER);
|
||||
} else if (fbr > 0 && fbc < nhfb - 1) {
|
||||
copy_rect(&src[hsize + CDEF_HBORDER], CDEF_BSTRIDE,
|
||||
&linebuf[pli][coffset + hsize], stride, CDEF_VBORDER,
|
||||
CDEF_HBORDER);
|
||||
} else {
|
||||
fill_rect(&src[hsize + OD_FILT_HBORDER], OD_FILT_BSTRIDE,
|
||||
OD_FILT_VBORDER, OD_FILT_HBORDER, OD_DERING_VERY_LARGE);
|
||||
fill_rect(&src[hsize + CDEF_HBORDER], CDEF_BSTRIDE, CDEF_VBORDER,
|
||||
CDEF_HBORDER, CDEF_VERY_LARGE);
|
||||
}
|
||||
if (dering_left) {
|
||||
if (cdef_left) {
|
||||
/* If we deringed the superblock on the left then we need to copy in
|
||||
saved pixels. */
|
||||
copy_rect(src, OD_FILT_BSTRIDE, colbuf[pli], OD_FILT_HBORDER,
|
||||
rend + OD_FILT_VBORDER, OD_FILT_HBORDER);
|
||||
copy_rect(src, CDEF_BSTRIDE, colbuf[pli], CDEF_HBORDER,
|
||||
rend + CDEF_VBORDER, CDEF_HBORDER);
|
||||
}
|
||||
/* Saving pixels in case we need to dering the superblock on the
|
||||
right. */
|
||||
copy_rect(colbuf[pli], OD_FILT_HBORDER, src + hsize, OD_FILT_BSTRIDE,
|
||||
rend + OD_FILT_VBORDER, OD_FILT_HBORDER);
|
||||
copy_rect(colbuf[pli], CDEF_HBORDER, src + hsize, CDEF_BSTRIDE,
|
||||
rend + CDEF_VBORDER, CDEF_HBORDER);
|
||||
copy_sb8_16(
|
||||
cm, &linebuf[pli][coffset], stride, xd->plane[pli].dst.buf,
|
||||
(MI_SIZE_64X64 << mi_high_l2[pli]) * (sbr + 1) - OD_FILT_VBORDER,
|
||||
coffset, xd->plane[pli].dst.stride, OD_FILT_VBORDER, hsize);
|
||||
(MI_SIZE_64X64 << mi_high_l2[pli]) * (fbr + 1) - CDEF_VBORDER,
|
||||
coffset, xd->plane[pli].dst.stride, CDEF_VBORDER, hsize);
|
||||
|
||||
if (tile_top) {
|
||||
fill_rect(src, OD_FILT_BSTRIDE, OD_FILT_VBORDER,
|
||||
hsize + 2 * OD_FILT_HBORDER, OD_DERING_VERY_LARGE);
|
||||
fill_rect(src, CDEF_BSTRIDE, CDEF_VBORDER, hsize + 2 * CDEF_HBORDER,
|
||||
CDEF_VERY_LARGE);
|
||||
}
|
||||
if (tile_left) {
|
||||
fill_rect(src, OD_FILT_BSTRIDE, vsize + 2 * OD_FILT_VBORDER,
|
||||
OD_FILT_HBORDER, OD_DERING_VERY_LARGE);
|
||||
fill_rect(src, CDEF_BSTRIDE, vsize + 2 * CDEF_VBORDER, CDEF_HBORDER,
|
||||
CDEF_VERY_LARGE);
|
||||
}
|
||||
if (tile_bottom) {
|
||||
fill_rect(&src[(vsize + OD_FILT_VBORDER) * OD_FILT_BSTRIDE],
|
||||
OD_FILT_BSTRIDE, OD_FILT_VBORDER,
|
||||
hsize + 2 * OD_FILT_HBORDER, OD_DERING_VERY_LARGE);
|
||||
fill_rect(&src[(vsize + CDEF_VBORDER) * CDEF_BSTRIDE], CDEF_BSTRIDE,
|
||||
CDEF_VBORDER, hsize + 2 * CDEF_HBORDER, CDEF_VERY_LARGE);
|
||||
}
|
||||
if (tile_right) {
|
||||
fill_rect(&src[hsize + OD_FILT_HBORDER], OD_FILT_BSTRIDE,
|
||||
vsize + 2 * OD_FILT_VBORDER, OD_FILT_HBORDER,
|
||||
OD_DERING_VERY_LARGE);
|
||||
fill_rect(&src[hsize + CDEF_HBORDER], CDEF_BSTRIDE,
|
||||
vsize + 2 * CDEF_VBORDER, CDEF_HBORDER, CDEF_VERY_LARGE);
|
||||
}
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
if (cm->use_highbitdepth) {
|
||||
od_dering(
|
||||
(uint8_t *)&CONVERT_TO_SHORTPTR(
|
||||
xd->plane[pli]
|
||||
.dst.buf)[xd->plane[pli].dst.stride *
|
||||
(MI_SIZE_64X64 * sbr << mi_high_l2[pli]) +
|
||||
(sbc * MI_SIZE_64X64 << mi_wide_l2[pli])],
|
||||
cdef_filter_fb(
|
||||
#if CONFIG_CDEF_SINGLEPASS
|
||||
NULL,
|
||||
&CONVERT_TO_SHORTPTR(xd->plane[pli].dst.buf)
|
||||
#else
|
||||
(uint8_t *)&CONVERT_TO_SHORTPTR(xd->plane[pli].dst.buf)
|
||||
#endif
|
||||
[xd->plane[pli].dst.stride *
|
||||
(MI_SIZE_64X64 * fbr << mi_high_l2[pli]) +
|
||||
(fbc * MI_SIZE_64X64 << mi_wide_l2[pli])],
|
||||
#if CONFIG_CDEF_SINGLEPASS
|
||||
xd->plane[pli].dst.stride,
|
||||
#else
|
||||
xd->plane[pli].dst.stride, dst,
|
||||
&src[OD_FILT_VBORDER * OD_FILT_BSTRIDE + OD_FILT_HBORDER],
|
||||
xdec[pli], ydec[pli], dir, NULL, var, pli, dlist, dering_count,
|
||||
level, clpf_strength, clpf_damping, dering_damping, coeff_shift,
|
||||
0, 1);
|
||||
#endif
|
||||
&src[CDEF_VBORDER * CDEF_BSTRIDE + CDEF_HBORDER], xdec[pli],
|
||||
ydec[pli], dir, NULL, var, pli, dlist, cdef_count, level,
|
||||
#if CONFIG_CDEF_SINGLEPASS
|
||||
sec_strength, pri_damping, sec_damping, coeff_shift);
|
||||
#else
|
||||
sec_strength, sec_damping, pri_damping, coeff_shift, 0, 1);
|
||||
#endif
|
||||
} else {
|
||||
#endif
|
||||
od_dering(&xd->plane[pli]
|
||||
.dst.buf[xd->plane[pli].dst.stride *
|
||||
(MI_SIZE_64X64 * sbr << mi_high_l2[pli]) +
|
||||
(sbc * MI_SIZE_64X64 << mi_wide_l2[pli])],
|
||||
xd->plane[pli].dst.stride, dst,
|
||||
&src[OD_FILT_VBORDER * OD_FILT_BSTRIDE + OD_FILT_HBORDER],
|
||||
xdec[pli], ydec[pli], dir, NULL, var, pli, dlist,
|
||||
dering_count, level, clpf_strength, clpf_damping,
|
||||
dering_damping, coeff_shift, 0, 0);
|
||||
cdef_filter_fb(
|
||||
&xd->plane[pli]
|
||||
.dst.buf[xd->plane[pli].dst.stride *
|
||||
(MI_SIZE_64X64 * fbr << mi_high_l2[pli]) +
|
||||
(fbc * MI_SIZE_64X64 << mi_wide_l2[pli])],
|
||||
#if CONFIG_CDEF_SINGLEPASS
|
||||
NULL, xd->plane[pli].dst.stride,
|
||||
#else
|
||||
xd->plane[pli].dst.stride, dst,
|
||||
#endif
|
||||
&src[CDEF_VBORDER * CDEF_BSTRIDE + CDEF_HBORDER], xdec[pli],
|
||||
ydec[pli], dir, NULL, var, pli, dlist, cdef_count, level,
|
||||
#if CONFIG_CDEF_SINGLEPASS
|
||||
sec_strength, pri_damping, sec_damping, coeff_shift);
|
||||
#else
|
||||
sec_strength, sec_damping, pri_damping, coeff_shift, 0, 0);
|
||||
#endif
|
||||
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
}
|
||||
#endif
|
||||
}
|
||||
dering_left = 1;
|
||||
cdef_left = 1;
|
||||
}
|
||||
{
|
||||
unsigned char *tmp;
|
||||
tmp = prev_row_dering;
|
||||
prev_row_dering = curr_row_dering;
|
||||
curr_row_dering = tmp;
|
||||
tmp = prev_row_cdef;
|
||||
prev_row_cdef = curr_row_cdef;
|
||||
curr_row_cdef = tmp;
|
||||
}
|
||||
}
|
||||
aom_free(row_dering);
|
||||
aom_free(row_cdef);
|
||||
for (pli = 0; pli < nplanes; pli++) {
|
||||
aom_free(linebuf[pli]);
|
||||
aom_free(colbuf[pli]);
|
||||
|
|
|
|||
31
third_party/aom/av1/common/cdef.h
vendored
31
third_party/aom/av1/common/cdef.h
vendored
|
|
@ -8,31 +8,28 @@
|
|||
* Media Patent License 1.0 was not distributed with this source code in the
|
||||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
#ifndef AV1_COMMON_DERING_H_
|
||||
#define AV1_COMMON_DERING_H_
|
||||
#ifndef AV1_COMMON_CDEF_H_
|
||||
#define AV1_COMMON_CDEF_H_
|
||||
|
||||
#define CDEF_STRENGTH_BITS 7
|
||||
|
||||
#define DERING_STRENGTHS 32
|
||||
#define CLPF_STRENGTHS 4
|
||||
#define CDEF_PRI_STRENGTHS 32
|
||||
#define CDEF_SEC_STRENGTHS 4
|
||||
|
||||
#include "./aom_config.h"
|
||||
#include "aom/aom_integer.h"
|
||||
#include "aom_ports/mem.h"
|
||||
#include "av1/common/od_dering.h"
|
||||
#include "av1/common/cdef_block.h"
|
||||
#include "av1/common/onyxc_int.h"
|
||||
#include "./od_dering.h"
|
||||
|
||||
static INLINE int sign(int i) { return i < 0 ? -1 : 1; }
|
||||
|
||||
static INLINE int constrain(int diff, int threshold, unsigned int damping) {
|
||||
return threshold
|
||||
? sign(diff) *
|
||||
AOMMIN(
|
||||
abs(diff),
|
||||
AOMMAX(0, threshold - (abs(diff) >>
|
||||
(damping - get_msb(threshold)))))
|
||||
: 0;
|
||||
static INLINE int constrain(int diff, int threshold, int damping) {
|
||||
if (!threshold) return 0;
|
||||
|
||||
const int shift = AOMMAX(0, damping - get_msb(threshold));
|
||||
return sign(diff) *
|
||||
AOMMIN(abs(diff), AOMMAX(0, threshold - (abs(diff) >> shift)));
|
||||
}
|
||||
|
||||
#ifdef __cplusplus
|
||||
|
|
@ -40,8 +37,8 @@ extern "C" {
|
|||
#endif
|
||||
|
||||
int sb_all_skip(const AV1_COMMON *const cm, int mi_row, int mi_col);
|
||||
int sb_compute_dering_list(const AV1_COMMON *const cm, int mi_row, int mi_col,
|
||||
dering_list *dlist, int filter_skip);
|
||||
int sb_compute_cdef_list(const AV1_COMMON *const cm, int mi_row, int mi_col,
|
||||
cdef_list *dlist, int filter_skip);
|
||||
void av1_cdef_frame(YV12_BUFFER_CONFIG *frame, AV1_COMMON *cm, MACROBLOCKD *xd);
|
||||
|
||||
void av1_cdef_search(YV12_BUFFER_CONFIG *frame, const YV12_BUFFER_CONFIG *ref,
|
||||
|
|
@ -50,4 +47,4 @@ void av1_cdef_search(YV12_BUFFER_CONFIG *frame, const YV12_BUFFER_CONFIG *ref,
|
|||
#ifdef __cplusplus
|
||||
} // extern "C"
|
||||
#endif
|
||||
#endif // AV1_COMMON_DERING_H_
|
||||
#endif // AV1_COMMON_CDEF_H_
|
||||
|
|
|
|||
584
third_party/aom/av1/common/cdef_block.c
vendored
Normal file
584
third_party/aom/av1/common/cdef_block.c
vendored
Normal file
|
|
@ -0,0 +1,584 @@
|
|||
/*
|
||||
* Copyright (c) 2016, Alliance for Open Media. All rights reserved
|
||||
*
|
||||
* This source code is subject to the terms of the BSD 2 Clause License and
|
||||
* the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
|
||||
* was not distributed with this source code in the LICENSE file, you can
|
||||
* obtain it at www.aomedia.org/license/software. If the Alliance for Open
|
||||
* Media Patent License 1.0 was not distributed with this source code in the
|
||||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#include <math.h>
|
||||
#include <stdlib.h>
|
||||
|
||||
#ifdef HAVE_CONFIG_H
|
||||
#include "./config.h"
|
||||
#endif
|
||||
|
||||
#include "./aom_dsp_rtcd.h"
|
||||
#include "./av1_rtcd.h"
|
||||
#include "./cdef.h"
|
||||
|
||||
/* Generated from gen_filter_tables.c. */
|
||||
#if !CONFIG_CDEF_SINGLEPASS || CDEF_FULL
|
||||
const int cdef_directions[8][3] = {
|
||||
{ -1 * CDEF_BSTRIDE + 1, -2 * CDEF_BSTRIDE + 2, -3 * CDEF_BSTRIDE + 3 },
|
||||
{ 0 * CDEF_BSTRIDE + 1, -1 * CDEF_BSTRIDE + 2, -1 * CDEF_BSTRIDE + 3 },
|
||||
{ 0 * CDEF_BSTRIDE + 1, 0 * CDEF_BSTRIDE + 2, 0 * CDEF_BSTRIDE + 3 },
|
||||
{ 0 * CDEF_BSTRIDE + 1, 1 * CDEF_BSTRIDE + 2, 1 * CDEF_BSTRIDE + 3 },
|
||||
{ 1 * CDEF_BSTRIDE + 1, 2 * CDEF_BSTRIDE + 2, 3 * CDEF_BSTRIDE + 3 },
|
||||
{ 1 * CDEF_BSTRIDE + 0, 2 * CDEF_BSTRIDE + 1, 3 * CDEF_BSTRIDE + 1 },
|
||||
{ 1 * CDEF_BSTRIDE + 0, 2 * CDEF_BSTRIDE + 0, 3 * CDEF_BSTRIDE + 0 },
|
||||
{ 1 * CDEF_BSTRIDE + 0, 2 * CDEF_BSTRIDE - 1, 3 * CDEF_BSTRIDE - 1 }
|
||||
};
|
||||
#else
|
||||
const int cdef_directions[8][2] = {
|
||||
{ -1 * CDEF_BSTRIDE + 1, -2 * CDEF_BSTRIDE + 2 },
|
||||
{ 0 * CDEF_BSTRIDE + 1, -1 * CDEF_BSTRIDE + 2 },
|
||||
{ 0 * CDEF_BSTRIDE + 1, 0 * CDEF_BSTRIDE + 2 },
|
||||
{ 0 * CDEF_BSTRIDE + 1, 1 * CDEF_BSTRIDE + 2 },
|
||||
{ 1 * CDEF_BSTRIDE + 1, 2 * CDEF_BSTRIDE + 2 },
|
||||
{ 1 * CDEF_BSTRIDE + 0, 2 * CDEF_BSTRIDE + 1 },
|
||||
{ 1 * CDEF_BSTRIDE + 0, 2 * CDEF_BSTRIDE + 0 },
|
||||
{ 1 * CDEF_BSTRIDE + 0, 2 * CDEF_BSTRIDE - 1 }
|
||||
};
|
||||
#endif
|
||||
|
||||
/* Detect direction. 0 means 45-degree up-right, 2 is horizontal, and so on.
|
||||
The search minimizes the weighted variance along all the lines in a
|
||||
particular direction, i.e. the squared error between the input and a
|
||||
"predicted" block where each pixel is replaced by the average along a line
|
||||
in a particular direction. Since each direction have the same sum(x^2) term,
|
||||
that term is never computed. See Section 2, step 2, of:
|
||||
http://jmvalin.ca/notes/intra_paint.pdf */
|
||||
int cdef_find_dir_c(const uint16_t *img, int stride, int32_t *var,
|
||||
int coeff_shift) {
|
||||
int i;
|
||||
int32_t cost[8] = { 0 };
|
||||
int partial[8][15] = { { 0 } };
|
||||
int32_t best_cost = 0;
|
||||
int best_dir = 0;
|
||||
/* Instead of dividing by n between 2 and 8, we multiply by 3*5*7*8/n.
|
||||
The output is then 840 times larger, but we don't care for finding
|
||||
the max. */
|
||||
static const int div_table[] = { 0, 840, 420, 280, 210, 168, 140, 120, 105 };
|
||||
for (i = 0; i < 8; i++) {
|
||||
int j;
|
||||
for (j = 0; j < 8; j++) {
|
||||
int x;
|
||||
/* We subtract 128 here to reduce the maximum range of the squared
|
||||
partial sums. */
|
||||
x = (img[i * stride + j] >> coeff_shift) - 128;
|
||||
partial[0][i + j] += x;
|
||||
partial[1][i + j / 2] += x;
|
||||
partial[2][i] += x;
|
||||
partial[3][3 + i - j / 2] += x;
|
||||
partial[4][7 + i - j] += x;
|
||||
partial[5][3 - i / 2 + j] += x;
|
||||
partial[6][j] += x;
|
||||
partial[7][i / 2 + j] += x;
|
||||
}
|
||||
}
|
||||
for (i = 0; i < 8; i++) {
|
||||
cost[2] += partial[2][i] * partial[2][i];
|
||||
cost[6] += partial[6][i] * partial[6][i];
|
||||
}
|
||||
cost[2] *= div_table[8];
|
||||
cost[6] *= div_table[8];
|
||||
for (i = 0; i < 7; i++) {
|
||||
cost[0] += (partial[0][i] * partial[0][i] +
|
||||
partial[0][14 - i] * partial[0][14 - i]) *
|
||||
div_table[i + 1];
|
||||
cost[4] += (partial[4][i] * partial[4][i] +
|
||||
partial[4][14 - i] * partial[4][14 - i]) *
|
||||
div_table[i + 1];
|
||||
}
|
||||
cost[0] += partial[0][7] * partial[0][7] * div_table[8];
|
||||
cost[4] += partial[4][7] * partial[4][7] * div_table[8];
|
||||
for (i = 1; i < 8; i += 2) {
|
||||
int j;
|
||||
for (j = 0; j < 4 + 1; j++) {
|
||||
cost[i] += partial[i][3 + j] * partial[i][3 + j];
|
||||
}
|
||||
cost[i] *= div_table[8];
|
||||
for (j = 0; j < 4 - 1; j++) {
|
||||
cost[i] += (partial[i][j] * partial[i][j] +
|
||||
partial[i][10 - j] * partial[i][10 - j]) *
|
||||
div_table[2 * j + 2];
|
||||
}
|
||||
}
|
||||
for (i = 0; i < 8; i++) {
|
||||
if (cost[i] > best_cost) {
|
||||
best_cost = cost[i];
|
||||
best_dir = i;
|
||||
}
|
||||
}
|
||||
/* Difference between the optimal variance and the variance along the
|
||||
orthogonal direction. Again, the sum(x^2) terms cancel out. */
|
||||
*var = best_cost - cost[(best_dir + 4) & 7];
|
||||
/* We'd normally divide by 840, but dividing by 1024 is close enough
|
||||
for what we're going to do with this. */
|
||||
*var >>= 10;
|
||||
return best_dir;
|
||||
}
|
||||
|
||||
#if CONFIG_CDEF_SINGLEPASS
|
||||
#if CDEF_FULL
|
||||
const int cdef_pri_taps[2][3] = { { 3, 2, 1 }, { 2, 2, 2 } };
|
||||
const int cdef_sec_taps[2][2] = { { 3, 1 }, { 3, 1 } };
|
||||
#else
|
||||
const int cdef_pri_taps[2][2] = { { 4, 2 }, { 3, 3 } };
|
||||
const int cdef_sec_taps[2][2] = { { 2, 1 }, { 2, 1 } };
|
||||
#endif
|
||||
|
||||
/* Smooth in the direction detected. */
|
||||
#if CDEF_CAP
|
||||
void cdef_filter_block_c(uint8_t *dst8, uint16_t *dst16, int dstride,
|
||||
const uint16_t *in, int pri_strength, int sec_strength,
|
||||
int dir, int pri_damping, int sec_damping, int bsize,
|
||||
UNUSED int max_unused)
|
||||
#else
|
||||
void cdef_filter_block_c(uint8_t *dst8, uint16_t *dst16, int dstride,
|
||||
const uint16_t *in, int pri_strength, int sec_strength,
|
||||
int dir, int pri_damping, int sec_damping, int bsize,
|
||||
int max)
|
||||
#endif
|
||||
{
|
||||
int i, j, k;
|
||||
const int s = CDEF_BSTRIDE;
|
||||
const int *pri_taps = cdef_pri_taps[pri_strength & 1];
|
||||
const int *sec_taps = cdef_sec_taps[pri_strength & 1];
|
||||
for (i = 0; i < 4 << (bsize == BLOCK_8X8); i++) {
|
||||
for (j = 0; j < 4 << (bsize == BLOCK_8X8); j++) {
|
||||
int16_t sum = 0;
|
||||
int16_t y;
|
||||
int16_t x = in[i * s + j];
|
||||
#if CDEF_CAP
|
||||
int max = x;
|
||||
int min = x;
|
||||
#endif
|
||||
#if CDEF_FULL
|
||||
for (k = 0; k < 3; k++)
|
||||
#else
|
||||
for (k = 0; k < 2; k++)
|
||||
#endif
|
||||
{
|
||||
int16_t p0 = in[i * s + j + cdef_directions[dir][k]];
|
||||
int16_t p1 = in[i * s + j - cdef_directions[dir][k]];
|
||||
sum += pri_taps[k] * constrain(p0 - x, pri_strength, pri_damping);
|
||||
sum += pri_taps[k] * constrain(p1 - x, pri_strength, pri_damping);
|
||||
#if CDEF_CAP
|
||||
if (p0 != CDEF_VERY_LARGE) max = AOMMAX(p0, max);
|
||||
if (p1 != CDEF_VERY_LARGE) max = AOMMAX(p1, max);
|
||||
min = AOMMIN(p0, min);
|
||||
min = AOMMIN(p1, min);
|
||||
#endif
|
||||
#if CDEF_FULL
|
||||
if (k == 2) continue;
|
||||
#endif
|
||||
int16_t s0 = in[i * s + j + cdef_directions[(dir + 2) & 7][k]];
|
||||
int16_t s1 = in[i * s + j - cdef_directions[(dir + 2) & 7][k]];
|
||||
int16_t s2 = in[i * s + j + cdef_directions[(dir + 6) & 7][k]];
|
||||
int16_t s3 = in[i * s + j - cdef_directions[(dir + 6) & 7][k]];
|
||||
#if CDEF_CAP
|
||||
if (s0 != CDEF_VERY_LARGE) max = AOMMAX(s0, max);
|
||||
if (s1 != CDEF_VERY_LARGE) max = AOMMAX(s1, max);
|
||||
if (s2 != CDEF_VERY_LARGE) max = AOMMAX(s2, max);
|
||||
if (s3 != CDEF_VERY_LARGE) max = AOMMAX(s3, max);
|
||||
min = AOMMIN(s0, min);
|
||||
min = AOMMIN(s1, min);
|
||||
min = AOMMIN(s2, min);
|
||||
min = AOMMIN(s3, min);
|
||||
#endif
|
||||
sum += sec_taps[k] * constrain(s0 - x, sec_strength, sec_damping);
|
||||
sum += sec_taps[k] * constrain(s1 - x, sec_strength, sec_damping);
|
||||
sum += sec_taps[k] * constrain(s2 - x, sec_strength, sec_damping);
|
||||
sum += sec_taps[k] * constrain(s3 - x, sec_strength, sec_damping);
|
||||
}
|
||||
#if CDEF_CAP
|
||||
y = clamp((int16_t)x + ((8 + sum - (sum < 0)) >> 4), min, max);
|
||||
#else
|
||||
y = clamp((int16_t)x + ((8 + sum - (sum < 0)) >> 4), 0, max);
|
||||
#endif
|
||||
if (dst8)
|
||||
dst8[i * dstride + j] = (uint8_t)y;
|
||||
else
|
||||
dst16[i * dstride + j] = (uint16_t)y;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#else
|
||||
|
||||
/* Smooth in the direction detected. */
|
||||
void cdef_direction_8x8_c(uint16_t *y, int ystride, const uint16_t *in,
|
||||
int threshold, int dir, int damping) {
|
||||
int i;
|
||||
int j;
|
||||
int k;
|
||||
static const int taps[3] = { 3, 2, 1 };
|
||||
for (i = 0; i < 8; i++) {
|
||||
for (j = 0; j < 8; j++) {
|
||||
int16_t sum;
|
||||
int16_t xx;
|
||||
int16_t yy;
|
||||
xx = in[i * CDEF_BSTRIDE + j];
|
||||
sum = 0;
|
||||
for (k = 0; k < 3; k++) {
|
||||
int16_t p0;
|
||||
int16_t p1;
|
||||
p0 = in[i * CDEF_BSTRIDE + j + cdef_directions[dir][k]] - xx;
|
||||
p1 = in[i * CDEF_BSTRIDE + j - cdef_directions[dir][k]] - xx;
|
||||
sum += taps[k] * constrain(p0, threshold, damping);
|
||||
sum += taps[k] * constrain(p1, threshold, damping);
|
||||
}
|
||||
sum = (sum + 8) >> 4;
|
||||
yy = xx + sum;
|
||||
y[i * ystride + j] = yy;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/* Smooth in the direction detected. */
|
||||
void cdef_direction_4x4_c(uint16_t *y, int ystride, const uint16_t *in,
|
||||
int threshold, int dir, int damping) {
|
||||
int i;
|
||||
int j;
|
||||
int k;
|
||||
static const int taps[2] = { 4, 1 };
|
||||
for (i = 0; i < 4; i++) {
|
||||
for (j = 0; j < 4; j++) {
|
||||
int16_t sum;
|
||||
int16_t xx;
|
||||
int16_t yy;
|
||||
xx = in[i * CDEF_BSTRIDE + j];
|
||||
sum = 0;
|
||||
for (k = 0; k < 2; k++) {
|
||||
int16_t p0;
|
||||
int16_t p1;
|
||||
p0 = in[i * CDEF_BSTRIDE + j + cdef_directions[dir][k]] - xx;
|
||||
p1 = in[i * CDEF_BSTRIDE + j - cdef_directions[dir][k]] - xx;
|
||||
sum += taps[k] * constrain(p0, threshold, damping);
|
||||
sum += taps[k] * constrain(p1, threshold, damping);
|
||||
}
|
||||
sum = (sum + 8) >> 4;
|
||||
yy = xx + sum;
|
||||
y[i * ystride + j] = yy;
|
||||
}
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
/* Compute the primary filter strength for an 8x8 block based on the
|
||||
directional variance difference. A high variance difference means
|
||||
that we have a highly directional pattern (e.g. a high contrast
|
||||
edge), so we can apply more deringing. A low variance means that we
|
||||
either have a low contrast edge, or a non-directional texture, so
|
||||
we want to be careful not to blur. */
|
||||
static INLINE int adjust_strength(int strength, int32_t var) {
|
||||
const int i = var >> 6 ? AOMMIN(get_msb(var >> 6), 12) : 0;
|
||||
/* We use the variance of 8x8 blocks to adjust the strength. */
|
||||
return var ? (strength * (4 + i) + 8) >> 4 : 0;
|
||||
}
|
||||
|
||||
#if !CONFIG_CDEF_SINGLEPASS
|
||||
void copy_8x8_16bit_to_16bit_c(uint16_t *dst, int dstride, const uint16_t *src,
|
||||
int sstride) {
|
||||
int i, j;
|
||||
for (i = 0; i < 8; i++)
|
||||
for (j = 0; j < 8; j++) dst[i * dstride + j] = src[i * sstride + j];
|
||||
}
|
||||
|
||||
void copy_4x4_16bit_to_16bit_c(uint16_t *dst, int dstride, const uint16_t *src,
|
||||
int sstride) {
|
||||
int i, j;
|
||||
for (i = 0; i < 4; i++)
|
||||
for (j = 0; j < 4; j++) dst[i * dstride + j] = src[i * sstride + j];
|
||||
}
|
||||
|
||||
static void copy_block_16bit_to_16bit(uint16_t *dst, int dstride, uint16_t *src,
|
||||
cdef_list *dlist, int cdef_count,
|
||||
int bsize) {
|
||||
int bi, bx, by;
|
||||
|
||||
if (bsize == BLOCK_8X8) {
|
||||
for (bi = 0; bi < cdef_count; bi++) {
|
||||
by = dlist[bi].by;
|
||||
bx = dlist[bi].bx;
|
||||
copy_8x8_16bit_to_16bit(&dst[(by << 3) * dstride + (bx << 3)], dstride,
|
||||
&src[bi << (3 + 3)], 8);
|
||||
}
|
||||
} else if (bsize == BLOCK_4X8) {
|
||||
for (bi = 0; bi < cdef_count; bi++) {
|
||||
by = dlist[bi].by;
|
||||
bx = dlist[bi].bx;
|
||||
copy_4x4_16bit_to_16bit(&dst[(by << 3) * dstride + (bx << 2)], dstride,
|
||||
&src[bi << (3 + 2)], 4);
|
||||
copy_4x4_16bit_to_16bit(&dst[((by << 3) + 4) * dstride + (bx << 2)],
|
||||
dstride, &src[(bi << (3 + 2)) + 4 * 4], 4);
|
||||
}
|
||||
} else if (bsize == BLOCK_8X4) {
|
||||
for (bi = 0; bi < cdef_count; bi++) {
|
||||
by = dlist[bi].by;
|
||||
bx = dlist[bi].bx;
|
||||
copy_4x4_16bit_to_16bit(&dst[(by << 2) * dstride + (bx << 3)], dstride,
|
||||
&src[bi << (2 + 3)], 8);
|
||||
copy_4x4_16bit_to_16bit(&dst[(by << 2) * dstride + (bx << 3) + 4],
|
||||
dstride, &src[(bi << (2 + 3)) + 4], 8);
|
||||
}
|
||||
} else {
|
||||
assert(bsize == BLOCK_4X4);
|
||||
for (bi = 0; bi < cdef_count; bi++) {
|
||||
by = dlist[bi].by;
|
||||
bx = dlist[bi].bx;
|
||||
copy_4x4_16bit_to_16bit(&dst[(by << 2) * dstride + (bx << 2)], dstride,
|
||||
&src[bi << (2 + 2)], 4);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void copy_8x8_16bit_to_8bit_c(uint8_t *dst, int dstride, const uint16_t *src,
|
||||
int sstride) {
|
||||
int i, j;
|
||||
for (i = 0; i < 8; i++)
|
||||
for (j = 0; j < 8; j++)
|
||||
dst[i * dstride + j] = (uint8_t)src[i * sstride + j];
|
||||
}
|
||||
|
||||
void copy_4x4_16bit_to_8bit_c(uint8_t *dst, int dstride, const uint16_t *src,
|
||||
int sstride) {
|
||||
int i, j;
|
||||
for (i = 0; i < 4; i++)
|
||||
for (j = 0; j < 4; j++)
|
||||
dst[i * dstride + j] = (uint8_t)src[i * sstride + j];
|
||||
}
|
||||
|
||||
static void copy_block_16bit_to_8bit(uint8_t *dst, int dstride,
|
||||
const uint16_t *src, cdef_list *dlist,
|
||||
int cdef_count, int bsize) {
|
||||
int bi, bx, by;
|
||||
if (bsize == BLOCK_8X8) {
|
||||
for (bi = 0; bi < cdef_count; bi++) {
|
||||
by = dlist[bi].by;
|
||||
bx = dlist[bi].bx;
|
||||
copy_8x8_16bit_to_8bit(&dst[(by << 3) * dstride + (bx << 3)], dstride,
|
||||
&src[bi << (3 + 3)], 8);
|
||||
}
|
||||
} else if (bsize == BLOCK_4X8) {
|
||||
for (bi = 0; bi < cdef_count; bi++) {
|
||||
by = dlist[bi].by;
|
||||
bx = dlist[bi].bx;
|
||||
copy_4x4_16bit_to_8bit(&dst[(by << 3) * dstride + (bx << 2)], dstride,
|
||||
&src[bi << (3 + 2)], 4);
|
||||
copy_4x4_16bit_to_8bit(&dst[((by << 3) + 4) * dstride + (bx << 2)],
|
||||
dstride, &src[(bi << (3 + 2)) + 4 * 4], 4);
|
||||
}
|
||||
} else if (bsize == BLOCK_8X4) {
|
||||
for (bi = 0; bi < cdef_count; bi++) {
|
||||
by = dlist[bi].by;
|
||||
bx = dlist[bi].bx;
|
||||
copy_4x4_16bit_to_8bit(&dst[(by << 2) * dstride + (bx << 3)], dstride,
|
||||
&src[bi << (2 + 3)], 8);
|
||||
copy_4x4_16bit_to_8bit(&dst[(by << 2) * dstride + (bx << 3) + 4], dstride,
|
||||
&src[(bi << (2 + 3)) + 4], 8);
|
||||
}
|
||||
} else {
|
||||
assert(bsize == BLOCK_4X4);
|
||||
for (bi = 0; bi < cdef_count; bi++) {
|
||||
by = dlist[bi].by;
|
||||
bx = dlist[bi].bx;
|
||||
copy_4x4_16bit_to_8bit(&dst[(by << 2) * dstride + (bx << 2)], dstride,
|
||||
&src[bi << (2 * 2)], 4);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
int get_filter_skip(int level) {
|
||||
int filter_skip = level & 1;
|
||||
if (level == 1) filter_skip = 0;
|
||||
return filter_skip;
|
||||
}
|
||||
|
||||
void cdef_filter_fb(uint8_t *dst, int dstride, uint16_t *y, uint16_t *in,
|
||||
int xdec, int ydec, int dir[CDEF_NBLOCKS][CDEF_NBLOCKS],
|
||||
int *dirinit, int var[CDEF_NBLOCKS][CDEF_NBLOCKS], int pli,
|
||||
cdef_list *dlist, int cdef_count, int level,
|
||||
int sec_strength, int sec_damping, int pri_damping,
|
||||
int coeff_shift, int skip_dering, int hbd) {
|
||||
#else
|
||||
|
||||
void cdef_filter_fb(uint8_t *dst8, uint16_t *dst16, int dstride, uint16_t *in,
|
||||
int xdec, int ydec, int dir[CDEF_NBLOCKS][CDEF_NBLOCKS],
|
||||
int *dirinit, int var[CDEF_NBLOCKS][CDEF_NBLOCKS], int pli,
|
||||
cdef_list *dlist, int cdef_count, int level,
|
||||
int sec_strength, int pri_damping, int sec_damping,
|
||||
int coeff_shift) {
|
||||
#endif
|
||||
int bi;
|
||||
int bx;
|
||||
int by;
|
||||
int bsize, bsizex, bsizey;
|
||||
|
||||
#if CONFIG_CDEF_SINGLEPASS
|
||||
int pri_strength = (level >> 1) << coeff_shift;
|
||||
int filter_skip = level & 1;
|
||||
if (!pri_strength && !sec_strength && filter_skip) {
|
||||
pri_strength = 19 << coeff_shift;
|
||||
sec_strength = 7 << coeff_shift;
|
||||
}
|
||||
#else
|
||||
int threshold = (level >> 1) << coeff_shift;
|
||||
int filter_skip = get_filter_skip(level);
|
||||
if (level == 1) threshold = 31 << coeff_shift;
|
||||
|
||||
cdef_direction_func cdef_direction[] = { cdef_direction_4x4,
|
||||
cdef_direction_8x8 };
|
||||
#endif
|
||||
sec_damping += coeff_shift - (pli != AOM_PLANE_Y);
|
||||
pri_damping += coeff_shift - (pli != AOM_PLANE_Y);
|
||||
bsize =
|
||||
ydec ? (xdec ? BLOCK_4X4 : BLOCK_8X4) : (xdec ? BLOCK_4X8 : BLOCK_8X8);
|
||||
bsizex = 3 - xdec;
|
||||
bsizey = 3 - ydec;
|
||||
#if CONFIG_CDEF_SINGLEPASS
|
||||
if (dirinit && pri_strength == 0 && sec_strength == 0)
|
||||
#else
|
||||
if (!skip_dering)
|
||||
#endif
|
||||
{
|
||||
#if CONFIG_CDEF_SINGLEPASS
|
||||
// If we're here, both primary and secondary strengths are 0, and
|
||||
// we still haven't written anything to y[] yet, so we just copy
|
||||
// the input to y[]. This is necessary only for av1_cdef_search()
|
||||
// and only av1_cdef_search() sets dirinit.
|
||||
for (bi = 0; bi < cdef_count; bi++) {
|
||||
by = dlist[bi].by;
|
||||
bx = dlist[bi].bx;
|
||||
#else
|
||||
if (pli == 0) {
|
||||
if (!dirinit || !*dirinit) {
|
||||
for (bi = 0; bi < cdef_count; bi++) {
|
||||
by = dlist[bi].by;
|
||||
bx = dlist[bi].bx;
|
||||
dir[by][bx] = cdef_find_dir(&in[8 * by * CDEF_BSTRIDE + 8 * bx],
|
||||
CDEF_BSTRIDE, &var[by][bx], coeff_shift);
|
||||
}
|
||||
if (dirinit) *dirinit = 1;
|
||||
}
|
||||
}
|
||||
// Only run dering for non-zero threshold (which is always the case for
|
||||
// 4:2:2 or 4:4:0). If we don't dering, we still need to eventually write
|
||||
// something out in y[] later.
|
||||
if (threshold != 0) {
|
||||
assert(bsize == BLOCK_8X8 || bsize == BLOCK_4X4);
|
||||
for (bi = 0; bi < cdef_count; bi++) {
|
||||
int t = !filter_skip && dlist[bi].skip ? 0 : threshold;
|
||||
by = dlist[bi].by;
|
||||
bx = dlist[bi].bx;
|
||||
(cdef_direction[bsize == BLOCK_8X8])(
|
||||
&y[bi << (bsizex + bsizey)], 1 << bsizex,
|
||||
&in[(by * CDEF_BSTRIDE << bsizey) + (bx << bsizex)],
|
||||
pli ? t : adjust_strength(t, var[by][bx]), dir[by][bx],
|
||||
pri_damping);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (sec_strength) {
|
||||
if (threshold && !skip_dering)
|
||||
copy_block_16bit_to_16bit(in, CDEF_BSTRIDE, y, dlist, cdef_count, bsize);
|
||||
for (bi = 0; bi < cdef_count; bi++) {
|
||||
by = dlist[bi].by;
|
||||
bx = dlist[bi].bx;
|
||||
int py = by << bsizey;
|
||||
int px = bx << bsizex;
|
||||
|
||||
if (!filter_skip && dlist[bi].skip) continue;
|
||||
if (!dst || hbd) {
|
||||
// 16 bit destination if high bitdepth or 8 bit destination not given
|
||||
(!threshold || (dir[by][bx] < 4 && dir[by][bx]) ? aom_clpf_block_hbd
|
||||
: aom_clpf_hblock_hbd)(
|
||||
dst ? (uint16_t *)dst + py * dstride + px
|
||||
: &y[bi << (bsizex + bsizey)],
|
||||
in + py * CDEF_BSTRIDE + px, dst && hbd ? dstride : 1 << bsizex,
|
||||
CDEF_BSTRIDE, 1 << bsizex, 1 << bsizey, sec_strength << coeff_shift,
|
||||
sec_damping);
|
||||
} else {
|
||||
// Do clpf and write the result to an 8 bit destination
|
||||
(!threshold || (dir[by][bx] < 4 && dir[by][bx]) ? aom_clpf_block
|
||||
: aom_clpf_hblock)(
|
||||
dst + py * dstride + px, in + py * CDEF_BSTRIDE + px, dstride,
|
||||
CDEF_BSTRIDE, 1 << bsizex, 1 << bsizey, sec_strength << coeff_shift,
|
||||
sec_damping);
|
||||
}
|
||||
}
|
||||
} else if (threshold != 0) {
|
||||
// No clpf, so copy instead
|
||||
if (hbd) {
|
||||
copy_block_16bit_to_16bit((uint16_t *)dst, dstride, y, dlist, cdef_count,
|
||||
bsize);
|
||||
} else {
|
||||
copy_block_16bit_to_8bit(dst, dstride, y, dlist, cdef_count, bsize);
|
||||
}
|
||||
} else if (dirinit) {
|
||||
// If we're here, both dering and clpf are off, and we still haven't written
|
||||
// anything to y[] yet, so we just copy the input to y[]. This is necessary
|
||||
// only for av1_cdef_search() and only av1_cdef_search() sets dirinit.
|
||||
for (bi = 0; bi < cdef_count; bi++) {
|
||||
by = dlist[bi].by;
|
||||
bx = dlist[bi].bx;
|
||||
#endif
|
||||
int iy, ix;
|
||||
// TODO(stemidts/jmvalin): SIMD optimisations
|
||||
for (iy = 0; iy < 1 << bsizey; iy++)
|
||||
for (ix = 0; ix < 1 << bsizex; ix++)
|
||||
#if CONFIG_CDEF_SINGLEPASS
|
||||
dst16[(bi << (bsizex + bsizey)) + (iy << bsizex) + ix] =
|
||||
#else
|
||||
y[(bi << (bsizex + bsizey)) + (iy << bsizex) + ix] =
|
||||
#endif
|
||||
in[((by << bsizey) + iy) * CDEF_BSTRIDE + (bx << bsizex) + ix];
|
||||
}
|
||||
#if CONFIG_CDEF_SINGLEPASS
|
||||
return;
|
||||
#endif
|
||||
}
|
||||
|
||||
#if CONFIG_CDEF_SINGLEPASS
|
||||
if (pli == 0) {
|
||||
if (!dirinit || !*dirinit) {
|
||||
for (bi = 0; bi < cdef_count; bi++) {
|
||||
by = dlist[bi].by;
|
||||
bx = dlist[bi].bx;
|
||||
dir[by][bx] = cdef_find_dir(&in[8 * by * CDEF_BSTRIDE + 8 * bx],
|
||||
CDEF_BSTRIDE, &var[by][bx], coeff_shift);
|
||||
}
|
||||
if (dirinit) *dirinit = 1;
|
||||
}
|
||||
}
|
||||
|
||||
assert(bsize == BLOCK_8X8 || bsize == BLOCK_4X4);
|
||||
for (bi = 0; bi < cdef_count; bi++) {
|
||||
int t = !filter_skip && dlist[bi].skip ? 0 : pri_strength;
|
||||
int s = !filter_skip && dlist[bi].skip ? 0 : sec_strength;
|
||||
by = dlist[bi].by;
|
||||
bx = dlist[bi].bx;
|
||||
if (dst8)
|
||||
cdef_filter_block(
|
||||
&dst8[(by << bsizey) * dstride + (bx << bsizex)], NULL, dstride,
|
||||
&in[(by * CDEF_BSTRIDE << bsizey) + (bx << bsizex)],
|
||||
(pli ? t : adjust_strength(t, var[by][bx])), s, t ? dir[by][bx] : 0,
|
||||
pri_damping, sec_damping, bsize, (256 << coeff_shift) - 1);
|
||||
else
|
||||
cdef_filter_block(
|
||||
NULL,
|
||||
&dst16[dirinit ? bi << (bsizex + bsizey)
|
||||
: (by << bsizey) * dstride + (bx << bsizex)],
|
||||
dirinit ? 1 << bsizex : dstride,
|
||||
&in[(by * CDEF_BSTRIDE << bsizey) + (bx << bsizex)],
|
||||
(pli ? t : adjust_strength(t, var[by][bx])), s, t ? dir[by][bx] : 0,
|
||||
pri_damping, sec_damping, bsize, (256 << coeff_shift) - 1);
|
||||
}
|
||||
#endif
|
||||
}
|
||||
90
third_party/aom/av1/common/cdef_block.h
vendored
Normal file
90
third_party/aom/av1/common/cdef_block.h
vendored
Normal file
|
|
@ -0,0 +1,90 @@
|
|||
/*
|
||||
* Copyright (c) 2016, Alliance for Open Media. All rights reserved
|
||||
*
|
||||
* This source code is subject to the terms of the BSD 2 Clause License and
|
||||
* the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
|
||||
* was not distributed with this source code in the LICENSE file, you can
|
||||
* obtain it at www.aomedia.org/license/software. If the Alliance for Open
|
||||
* Media Patent License 1.0 was not distributed with this source code in the
|
||||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#if !defined(_CDEF_BLOCK_H)
|
||||
#define _CDEF_BLOCK_H (1)
|
||||
|
||||
#include "./odintrin.h"
|
||||
|
||||
#define CDEF_BLOCKSIZE 64
|
||||
#define CDEF_BLOCKSIZE_LOG2 6
|
||||
#define CDEF_NBLOCKS (CDEF_BLOCKSIZE / 8)
|
||||
#if CONFIG_CDEF_SINGLEPASS
|
||||
#define CDEF_SB_SHIFT (MAX_SB_SIZE_LOG2 - CDEF_BLOCKSIZE_LOG2)
|
||||
#endif
|
||||
|
||||
/* We need to buffer three vertical lines. */
|
||||
#define CDEF_VBORDER (3)
|
||||
/* We only need to buffer three horizontal pixels too, but let's align to
|
||||
16 bytes (8 x 16 bits) to make vectorization easier. */
|
||||
#define CDEF_HBORDER (8)
|
||||
#define CDEF_BSTRIDE ALIGN_POWER_OF_TWO(CDEF_BLOCKSIZE + 2 * CDEF_HBORDER, 3)
|
||||
|
||||
#define CDEF_VERY_LARGE (30000)
|
||||
#define CDEF_INBUF_SIZE (CDEF_BSTRIDE * (CDEF_BLOCKSIZE + 2 * CDEF_VBORDER))
|
||||
|
||||
#if CONFIG_CDEF_SINGLEPASS
|
||||
// Filter configuration
|
||||
#define CDEF_CAP 1 // 1 = Cap change to largest diff
|
||||
#define CDEF_FULL 0 // 1 = 7x7 filter, 0 = 5x5 filter
|
||||
|
||||
#if CDEF_FULL
|
||||
extern const int cdef_pri_taps[2][3];
|
||||
extern const int cdef_sec_taps[2][2];
|
||||
extern const int cdef_directions[8][3];
|
||||
#else
|
||||
extern const int cdef_pri_taps[2][2];
|
||||
extern const int cdef_sec_taps[2][2];
|
||||
extern const int cdef_directions[8][2];
|
||||
#endif
|
||||
|
||||
#else // CONFIG_CDEF_SINGLEPASS
|
||||
extern const int cdef_directions[8][3];
|
||||
#endif
|
||||
|
||||
typedef struct {
|
||||
uint8_t by;
|
||||
uint8_t bx;
|
||||
uint8_t skip;
|
||||
} cdef_list;
|
||||
|
||||
#if CONFIG_CDEF_SINGLEPASS
|
||||
typedef void (*cdef_filter_block_func)(uint8_t *dst8, uint16_t *dst16,
|
||||
int dstride, const uint16_t *in,
|
||||
int pri_strength, int sec_strength,
|
||||
int dir, int pri_damping,
|
||||
int sec_damping, int bsize, int max);
|
||||
void copy_cdef_16bit_to_16bit(uint16_t *dst, int dstride, uint16_t *src,
|
||||
cdef_list *dlist, int cdef_count, int bsize);
|
||||
#else
|
||||
typedef void (*cdef_direction_func)(uint16_t *y, int ystride,
|
||||
const uint16_t *in, int threshold, int dir,
|
||||
int damping);
|
||||
|
||||
int get_filter_skip(int level);
|
||||
#endif
|
||||
|
||||
#if CONFIG_CDEF_SINGLEPASS
|
||||
void cdef_filter_fb(uint8_t *dst8, uint16_t *dst16, int dstride, uint16_t *in,
|
||||
int xdec, int ydec, int dir[CDEF_NBLOCKS][CDEF_NBLOCKS],
|
||||
int *dirinit, int var[CDEF_NBLOCKS][CDEF_NBLOCKS], int pli,
|
||||
cdef_list *dlist, int cdef_count, int level,
|
||||
int sec_strength, int pri_damping, int sec_damping,
|
||||
int coeff_shift);
|
||||
#else
|
||||
void cdef_filter_fb(uint8_t *dst, int dstride, uint16_t *y, uint16_t *in,
|
||||
int xdec, int ydec, int dir[CDEF_NBLOCKS][CDEF_NBLOCKS],
|
||||
int *dirinit, int var[CDEF_NBLOCKS][CDEF_NBLOCKS], int pli,
|
||||
cdef_list *dlist, int cdef_count, int level,
|
||||
int sec_strength, int sec_damping, int pri_damping,
|
||||
int coeff_shift, int skip_dering, int hbd);
|
||||
#endif
|
||||
#endif
|
||||
14
third_party/aom/av1/common/cdef_block_avx2.c
vendored
Normal file
14
third_party/aom/av1/common/cdef_block_avx2.c
vendored
Normal file
|
|
@ -0,0 +1,14 @@
|
|||
/*
|
||||
* Copyright (c) 2016, Alliance for Open Media. All rights reserved
|
||||
*
|
||||
* This source code is subject to the terms of the BSD 2 Clause License and
|
||||
* the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
|
||||
* was not distributed with this source code in the LICENSE file, you can
|
||||
* obtain it at www.aomedia.org/license/software. If the Alliance for Open
|
||||
* Media Patent License 1.0 was not distributed with this source code in the
|
||||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#include "aom_dsp/aom_simd.h"
|
||||
#define SIMD_FUNC(name) name##_avx2
|
||||
#include "./cdef_block_simd.h"
|
||||
|
|
@ -11,4 +11,4 @@
|
|||
|
||||
#include "aom_dsp/aom_simd.h"
|
||||
#define SIMD_FUNC(name) name##_neon
|
||||
#include "./od_dering_simd.h"
|
||||
#include "./cdef_block_simd.h"
|
||||
1214
third_party/aom/av1/common/cdef_block_simd.h
vendored
Normal file
1214
third_party/aom/av1/common/cdef_block_simd.h
vendored
Normal file
File diff suppressed because it is too large
Load diff
|
|
@ -11,4 +11,4 @@
|
|||
|
||||
#include "aom_dsp/aom_simd.h"
|
||||
#define SIMD_FUNC(name) name##_sse2
|
||||
#include "./od_dering_simd.h"
|
||||
#include "./cdef_block_simd.h"
|
||||
|
|
@ -11,4 +11,4 @@
|
|||
|
||||
#include "aom_dsp/aom_simd.h"
|
||||
#define SIMD_FUNC(name) name##_sse4_1
|
||||
#include "./od_dering_simd.h"
|
||||
#include "./cdef_block_simd.h"
|
||||
|
|
@ -11,4 +11,4 @@
|
|||
|
||||
#include "aom_dsp/aom_simd.h"
|
||||
#define SIMD_FUNC(name) name##_ssse3
|
||||
#include "./od_dering_simd.h"
|
||||
#include "./cdef_block_simd.h"
|
||||
27
third_party/aom/av1/common/cdef_simd.h
vendored
27
third_party/aom/av1/common/cdef_simd.h
vendored
|
|
@ -1,27 +0,0 @@
|
|||
/*
|
||||
* Copyright (c) 2016, Alliance for Open Media. All rights reserved
|
||||
*
|
||||
* This source code is subject to the terms of the BSD 2 Clause License and
|
||||
* the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
|
||||
* was not distributed with this source code in the LICENSE file, you can
|
||||
* obtain it at www.aomedia.org/license/software. If the Alliance for Open
|
||||
* Media Patent License 1.0 was not distributed with this source code in the
|
||||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
#ifndef AV1_COMMON_CDEF_SIMD_H_
|
||||
#define AV1_COMMON_CDEF_SIMD_H_
|
||||
|
||||
#include "aom_dsp/aom_simd.h"
|
||||
|
||||
// sign(a-b) * min(abs(a-b), max(0, threshold - (abs(a-b) >> adjdamp)))
|
||||
SIMD_INLINE v128 constrain16(v128 a, v128 b, unsigned int threshold,
|
||||
unsigned int adjdamp) {
|
||||
v128 diff = v128_sub_16(a, b);
|
||||
const v128 sign = v128_shr_n_s16(diff, 15);
|
||||
diff = v128_abs_s16(diff);
|
||||
const v128 s =
|
||||
v128_ssub_u16(v128_dup_16(threshold), v128_shr_u16(diff, adjdamp));
|
||||
return v128_xor(v128_add_16(sign, v128_min_s16(diff, s)), sign);
|
||||
}
|
||||
|
||||
#endif // AV1_COMMON_CDEF_SIMD_H_
|
||||
621
third_party/aom/av1/common/cfl.c
vendored
621
third_party/aom/av1/common/cfl.c
vendored
|
|
@ -13,117 +13,148 @@
|
|||
#include "av1/common/common_data.h"
|
||||
#include "av1/common/onyxc_int.h"
|
||||
|
||||
#include "aom/internal/aom_codec_internal.h"
|
||||
|
||||
void cfl_init(CFL_CTX *cfl, AV1_COMMON *cm) {
|
||||
if (!((cm->subsampling_x == 0 && cm->subsampling_y == 0) ||
|
||||
(cm->subsampling_x == 1 && cm->subsampling_y == 1))) {
|
||||
aom_internal_error(&cm->error, AOM_CODEC_UNSUP_BITSTREAM,
|
||||
"Only 4:4:4 and 4:2:0 are currently supported by CfL");
|
||||
}
|
||||
memset(&cfl->y_pix, 0, sizeof(uint8_t) * MAX_SB_SQUARE);
|
||||
memset(&cfl->pred_buf_q3, 0, sizeof(cfl->pred_buf_q3));
|
||||
cfl->subsampling_x = cm->subsampling_x;
|
||||
cfl->subsampling_y = cm->subsampling_y;
|
||||
cfl->are_parameters_computed = 0;
|
||||
cfl->store_y = 0;
|
||||
#if CONFIG_CHROMA_SUB8X8 && CONFIG_DEBUG
|
||||
cfl_clear_sub8x8_val(cfl);
|
||||
#endif // CONFIG_CHROMA_SUB8X8 && CONFIG_DEBUG
|
||||
}
|
||||
|
||||
// Load from the CfL pixel buffer into output
|
||||
static void cfl_load(CFL_CTX *cfl, int row, int col, int width, int height) {
|
||||
const int sub_x = cfl->subsampling_x;
|
||||
const int sub_y = cfl->subsampling_y;
|
||||
const int off_log2 = tx_size_wide_log2[0];
|
||||
|
||||
// TODO(ltrudeau) convert to uint16 to add HBD support
|
||||
const uint8_t *y_pix;
|
||||
// TODO(ltrudeau) convert to uint16 to add HBD support
|
||||
uint8_t *output = cfl->y_down_pix;
|
||||
|
||||
int pred_row_offset = 0;
|
||||
int output_row_offset = 0;
|
||||
|
||||
// TODO(ltrudeau) should be faster to downsample when we store the values
|
||||
// TODO(ltrudeau) add support for 4:2:2
|
||||
if (sub_y == 0 && sub_x == 0) {
|
||||
y_pix = &cfl->y_pix[(row * MAX_SB_SIZE + col) << off_log2];
|
||||
for (int j = 0; j < height; j++) {
|
||||
for (int i = 0; i < width; i++) {
|
||||
// In 4:4:4, pixels match 1 to 1
|
||||
output[output_row_offset + i] = y_pix[pred_row_offset + i];
|
||||
}
|
||||
pred_row_offset += MAX_SB_SIZE;
|
||||
output_row_offset += MAX_SB_SIZE;
|
||||
}
|
||||
} else if (sub_y == 1 && sub_x == 1) {
|
||||
y_pix = &cfl->y_pix[(row * MAX_SB_SIZE + col) << (off_log2 + sub_y)];
|
||||
for (int j = 0; j < height; j++) {
|
||||
for (int i = 0; i < width; i++) {
|
||||
int top_left = (pred_row_offset + i) << sub_y;
|
||||
int bot_left = top_left + MAX_SB_SIZE;
|
||||
// In 4:2:0, average pixels in 2x2 grid
|
||||
output[output_row_offset + i] = OD_SHR_ROUND(
|
||||
y_pix[top_left] + y_pix[top_left + 1] // Top row
|
||||
+ y_pix[bot_left] + y_pix[bot_left + 1] // Bottom row
|
||||
,
|
||||
2);
|
||||
}
|
||||
pred_row_offset += MAX_SB_SIZE;
|
||||
output_row_offset += MAX_SB_SIZE;
|
||||
}
|
||||
} else {
|
||||
assert(0); // Unsupported chroma subsampling
|
||||
}
|
||||
// Due to frame boundary issues, it is possible that the total area of
|
||||
// covered by Chroma exceeds that of Luma. When this happens, we write over
|
||||
// the broken data by repeating the last columns and/or rows.
|
||||
//
|
||||
// Note that in order to manage the case where both rows and columns
|
||||
// overrun,
|
||||
// we apply rows first. This way, when the rows overrun the bottom of the
|
||||
// frame, the columns will be copied over them.
|
||||
const int uv_width = (col << off_log2) + width;
|
||||
const int uv_height = (row << off_log2) + height;
|
||||
|
||||
const int diff_width = uv_width - (cfl->y_width >> sub_x);
|
||||
const int diff_height = uv_height - (cfl->y_height >> sub_y);
|
||||
// Due to frame boundary issues, it is possible that the total area covered by
|
||||
// chroma exceeds that of luma. When this happens, we fill the missing pixels by
|
||||
// repeating the last columns and/or rows.
|
||||
static INLINE void cfl_pad(CFL_CTX *cfl, int width, int height) {
|
||||
const int diff_width = width - cfl->buf_width;
|
||||
const int diff_height = height - cfl->buf_height;
|
||||
|
||||
if (diff_width > 0) {
|
||||
int last_pixel;
|
||||
output_row_offset = width - diff_width;
|
||||
|
||||
for (int j = 0; j < height; j++) {
|
||||
last_pixel = output_row_offset - 1;
|
||||
const int min_height = height - diff_height;
|
||||
int16_t *pred_buf_q3 = cfl->pred_buf_q3 + (width - diff_width);
|
||||
for (int j = 0; j < min_height; j++) {
|
||||
const int last_pixel = pred_buf_q3[-1];
|
||||
for (int i = 0; i < diff_width; i++) {
|
||||
output[output_row_offset + i] = output[last_pixel];
|
||||
pred_buf_q3[i] = last_pixel;
|
||||
}
|
||||
output_row_offset += MAX_SB_SIZE;
|
||||
pred_buf_q3 += MAX_SB_SIZE;
|
||||
}
|
||||
cfl->buf_width = width;
|
||||
}
|
||||
|
||||
if (diff_height > 0) {
|
||||
output_row_offset = (height - diff_height) * MAX_SB_SIZE;
|
||||
const int last_row_offset = output_row_offset - MAX_SB_SIZE;
|
||||
|
||||
int16_t *pred_buf_q3 =
|
||||
cfl->pred_buf_q3 + ((height - diff_height) * MAX_SB_SIZE);
|
||||
for (int j = 0; j < diff_height; j++) {
|
||||
const int16_t *last_row_q3 = pred_buf_q3 - MAX_SB_SIZE;
|
||||
for (int i = 0; i < width; i++) {
|
||||
output[output_row_offset + i] = output[last_row_offset + i];
|
||||
pred_buf_q3[i] = last_row_q3[i];
|
||||
}
|
||||
output_row_offset += MAX_SB_SIZE;
|
||||
pred_buf_q3 += MAX_SB_SIZE;
|
||||
}
|
||||
cfl->buf_height = height;
|
||||
}
|
||||
}
|
||||
|
||||
static void sum_above_row_lbd(const uint8_t *above_u, const uint8_t *above_v,
|
||||
int width, int *out_sum_u, int *out_sum_v) {
|
||||
int sum_u = 0;
|
||||
int sum_v = 0;
|
||||
for (int i = 0; i < width; i++) {
|
||||
sum_u += above_u[i];
|
||||
sum_v += above_v[i];
|
||||
}
|
||||
*out_sum_u += sum_u;
|
||||
*out_sum_v += sum_v;
|
||||
}
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
static void sum_above_row_hbd(const uint16_t *above_u, const uint16_t *above_v,
|
||||
int width, int *out_sum_u, int *out_sum_v) {
|
||||
int sum_u = 0;
|
||||
int sum_v = 0;
|
||||
for (int i = 0; i < width; i++) {
|
||||
sum_u += above_u[i];
|
||||
sum_v += above_v[i];
|
||||
}
|
||||
*out_sum_u += sum_u;
|
||||
*out_sum_v += sum_v;
|
||||
}
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
|
||||
static void sum_above_row(const MACROBLOCKD *xd, int width, int *out_sum_u,
|
||||
int *out_sum_v) {
|
||||
const struct macroblockd_plane *const pd_u = &xd->plane[AOM_PLANE_U];
|
||||
const struct macroblockd_plane *const pd_v = &xd->plane[AOM_PLANE_V];
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
if (get_bitdepth_data_path_index(xd)) {
|
||||
const uint16_t *above_u_16 =
|
||||
CONVERT_TO_SHORTPTR(pd_u->dst.buf) - pd_u->dst.stride;
|
||||
const uint16_t *above_v_16 =
|
||||
CONVERT_TO_SHORTPTR(pd_v->dst.buf) - pd_v->dst.stride;
|
||||
sum_above_row_hbd(above_u_16, above_v_16, width, out_sum_u, out_sum_v);
|
||||
return;
|
||||
}
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
const uint8_t *above_u = pd_u->dst.buf - pd_u->dst.stride;
|
||||
const uint8_t *above_v = pd_v->dst.buf - pd_v->dst.stride;
|
||||
sum_above_row_lbd(above_u, above_v, width, out_sum_u, out_sum_v);
|
||||
}
|
||||
|
||||
static void sum_left_col_lbd(const uint8_t *left_u, int u_stride,
|
||||
const uint8_t *left_v, int v_stride, int height,
|
||||
int *out_sum_u, int *out_sum_v) {
|
||||
int sum_u = 0;
|
||||
int sum_v = 0;
|
||||
for (int i = 0; i < height; i++) {
|
||||
sum_u += left_u[i * u_stride];
|
||||
sum_v += left_v[i * v_stride];
|
||||
}
|
||||
*out_sum_u += sum_u;
|
||||
*out_sum_v += sum_v;
|
||||
}
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
static void sum_left_col_hbd(const uint16_t *left_u, int u_stride,
|
||||
const uint16_t *left_v, int v_stride, int height,
|
||||
int *out_sum_u, int *out_sum_v) {
|
||||
int sum_u = 0;
|
||||
int sum_v = 0;
|
||||
for (int i = 0; i < height; i++) {
|
||||
sum_u += left_u[i * u_stride];
|
||||
sum_v += left_v[i * v_stride];
|
||||
}
|
||||
*out_sum_u += sum_u;
|
||||
*out_sum_v += sum_v;
|
||||
}
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
static void sum_left_col(const MACROBLOCKD *xd, int height, int *out_sum_u,
|
||||
int *out_sum_v) {
|
||||
const struct macroblockd_plane *const pd_u = &xd->plane[AOM_PLANE_U];
|
||||
const struct macroblockd_plane *const pd_v = &xd->plane[AOM_PLANE_V];
|
||||
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
if (get_bitdepth_data_path_index(xd)) {
|
||||
const uint16_t *left_u_16 = CONVERT_TO_SHORTPTR(pd_u->dst.buf) - 1;
|
||||
const uint16_t *left_v_16 = CONVERT_TO_SHORTPTR(pd_v->dst.buf) - 1;
|
||||
sum_left_col_hbd(left_u_16, pd_u->dst.stride, left_v_16, pd_v->dst.stride,
|
||||
height, out_sum_u, out_sum_v);
|
||||
return;
|
||||
}
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
const uint8_t *left_u = pd_u->dst.buf - 1;
|
||||
const uint8_t *left_v = pd_v->dst.buf - 1;
|
||||
sum_left_col_lbd(left_u, pd_u->dst.stride, left_v, pd_v->dst.stride, height,
|
||||
out_sum_u, out_sum_v);
|
||||
}
|
||||
|
||||
// CfL computes its own block-level DC_PRED. This is required to compute both
|
||||
// alpha_cb and alpha_cr before the prediction are computed.
|
||||
static void cfl_dc_pred(MACROBLOCKD *xd, BLOCK_SIZE plane_bsize) {
|
||||
const struct macroblockd_plane *const pd_u = &xd->plane[AOM_PLANE_U];
|
||||
const struct macroblockd_plane *const pd_v = &xd->plane[AOM_PLANE_V];
|
||||
|
||||
const uint8_t *const dst_u = pd_u->dst.buf;
|
||||
const uint8_t *const dst_v = pd_v->dst.buf;
|
||||
|
||||
const int dst_u_stride = pd_u->dst.stride;
|
||||
const int dst_v_stride = pd_v->dst.stride;
|
||||
|
||||
CFL_CTX *const cfl = xd->cfl;
|
||||
|
||||
// Compute DC_PRED until block boundary. We can't assume the neighbor will use
|
||||
|
|
@ -138,14 +169,13 @@ static void cfl_dc_pred(MACROBLOCKD *xd, BLOCK_SIZE plane_bsize) {
|
|||
int sum_u = 0;
|
||||
int sum_v = 0;
|
||||
|
||||
// Match behavior of build_intra_predictors (reconintra.c) at superblock
|
||||
// Match behavior of build_intra_predictors_high (reconintra.c) at superblock
|
||||
// boundaries:
|
||||
//
|
||||
// 127 127 127 .. 127 127 127 127 127 127
|
||||
// 129 A B .. Y Z
|
||||
// 129 C D .. W X
|
||||
// 129 E F .. U V
|
||||
// 129 G H .. S T T T T T
|
||||
// base-1 base-1 base-1 .. base-1 base-1 base-1 base-1 base-1 base-1
|
||||
// base+1 A B .. Y Z
|
||||
// base+1 C D .. W X
|
||||
// base+1 E F .. U V
|
||||
// base+1 G H .. S T T T T T
|
||||
// ..
|
||||
|
||||
#if CONFIG_CHROMA_SUB8X8
|
||||
|
|
@ -153,14 +183,11 @@ static void cfl_dc_pred(MACROBLOCKD *xd, BLOCK_SIZE plane_bsize) {
|
|||
#else
|
||||
if (xd->up_available && xd->mb_to_right_edge >= 0) {
|
||||
#endif
|
||||
// TODO(ltrudeau) replace this with DC_PRED assembly
|
||||
for (int i = 0; i < width; i++) {
|
||||
sum_u += dst_u[-dst_u_stride + i];
|
||||
sum_v += dst_v[-dst_v_stride + i];
|
||||
}
|
||||
sum_above_row(xd, width, &sum_u, &sum_v);
|
||||
} else {
|
||||
sum_u = width * 127;
|
||||
sum_v = width * 127;
|
||||
const int base = 128 << (xd->bd - 8);
|
||||
sum_u = width * (base - 1);
|
||||
sum_v = width * (base - 1);
|
||||
}
|
||||
|
||||
#if CONFIG_CHROMA_SUB8X8
|
||||
|
|
@ -168,13 +195,11 @@ static void cfl_dc_pred(MACROBLOCKD *xd, BLOCK_SIZE plane_bsize) {
|
|||
#else
|
||||
if (xd->left_available && xd->mb_to_bottom_edge >= 0) {
|
||||
#endif
|
||||
for (int i = 0; i < height; i++) {
|
||||
sum_u += dst_u[i * dst_u_stride - 1];
|
||||
sum_v += dst_v[i * dst_v_stride - 1];
|
||||
}
|
||||
sum_left_col(xd, height, &sum_u, &sum_v);
|
||||
} else {
|
||||
sum_u += height * 129;
|
||||
sum_v += height * 129;
|
||||
const int base = 128 << (xd->bd - 8);
|
||||
sum_u += height * (base + 1);
|
||||
sum_v += height * (base + 1);
|
||||
}
|
||||
|
||||
// TODO(ltrudeau) Because of max_block_wide and max_block_high, num_pel will
|
||||
|
|
@ -183,64 +208,103 @@ static void cfl_dc_pred(MACROBLOCKD *xd, BLOCK_SIZE plane_bsize) {
|
|||
cfl->dc_pred[CFL_PRED_V] = (sum_v + (num_pel >> 1)) / num_pel;
|
||||
}
|
||||
|
||||
static void cfl_compute_averages(CFL_CTX *cfl, TX_SIZE tx_size) {
|
||||
static void cfl_subtract_averages(CFL_CTX *cfl, TX_SIZE tx_size) {
|
||||
const int width = cfl->uv_width;
|
||||
const int height = cfl->uv_height;
|
||||
const int tx_height = tx_size_high[tx_size];
|
||||
const int tx_width = tx_size_wide[tx_size];
|
||||
const int stride = width >> tx_size_wide_log2[tx_size];
|
||||
const int block_row_stride = MAX_SB_SIZE << tx_size_high_log2[tx_size];
|
||||
const int num_pel_log2 =
|
||||
(tx_size_high_log2[tx_size] + tx_size_wide_log2[tx_size]);
|
||||
|
||||
// TODO(ltrudeau) Convert to uint16 for HBD support
|
||||
const uint8_t *y_pix = cfl->y_down_pix;
|
||||
// TODO(ltrudeau) Convert to uint16 for HBD support
|
||||
const uint8_t *t_y_pix;
|
||||
int *averages_q3 = cfl->y_averages_q3;
|
||||
int16_t *pred_buf_q3 = cfl->pred_buf_q3;
|
||||
|
||||
cfl_load(cfl, 0, 0, width, height);
|
||||
cfl_pad(cfl, width, height);
|
||||
|
||||
int a = 0;
|
||||
for (int b_j = 0; b_j < height; b_j += tx_height) {
|
||||
for (int b_i = 0; b_i < width; b_i += tx_width) {
|
||||
int sum = 0;
|
||||
t_y_pix = y_pix;
|
||||
int sum_q3 = 0;
|
||||
int16_t *tx_pred_buf_q3 = pred_buf_q3;
|
||||
for (int t_j = 0; t_j < tx_height; t_j++) {
|
||||
for (int t_i = b_i; t_i < b_i + tx_width; t_i++) {
|
||||
sum += t_y_pix[t_i];
|
||||
sum_q3 += tx_pred_buf_q3[t_i];
|
||||
}
|
||||
t_y_pix += MAX_SB_SIZE;
|
||||
tx_pred_buf_q3 += MAX_SB_SIZE;
|
||||
}
|
||||
averages_q3[a++] =
|
||||
((sum << 3) + (1 << (num_pel_log2 - 1))) >> num_pel_log2;
|
||||
|
||||
int avg_q3 = (sum_q3 + (1 << (num_pel_log2 - 1))) >> num_pel_log2;
|
||||
// Loss is never more than 1/2 (in Q3)
|
||||
assert(fabs((double)averages_q3[a - 1] -
|
||||
(sum / ((double)(1 << num_pel_log2))) * (1 << 3)) <= 0.5);
|
||||
assert(fabs((double)avg_q3 - (sum_q3 / ((double)(1 << num_pel_log2)))) <=
|
||||
0.5);
|
||||
|
||||
tx_pred_buf_q3 = pred_buf_q3;
|
||||
for (int t_j = 0; t_j < tx_height; t_j++) {
|
||||
for (int t_i = b_i; t_i < b_i + tx_width; t_i++) {
|
||||
tx_pred_buf_q3[t_i] -= avg_q3;
|
||||
}
|
||||
|
||||
tx_pred_buf_q3 += MAX_SB_SIZE;
|
||||
}
|
||||
}
|
||||
assert(a % stride == 0);
|
||||
y_pix += block_row_stride;
|
||||
pred_buf_q3 += block_row_stride;
|
||||
}
|
||||
|
||||
cfl->y_averages_stride = stride;
|
||||
assert(a <= MAX_NUM_TXB);
|
||||
}
|
||||
|
||||
static INLINE int cfl_idx_to_alpha(int alpha_idx, CFL_SIGN_TYPE alpha_sign,
|
||||
static INLINE int cfl_idx_to_alpha(int alpha_idx, int joint_sign,
|
||||
CFL_PRED_TYPE pred_type) {
|
||||
const int mag_idx = cfl_alpha_codes[alpha_idx][pred_type];
|
||||
const int abs_alpha_q3 = cfl_alpha_mags_q3[mag_idx];
|
||||
if (alpha_sign == CFL_SIGN_POS) {
|
||||
return abs_alpha_q3;
|
||||
} else {
|
||||
assert(abs_alpha_q3 != 0);
|
||||
assert(cfl_alpha_mags_q3[mag_idx + 1] == -abs_alpha_q3);
|
||||
return -abs_alpha_q3;
|
||||
const int alpha_sign = (pred_type == CFL_PRED_U) ? CFL_SIGN_U(joint_sign)
|
||||
: CFL_SIGN_V(joint_sign);
|
||||
if (alpha_sign == CFL_SIGN_ZERO) return 0;
|
||||
const int abs_alpha_q3 =
|
||||
(pred_type == CFL_PRED_U) ? CFL_IDX_U(alpha_idx) : CFL_IDX_V(alpha_idx);
|
||||
return (alpha_sign == CFL_SIGN_POS) ? abs_alpha_q3 + 1 : -abs_alpha_q3 - 1;
|
||||
}
|
||||
|
||||
static void cfl_build_prediction_lbd(const int16_t *pred_buf_q3, uint8_t *dst,
|
||||
int dst_stride, int width, int height,
|
||||
int alpha_q3, int dc_pred) {
|
||||
for (int j = 0; j < height; j++) {
|
||||
for (int i = 0; i < width; i++) {
|
||||
dst[i] =
|
||||
clip_pixel(get_scaled_luma_q0(alpha_q3, pred_buf_q3[i]) + dc_pred);
|
||||
}
|
||||
dst += dst_stride;
|
||||
pred_buf_q3 += MAX_SB_SIZE;
|
||||
}
|
||||
}
|
||||
|
||||
// Predict the current transform block using CfL.
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
static void cfl_build_prediction_hbd(const int16_t *pred_buf_q3, uint16_t *dst,
|
||||
int dst_stride, int width, int height,
|
||||
int alpha_q3, int dc_pred, int bit_depth) {
|
||||
for (int j = 0; j < height; j++) {
|
||||
for (int i = 0; i < width; i++) {
|
||||
dst[i] = clip_pixel_highbd(
|
||||
get_scaled_luma_q0(alpha_q3, pred_buf_q3[i]) + dc_pred, bit_depth);
|
||||
}
|
||||
dst += dst_stride;
|
||||
pred_buf_q3 += MAX_SB_SIZE;
|
||||
}
|
||||
}
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
|
||||
static void cfl_build_prediction(const int16_t *pred_buf_q3, uint8_t *dst,
|
||||
int dst_stride, int width, int height,
|
||||
int alpha_q3, int dc_pred, int use_hbd,
|
||||
int bit_depth) {
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
if (use_hbd) {
|
||||
uint16_t *dst_16 = CONVERT_TO_SHORTPTR(dst);
|
||||
cfl_build_prediction_hbd(pred_buf_q3, dst_16, dst_stride, width, height,
|
||||
alpha_q3, dc_pred, bit_depth);
|
||||
return;
|
||||
}
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
(void)use_hbd;
|
||||
(void)bit_depth;
|
||||
cfl_build_prediction_lbd(pred_buf_q3, dst, dst_stride, width, height,
|
||||
alpha_q3, dc_pred);
|
||||
}
|
||||
|
||||
void cfl_predict_block(MACROBLOCKD *const xd, uint8_t *dst, int dst_stride,
|
||||
int row, int col, TX_SIZE tx_size, int plane) {
|
||||
CFL_CTX *const cfl = xd->cfl;
|
||||
|
|
@ -249,74 +313,112 @@ void cfl_predict_block(MACROBLOCKD *const xd, uint8_t *dst, int dst_stride,
|
|||
// CfL parameters must be computed before prediction can be done.
|
||||
assert(cfl->are_parameters_computed == 1);
|
||||
|
||||
const int width = tx_size_wide[tx_size];
|
||||
const int height = tx_size_high[tx_size];
|
||||
// TODO(ltrudeau) Convert to uint16 to support HBD
|
||||
const uint8_t *y_pix = cfl->y_down_pix;
|
||||
const int16_t *pred_buf_q3 =
|
||||
cfl->pred_buf_q3 + ((row * MAX_SB_SIZE + col) << tx_size_wide_log2[0]);
|
||||
const int alpha_q3 =
|
||||
cfl_idx_to_alpha(mbmi->cfl_alpha_idx, mbmi->cfl_alpha_signs, plane - 1);
|
||||
|
||||
const int dc_pred = cfl->dc_pred[plane - 1];
|
||||
const int alpha_q3 = cfl_idx_to_alpha(
|
||||
mbmi->cfl_alpha_idx, mbmi->cfl_alpha_signs[plane - 1], plane - 1);
|
||||
cfl_build_prediction(pred_buf_q3, dst, dst_stride, tx_size_wide[tx_size],
|
||||
tx_size_high[tx_size], alpha_q3, cfl->dc_pred[plane - 1],
|
||||
get_bitdepth_data_path_index(xd), xd->bd);
|
||||
}
|
||||
|
||||
const int avg_row =
|
||||
(row << tx_size_wide_log2[0]) >> tx_size_wide_log2[tx_size];
|
||||
const int avg_col =
|
||||
(col << tx_size_high_log2[0]) >> tx_size_high_log2[tx_size];
|
||||
const int avg_q3 =
|
||||
cfl->y_averages_q3[cfl->y_averages_stride * avg_row + avg_col];
|
||||
|
||||
cfl_load(cfl, row, col, width, height);
|
||||
static void cfl_luma_subsampling_420_lbd(const uint8_t *input, int input_stride,
|
||||
int16_t *output_q3, int width,
|
||||
int height) {
|
||||
for (int j = 0; j < height; j++) {
|
||||
for (int i = 0; i < width; i++) {
|
||||
// TODO(ltrudeau) add support for HBD.
|
||||
dst[i] =
|
||||
clip_pixel(get_scaled_luma_q0(alpha_q3, y_pix[i], avg_q3) + dc_pred);
|
||||
int top = i << 1;
|
||||
int bot = top + input_stride;
|
||||
output_q3[i] = (input[top] + input[top + 1] + input[bot] + input[bot + 1])
|
||||
<< 1;
|
||||
}
|
||||
dst += dst_stride;
|
||||
y_pix += MAX_SB_SIZE;
|
||||
input += input_stride << 1;
|
||||
output_q3 += MAX_SB_SIZE;
|
||||
}
|
||||
}
|
||||
|
||||
void cfl_store(CFL_CTX *cfl, const uint8_t *input, int input_stride, int row,
|
||||
int col, TX_SIZE tx_size, BLOCK_SIZE bsize) {
|
||||
const int tx_width = tx_size_wide[tx_size];
|
||||
const int tx_height = tx_size_high[tx_size];
|
||||
const int tx_off_log2 = tx_size_wide_log2[0];
|
||||
|
||||
#if CONFIG_CHROMA_SUB8X8
|
||||
if (bsize < BLOCK_8X8) {
|
||||
// Transform cannot be smaller than
|
||||
assert(tx_width >= 4);
|
||||
assert(tx_height >= 4);
|
||||
|
||||
const int bw = block_size_wide[bsize];
|
||||
const int bh = block_size_high[bsize];
|
||||
|
||||
// For chroma_sub8x8, the CfL prediction for prediction blocks smaller than
|
||||
// 8X8 uses non chroma reference reconstructed luma pixels. To do so, we
|
||||
// combine the 4X4 non chroma reference into the CfL pixel buffers based on
|
||||
// their row and column index.
|
||||
|
||||
// The following code is adapted from the is_chroma_reference() function.
|
||||
if ((cfl->mi_row &
|
||||
0x01) // Increment the row index for odd indexed 4X4 blocks
|
||||
&& (bh == 4) // But not for 4X8 blocks
|
||||
&& cfl->subsampling_y) { // And only when chroma is subsampled
|
||||
assert(row == 0);
|
||||
row++;
|
||||
}
|
||||
|
||||
if ((cfl->mi_col &
|
||||
0x01) // Increment the col index for odd indexed 4X4 blocks
|
||||
&& (bw == 4) // But not for 8X4 blocks
|
||||
&& cfl->subsampling_x) { // And only when chroma is subsampled
|
||||
assert(col == 0);
|
||||
col++;
|
||||
static void cfl_luma_subsampling_444_lbd(const uint8_t *input, int input_stride,
|
||||
int16_t *output_q3, int width,
|
||||
int height) {
|
||||
for (int j = 0; j < height; j++) {
|
||||
for (int i = 0; i < width; i++) {
|
||||
output_q3[i] = input[i] << 3;
|
||||
}
|
||||
input += input_stride;
|
||||
output_q3 += MAX_SB_SIZE;
|
||||
}
|
||||
#else
|
||||
(void)bsize;
|
||||
#endif
|
||||
}
|
||||
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
static void cfl_luma_subsampling_420_hbd(const uint16_t *input,
|
||||
int input_stride, int16_t *output_q3,
|
||||
int width, int height) {
|
||||
for (int j = 0; j < height; j++) {
|
||||
for (int i = 0; i < width; i++) {
|
||||
int top = i << 1;
|
||||
int bot = top + input_stride;
|
||||
output_q3[i] = (input[top] + input[top + 1] + input[bot] + input[bot + 1])
|
||||
<< 1;
|
||||
}
|
||||
input += input_stride << 1;
|
||||
output_q3 += MAX_SB_SIZE;
|
||||
}
|
||||
}
|
||||
|
||||
static void cfl_luma_subsampling_444_hbd(const uint16_t *input,
|
||||
int input_stride, int16_t *output_q3,
|
||||
int width, int height) {
|
||||
for (int j = 0; j < height; j++) {
|
||||
for (int i = 0; i < width; i++) {
|
||||
output_q3[i] = input[i] << 3;
|
||||
}
|
||||
input += input_stride;
|
||||
output_q3 += MAX_SB_SIZE;
|
||||
}
|
||||
}
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
|
||||
static void cfl_luma_subsampling_420(const uint8_t *input, int input_stride,
|
||||
int16_t *output_q3, int width, int height,
|
||||
int use_hbd) {
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
if (use_hbd) {
|
||||
const uint16_t *input_16 = CONVERT_TO_SHORTPTR(input);
|
||||
cfl_luma_subsampling_420_hbd(input_16, input_stride, output_q3, width,
|
||||
height);
|
||||
return;
|
||||
}
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
(void)use_hbd;
|
||||
cfl_luma_subsampling_420_lbd(input, input_stride, output_q3, width, height);
|
||||
}
|
||||
|
||||
static void cfl_luma_subsampling_444(const uint8_t *input, int input_stride,
|
||||
int16_t *output_q3, int width, int height,
|
||||
int use_hbd) {
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
if (use_hbd) {
|
||||
uint16_t *input_16 = CONVERT_TO_SHORTPTR(input);
|
||||
cfl_luma_subsampling_444_hbd(input_16, input_stride, output_q3, width,
|
||||
height);
|
||||
return;
|
||||
}
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
(void)use_hbd;
|
||||
cfl_luma_subsampling_444_lbd(input, input_stride, output_q3, width, height);
|
||||
}
|
||||
|
||||
static INLINE void cfl_store(CFL_CTX *cfl, const uint8_t *input,
|
||||
int input_stride, int row, int col, int width,
|
||||
int height, int use_hbd) {
|
||||
const int tx_off_log2 = tx_size_wide_log2[0];
|
||||
const int sub_x = cfl->subsampling_x;
|
||||
const int sub_y = cfl->subsampling_y;
|
||||
const int store_row = row << (tx_off_log2 - sub_y);
|
||||
const int store_col = col << (tx_off_log2 - sub_x);
|
||||
const int store_height = height >> sub_y;
|
||||
const int store_width = width >> sub_x;
|
||||
|
||||
// Invalidate current parameters
|
||||
cfl->are_parameters_computed = 0;
|
||||
|
|
@ -325,30 +427,110 @@ void cfl_store(CFL_CTX *cfl, const uint8_t *input, int input_stride, int row,
|
|||
// can manage chroma overrun (e.g. when the chroma surfaces goes beyond the
|
||||
// frame boundary)
|
||||
if (col == 0 && row == 0) {
|
||||
cfl->y_width = tx_width;
|
||||
cfl->y_height = tx_height;
|
||||
cfl->buf_width = store_width;
|
||||
cfl->buf_height = store_height;
|
||||
} else {
|
||||
cfl->y_width = OD_MAXI((col << tx_off_log2) + tx_width, cfl->y_width);
|
||||
cfl->y_height = OD_MAXI((row << tx_off_log2) + tx_height, cfl->y_height);
|
||||
cfl->buf_width = OD_MAXI(store_col + store_width, cfl->buf_width);
|
||||
cfl->buf_height = OD_MAXI(store_row + store_height, cfl->buf_height);
|
||||
}
|
||||
|
||||
// Check that we will remain inside the pixel buffer.
|
||||
assert((row << tx_off_log2) + tx_height <= MAX_SB_SIZE);
|
||||
assert((col << tx_off_log2) + tx_width <= MAX_SB_SIZE);
|
||||
assert(store_row + store_height <= MAX_SB_SIZE);
|
||||
assert(store_col + store_width <= MAX_SB_SIZE);
|
||||
|
||||
// Store the input into the CfL pixel buffer
|
||||
uint8_t *y_pix = &cfl->y_pix[(row * MAX_SB_SIZE + col) << tx_off_log2];
|
||||
int16_t *pred_buf_q3 =
|
||||
cfl->pred_buf_q3 + (store_row * MAX_SB_SIZE + store_col);
|
||||
|
||||
// TODO(ltrudeau) Speedup possible by moving the downsampling to cfl_store
|
||||
for (int j = 0; j < tx_height; j++) {
|
||||
for (int i = 0; i < tx_width; i++) {
|
||||
y_pix[i] = input[i];
|
||||
}
|
||||
y_pix += MAX_SB_SIZE;
|
||||
input += input_stride;
|
||||
if (sub_y == 0 && sub_x == 0) {
|
||||
cfl_luma_subsampling_444(input, input_stride, pred_buf_q3, store_width,
|
||||
store_height, use_hbd);
|
||||
} else if (sub_y == 1 && sub_x == 1) {
|
||||
cfl_luma_subsampling_420(input, input_stride, pred_buf_q3, store_width,
|
||||
store_height, use_hbd);
|
||||
} else {
|
||||
// TODO(ltrudeau) add support for 4:2:2
|
||||
assert(0); // Unsupported chroma subsampling
|
||||
}
|
||||
}
|
||||
|
||||
#if CONFIG_CHROMA_SUB8X8
|
||||
// Adjust the row and column of blocks smaller than 8X8, as chroma-referenced
|
||||
// and non-chroma-referenced blocks are stored together in the CfL buffer.
|
||||
static INLINE void sub8x8_adjust_offset(const CFL_CTX *cfl, int *row_out,
|
||||
int *col_out) {
|
||||
// Increment row index for bottom: 8x4, 16x4 or both bottom 4x4s.
|
||||
if ((cfl->mi_row & 0x01) && cfl->subsampling_y) {
|
||||
assert(*row_out == 0);
|
||||
(*row_out)++;
|
||||
}
|
||||
|
||||
// Increment col index for right: 4x8, 4x16 or both right 4x4s.
|
||||
if ((cfl->mi_col & 0x01) && cfl->subsampling_x) {
|
||||
assert(*col_out == 0);
|
||||
(*col_out)++;
|
||||
}
|
||||
}
|
||||
#if CONFIG_DEBUG
|
||||
static INLINE void sub8x8_set_val(CFL_CTX *cfl, int row, int col, int val_high,
|
||||
int val_wide) {
|
||||
for (int val_r = 0; val_r < val_high; val_r++) {
|
||||
assert(row + val_r < CFL_SUB8X8_VAL_MI_SIZE);
|
||||
int row_off = (row + val_r) * CFL_SUB8X8_VAL_MI_SIZE;
|
||||
for (int val_c = 0; val_c < val_wide; val_c++) {
|
||||
assert(col + val_c < CFL_SUB8X8_VAL_MI_SIZE);
|
||||
assert(cfl->sub8x8_val[row_off + col + val_c] == 0);
|
||||
cfl->sub8x8_val[row_off + col + val_c]++;
|
||||
}
|
||||
}
|
||||
}
|
||||
#endif // CONFIG_DEBUG
|
||||
#endif // CONFIG_CHROMA_SUB8X8
|
||||
|
||||
void cfl_store_tx(MACROBLOCKD *const xd, int row, int col, TX_SIZE tx_size,
|
||||
BLOCK_SIZE bsize) {
|
||||
CFL_CTX *const cfl = xd->cfl;
|
||||
struct macroblockd_plane *const pd = &xd->plane[AOM_PLANE_Y];
|
||||
uint8_t *dst =
|
||||
&pd->dst.buf[(row * pd->dst.stride + col) << tx_size_wide_log2[0]];
|
||||
(void)bsize;
|
||||
#if CONFIG_CHROMA_SUB8X8
|
||||
|
||||
if (block_size_high[bsize] == 4 || block_size_wide[bsize] == 4) {
|
||||
// Only dimensions of size 4 can have an odd offset.
|
||||
assert(!((col & 1) && tx_size_wide[tx_size] != 4));
|
||||
assert(!((row & 1) && tx_size_high[tx_size] != 4));
|
||||
sub8x8_adjust_offset(cfl, &row, &col);
|
||||
#if CONFIG_DEBUG
|
||||
sub8x8_set_val(cfl, row, col, tx_size_high_unit[tx_size],
|
||||
tx_size_wide_unit[tx_size]);
|
||||
#endif // CONFIG_DEBUG
|
||||
}
|
||||
#endif
|
||||
cfl_store(cfl, dst, pd->dst.stride, row, col, tx_size_wide[tx_size],
|
||||
tx_size_high[tx_size], get_bitdepth_data_path_index(xd));
|
||||
}
|
||||
|
||||
void cfl_store_block(MACROBLOCKD *const xd, BLOCK_SIZE bsize, TX_SIZE tx_size) {
|
||||
CFL_CTX *const cfl = xd->cfl;
|
||||
struct macroblockd_plane *const pd = &xd->plane[AOM_PLANE_Y];
|
||||
int row = 0;
|
||||
int col = 0;
|
||||
#if CONFIG_CHROMA_SUB8X8
|
||||
bsize = AOMMAX(BLOCK_4X4, bsize);
|
||||
if (block_size_high[bsize] == 4 || block_size_wide[bsize] == 4) {
|
||||
sub8x8_adjust_offset(cfl, &row, &col);
|
||||
#if CONFIG_DEBUG
|
||||
sub8x8_set_val(cfl, row, col, mi_size_high[bsize], mi_size_wide[bsize]);
|
||||
#endif // CONFIG_DEBUG
|
||||
}
|
||||
#endif // CONFIG_CHROMA_SUB8X8
|
||||
const int width = max_intra_block_width(xd, bsize, AOM_PLANE_Y, tx_size);
|
||||
const int height = max_intra_block_height(xd, bsize, AOM_PLANE_Y, tx_size);
|
||||
cfl_store(cfl, pd->dst.buf, pd->dst.stride, row, col, width, height,
|
||||
get_bitdepth_data_path_index(xd));
|
||||
}
|
||||
|
||||
void cfl_compute_parameters(MACROBLOCKD *const xd, TX_SIZE tx_size) {
|
||||
CFL_CTX *const cfl = xd->cfl;
|
||||
MB_MODE_INFO *mbmi = &xd->mi[0]->mbmi;
|
||||
|
|
@ -359,6 +541,16 @@ void cfl_compute_parameters(MACROBLOCKD *const xd, TX_SIZE tx_size) {
|
|||
#if CONFIG_CHROMA_SUB8X8
|
||||
const BLOCK_SIZE plane_bsize = AOMMAX(
|
||||
BLOCK_4X4, get_plane_block_size(mbmi->sb_type, &xd->plane[AOM_PLANE_U]));
|
||||
#if CONFIG_DEBUG
|
||||
if (mbmi->sb_type < BLOCK_8X8) {
|
||||
for (int val_r = 0; val_r < mi_size_high[mbmi->sb_type]; val_r++) {
|
||||
for (int val_c = 0; val_c < mi_size_wide[mbmi->sb_type]; val_c++) {
|
||||
assert(cfl->sub8x8_val[val_r * CFL_SUB8X8_VAL_MI_SIZE + val_c] == 1);
|
||||
}
|
||||
}
|
||||
cfl_clear_sub8x8_val(cfl);
|
||||
}
|
||||
#endif // CONFIG_DEBUG
|
||||
#else
|
||||
const BLOCK_SIZE plane_bsize =
|
||||
get_plane_block_size(mbmi->sb_type, &xd->plane[AOM_PLANE_U]);
|
||||
|
|
@ -368,17 +560,10 @@ void cfl_compute_parameters(MACROBLOCKD *const xd, TX_SIZE tx_size) {
|
|||
cfl->uv_height =
|
||||
max_intra_block_height(xd, plane_bsize, AOM_PLANE_U, tx_size);
|
||||
|
||||
#if CONFIG_DEBUG
|
||||
if (mbmi->sb_type >= BLOCK_8X8) {
|
||||
assert(cfl->y_width <= cfl->uv_width << cfl->subsampling_x);
|
||||
assert(cfl->y_height <= cfl->uv_height << cfl->subsampling_y);
|
||||
}
|
||||
#endif
|
||||
assert(cfl->buf_width <= cfl->uv_width);
|
||||
assert(cfl->buf_height <= cfl->uv_height);
|
||||
|
||||
// Compute block-level DC_PRED for both chromatic planes.
|
||||
// DC_PRED replaces beta in the linear model.
|
||||
cfl_dc_pred(xd, plane_bsize);
|
||||
// Compute transform-level average on reconstructed luma input.
|
||||
cfl_compute_averages(cfl, tx_size);
|
||||
cfl_subtract_averages(cfl, tx_size);
|
||||
cfl->are_parameters_computed = 1;
|
||||
}
|
||||
|
|
|
|||
75
third_party/aom/av1/common/cfl.h
vendored
75
third_party/aom/av1/common/cfl.h
vendored
|
|
@ -12,79 +12,20 @@
|
|||
#ifndef AV1_COMMON_CFL_H_
|
||||
#define AV1_COMMON_CFL_H_
|
||||
|
||||
#include <assert.h>
|
||||
#include "av1/common/blockd.h"
|
||||
|
||||
#include "av1/common/enums.h"
|
||||
|
||||
// Forward declaration of AV1_COMMON, in order to avoid creating a cyclic
|
||||
// dependency by importing av1/common/onyxc_int.h
|
||||
typedef struct AV1Common AV1_COMMON;
|
||||
|
||||
// Forward declaration of MACROBLOCK, in order to avoid creating a cyclic
|
||||
// dependency by importing av1/common/blockd.h
|
||||
typedef struct macroblockd MACROBLOCKD;
|
||||
|
||||
typedef struct {
|
||||
// Pixel buffer containing the luma pixels used as prediction for chroma
|
||||
// TODO(ltrudeau) Convert to uint16 for HBD support
|
||||
uint8_t y_pix[MAX_SB_SQUARE];
|
||||
|
||||
// Pixel buffer containing the downsampled luma pixels used as prediction for
|
||||
// chroma
|
||||
// TODO(ltrudeau) Convert to uint16 for HBD support
|
||||
uint8_t y_down_pix[MAX_SB_SQUARE];
|
||||
|
||||
// Height and width of the luma prediction block currently in the pixel buffer
|
||||
int y_height, y_width;
|
||||
|
||||
// Height and width of the chroma prediction block currently associated with
|
||||
// this context
|
||||
int uv_height, uv_width;
|
||||
|
||||
// Transform level averages of the luma reconstructed values over the entire
|
||||
// prediction unit
|
||||
// Fixed point y_averages is Q12.3:
|
||||
// * Worst case division is 1/1024
|
||||
// * Max error will be 1/16th.
|
||||
// Note: 3 is chosen so that y_averages fits in 15 bits when 12 bit input is
|
||||
// used
|
||||
int y_averages_q3[MAX_NUM_TXB];
|
||||
int y_averages_stride;
|
||||
|
||||
int are_parameters_computed;
|
||||
|
||||
// Chroma subsampling
|
||||
int subsampling_x, subsampling_y;
|
||||
|
||||
// Block level DC_PRED for each chromatic plane
|
||||
int dc_pred[CFL_PRED_PLANES];
|
||||
|
||||
// The rate associated with each alpha codeword
|
||||
int costs[CFL_ALPHABET_SIZE];
|
||||
|
||||
int mi_row, mi_col;
|
||||
} CFL_CTX;
|
||||
|
||||
static const int cfl_alpha_mags_q3[CFL_MAGS_SIZE] = { 0, 1, -1, 2, -2, 4, -4 };
|
||||
|
||||
static const int cfl_alpha_codes[CFL_ALPHABET_SIZE][CFL_PRED_PLANES] = {
|
||||
// barrbrain's simple 1D quant ordered by subset 3 likelihood
|
||||
{ 0, 0 }, { 1, 1 }, { 3, 0 }, { 3, 3 }, { 1, 0 }, { 3, 1 },
|
||||
{ 5, 5 }, { 0, 1 }, { 5, 3 }, { 5, 0 }, { 3, 5 }, { 1, 3 },
|
||||
{ 0, 3 }, { 5, 1 }, { 1, 5 }, { 0, 5 }
|
||||
};
|
||||
|
||||
static INLINE int get_scaled_luma_q0(int alpha_q3, int y_pix, int avg_q3) {
|
||||
return (alpha_q3 * ((y_pix << 3) - avg_q3) + 32) >> 6;
|
||||
static INLINE int get_scaled_luma_q0(int alpha_q3, int16_t pred_buf_q3) {
|
||||
int scaled_luma_q6 = alpha_q3 * pred_buf_q3;
|
||||
return ROUND_POWER_OF_TWO_SIGNED(scaled_luma_q6, 6);
|
||||
}
|
||||
|
||||
void cfl_init(CFL_CTX *cfl, AV1_COMMON *cm);
|
||||
|
||||
void cfl_predict_block(MACROBLOCKD *const xd, uint8_t *dst, int dst_stride,
|
||||
int row, int col, TX_SIZE tx_size, int plane);
|
||||
|
||||
void cfl_store(CFL_CTX *cfl, const uint8_t *input, int input_stride, int row,
|
||||
int col, TX_SIZE tx_size, BLOCK_SIZE bsize);
|
||||
void cfl_store_block(MACROBLOCKD *const xd, BLOCK_SIZE bsize, TX_SIZE tx_size);
|
||||
|
||||
void cfl_store_tx(MACROBLOCKD *const xd, int row, int col, TX_SIZE tx_size,
|
||||
BLOCK_SIZE bsize);
|
||||
|
||||
void cfl_compute_parameters(MACROBLOCKD *const xd, TX_SIZE tx_size);
|
||||
|
||||
|
|
|
|||
12
third_party/aom/av1/common/clpf_simd.h
vendored
12
third_party/aom/av1/common/clpf_simd.h
vendored
|
|
@ -10,10 +10,20 @@
|
|||
*/
|
||||
|
||||
#include "./av1_rtcd.h"
|
||||
#include "./cdef_simd.h"
|
||||
#include "aom_ports/bitops.h"
|
||||
#include "aom_ports/mem.h"
|
||||
|
||||
// sign(a-b) * min(abs(a-b), max(0, threshold - (abs(a-b) >> adjdamp)))
|
||||
SIMD_INLINE v128 constrain16(v128 a, v128 b, unsigned int threshold,
|
||||
unsigned int adjdamp) {
|
||||
v128 diff = v128_sub_16(a, b);
|
||||
const v128 sign = v128_shr_n_s16(diff, 15);
|
||||
diff = v128_abs_s16(diff);
|
||||
const v128 s =
|
||||
v128_ssub_u16(v128_dup_16(threshold), v128_shr_u16(diff, adjdamp));
|
||||
return v128_xor(v128_add_16(sign, v128_min_s16(diff, s)), sign);
|
||||
}
|
||||
|
||||
// sign(a - b) * min(abs(a - b), max(0, strength - (abs(a - b) >> adjdamp)))
|
||||
SIMD_INLINE v128 constrain(v256 a, v256 b, unsigned int strength,
|
||||
unsigned int adjdamp) {
|
||||
|
|
|
|||
4
third_party/aom/av1/common/common.h
vendored
4
third_party/aom/av1/common/common.h
vendored
|
|
@ -50,10 +50,6 @@ static INLINE int get_unsigned_bits(unsigned int num_values) {
|
|||
|
||||
#define CHECK_MEM_ERROR(cm, lval, expr) \
|
||||
AOM_CHECK_MEM_ERROR(&cm->error, lval, expr)
|
||||
// TODO(yaowu: validate the usage of these codes or develop new ones.)
|
||||
#define AV1_SYNC_CODE_0 0x49
|
||||
#define AV1_SYNC_CODE_1 0x83
|
||||
#define AV1_SYNC_CODE_2 0x43
|
||||
|
||||
#define AOM_FRAME_MARKER 0x2
|
||||
|
||||
|
|
|
|||
934
third_party/aom/av1/common/common_data.h
vendored
934
third_party/aom/av1/common/common_data.h
vendored
File diff suppressed because it is too large
Load diff
738
third_party/aom/av1/common/convolve.c
vendored
738
third_party/aom/av1/common/convolve.c
vendored
File diff suppressed because it is too large
Load diff
91
third_party/aom/av1/common/convolve.h
vendored
91
third_party/aom/av1/common/convolve.h
vendored
|
|
@ -47,15 +47,49 @@ static INLINE ConvolveParams get_conv_params(int ref, int do_average,
|
|||
conv_params.do_post_rounding = 0;
|
||||
return conv_params;
|
||||
}
|
||||
|
||||
#if CONFIG_DUAL_FILTER && USE_EXTRA_FILTER
|
||||
static INLINE void av1_convolve_filter_params_fixup_1212(
|
||||
const InterpFilterParams *params_x, InterpFilterParams *params_y) {
|
||||
if (params_x->interp_filter == MULTITAP_SHARP &&
|
||||
params_y->interp_filter == MULTITAP_SHARP) {
|
||||
// Avoid two directions both using 12-tap filter.
|
||||
// This will reduce hardware implementation cost.
|
||||
*params_y = av1_get_interp_filter_params(EIGHTTAP_SHARP);
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
static INLINE void av1_get_convolve_filter_params(
|
||||
InterpFilters interp_filters, int avoid_1212, InterpFilterParams *params_x,
|
||||
InterpFilterParams *params_y) {
|
||||
#if CONFIG_DUAL_FILTER
|
||||
InterpFilter filter_x = av1_extract_interp_filter(interp_filters, 1);
|
||||
InterpFilter filter_y = av1_extract_interp_filter(interp_filters, 0);
|
||||
#else
|
||||
InterpFilter filter_x = av1_extract_interp_filter(interp_filters, 0);
|
||||
InterpFilter filter_y = av1_extract_interp_filter(interp_filters, 0);
|
||||
#endif
|
||||
|
||||
*params_x = av1_get_interp_filter_params(filter_x);
|
||||
*params_y = av1_get_interp_filter_params(filter_y);
|
||||
|
||||
if (avoid_1212) {
|
||||
#if CONFIG_DUAL_FILTER && USE_EXTRA_FILTER
|
||||
convolve_filter_params_fixup_1212(params_x, params_y);
|
||||
#endif
|
||||
}
|
||||
}
|
||||
|
||||
struct AV1Common;
|
||||
void av1_convolve_init(struct AV1Common *cm);
|
||||
|
||||
#if CONFIG_CONVOLVE_ROUND
|
||||
void av1_convolve_2d_facade(const uint8_t *src, int src_stride, uint8_t *dst,
|
||||
int dst_stride, int w, int h,
|
||||
const InterpFilter *interp_filter,
|
||||
const int subpel_x_q4, int x_step_q4,
|
||||
const int subpel_y_q4, int y_step_q4,
|
||||
ConvolveParams *conv_params);
|
||||
InterpFilters interp_filters, const int subpel_x_q4,
|
||||
int x_step_q4, const int subpel_y_q4, int y_step_q4,
|
||||
int scaled, ConvolveParams *conv_params);
|
||||
|
||||
static INLINE ConvolveParams get_conv_params_no_round(int ref, int do_average,
|
||||
int plane, int32_t *dst,
|
||||
|
|
@ -80,63 +114,42 @@ static INLINE ConvolveParams get_conv_params_no_round(int ref, int do_average,
|
|||
#if CONFIG_HIGHBITDEPTH
|
||||
void av1_highbd_convolve_2d_facade(const uint8_t *src8, int src_stride,
|
||||
uint8_t *dst, int dst_stride, int w, int h,
|
||||
const InterpFilter *interp_filter,
|
||||
InterpFilters interp_filters,
|
||||
const int subpel_x_q4, int x_step_q4,
|
||||
const int subpel_y_q4, int y_step_q4,
|
||||
ConvolveParams *conv_params, int bd);
|
||||
int scaled, ConvolveParams *conv_params,
|
||||
int bd);
|
||||
#endif
|
||||
#endif // CONFIG_CONVOLVE_ROUND
|
||||
|
||||
void av1_convolve(const uint8_t *src, int src_stride, uint8_t *dst,
|
||||
int dst_stride, int w, int h,
|
||||
#if CONFIG_DUAL_FILTER
|
||||
const InterpFilter *interp_filter,
|
||||
#else
|
||||
const InterpFilter interp_filter,
|
||||
#endif
|
||||
int dst_stride, int w, int h, InterpFilters interp_filters,
|
||||
const int subpel_x, int xstep, const int subpel_y, int ystep,
|
||||
ConvolveParams *conv_params);
|
||||
|
||||
void av1_convolve_c(const uint8_t *src, int src_stride, uint8_t *dst,
|
||||
int dst_stride, int w, int h,
|
||||
#if CONFIG_DUAL_FILTER
|
||||
const InterpFilter *interp_filter,
|
||||
#else
|
||||
const InterpFilter interp_filter,
|
||||
#endif
|
||||
int dst_stride, int w, int h, InterpFilters interp_filters,
|
||||
const int subpel_x, int xstep, const int subpel_y,
|
||||
int ystep, ConvolveParams *conv_params);
|
||||
|
||||
void av1_convolve_scale(const uint8_t *src, int src_stride, uint8_t *dst,
|
||||
int dst_stride, int w, int h,
|
||||
#if CONFIG_DUAL_FILTER
|
||||
const InterpFilter *interp_filter,
|
||||
#else
|
||||
const InterpFilter interp_filter,
|
||||
#endif
|
||||
const int subpel_x, int xstep, const int subpel_y,
|
||||
int ystep, ConvolveParams *conv_params);
|
||||
InterpFilters interp_filters, const int subpel_x,
|
||||
int xstep, const int subpel_y, int ystep,
|
||||
ConvolveParams *conv_params);
|
||||
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
void av1_highbd_convolve(const uint8_t *src, int src_stride, uint8_t *dst,
|
||||
int dst_stride, int w, int h,
|
||||
#if CONFIG_DUAL_FILTER
|
||||
const InterpFilter *interp_filter,
|
||||
#else
|
||||
const InterpFilter interp_filter,
|
||||
#endif
|
||||
const int subpel_x, int xstep, const int subpel_y,
|
||||
int ystep, int avg, int bd);
|
||||
InterpFilters interp_filters, const int subpel_x,
|
||||
int xstep, const int subpel_y, int ystep, int avg,
|
||||
int bd);
|
||||
|
||||
void av1_highbd_convolve_scale(const uint8_t *src, int src_stride, uint8_t *dst,
|
||||
int dst_stride, int w, int h,
|
||||
#if CONFIG_DUAL_FILTER
|
||||
const InterpFilter *interp_filter,
|
||||
#else
|
||||
const InterpFilter interp_filter,
|
||||
#endif // CONFIG_DUAL_FILTER
|
||||
const int subpel_x, int xstep,
|
||||
const int subpel_y, int ystep, int avg, int bd);
|
||||
InterpFilters interp_filters, const int subpel_x,
|
||||
int xstep, const int subpel_y, int ystep,
|
||||
int avg, int bd);
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
|
||||
#ifdef __cplusplus
|
||||
|
|
|
|||
3742
third_party/aom/av1/common/daala_tx.c
vendored
3742
third_party/aom/av1/common/daala_tx.c
vendored
File diff suppressed because it is too large
Load diff
42
third_party/aom/av1/common/daala_tx.h
vendored
42
third_party/aom/av1/common/daala_tx.h
vendored
|
|
@ -1,13 +1,53 @@
|
|||
#ifndef AOM_DSP_DAALA_TX_H_
|
||||
#define AOM_DSP_DAALA_TX_H_
|
||||
|
||||
#include "aom_dsp/aom_dsp_common.h"
|
||||
#include "av1/common/odintrin.h"
|
||||
|
||||
void daala_fdct4(const tran_low_t *input, tran_low_t *output);
|
||||
void daala_idct4(const tran_low_t *input, tran_low_t *output);
|
||||
void daala_fdst4(const tran_low_t *input, tran_low_t *output);
|
||||
void daala_idst4(const tran_low_t *input, tran_low_t *output);
|
||||
void daala_idtx4(const tran_low_t *input, tran_low_t *output);
|
||||
void daala_fdct8(const tran_low_t *input, tran_low_t *output);
|
||||
void daala_idct8(const tran_low_t *input, tran_low_t *output);
|
||||
void daala_fdst8(const tran_low_t *input, tran_low_t *output);
|
||||
void daala_idst8(const tran_low_t *input, tran_low_t *output);
|
||||
void daala_idtx8(const tran_low_t *input, tran_low_t *output);
|
||||
void daala_fdct16(const tran_low_t *input, tran_low_t *output);
|
||||
void daala_idct16(const tran_low_t *input, tran_low_t *output);
|
||||
void daala_fdst16(const tran_low_t *input, tran_low_t *output);
|
||||
void daala_idst16(const tran_low_t *input, tran_low_t *output);
|
||||
void daala_idtx16(const tran_low_t *input, tran_low_t *output);
|
||||
void daala_fdct32(const tran_low_t *input, tran_low_t *output);
|
||||
void daala_idct32(const tran_low_t *input, tran_low_t *output);
|
||||
void daala_fdst32(const tran_low_t *input, tran_low_t *output);
|
||||
void daala_idst32(const tran_low_t *input, tran_low_t *output);
|
||||
void daala_idtx32(const tran_low_t *input, tran_low_t *output);
|
||||
#if CONFIG_TX64X64
|
||||
void daala_fdct64(const tran_low_t *input, tran_low_t *output);
|
||||
void daala_idct64(const tran_low_t *input, tran_low_t *output);
|
||||
void daala_fdst64(const tran_low_t *input, tran_low_t *output);
|
||||
void daala_idst64(const tran_low_t *input, tran_low_t *output);
|
||||
void daala_idtx64(const tran_low_t *input, tran_low_t *output);
|
||||
#endif
|
||||
|
||||
void od_bin_fdct4(od_coeff y[4], const od_coeff *x, int xstride);
|
||||
void od_bin_idct4(od_coeff *x, int xstride, const od_coeff y[4]);
|
||||
void od_bin_fdst4(od_coeff y[4], const od_coeff *x, int xstride);
|
||||
void od_bin_idst4(od_coeff *x, int xstride, const od_coeff y[4]);
|
||||
void od_bin_fdct8(od_coeff y[8], const od_coeff *x, int xstride);
|
||||
void od_bin_idct8(od_coeff *x, int xstride, const od_coeff y[8]);
|
||||
void od_bin_fdst8(od_coeff y[8], const od_coeff *x, int xstride);
|
||||
void od_bin_idst8(od_coeff *x, int xstride, const od_coeff y[8]);
|
||||
|
||||
void od_bin_fdct16(od_coeff y[16], const od_coeff *x, int xstride);
|
||||
void od_bin_idct16(od_coeff *x, int xstride, const od_coeff y[16]);
|
||||
void od_bin_fdst16(od_coeff y[16], const od_coeff *x, int xstride);
|
||||
void od_bin_idst16(od_coeff *x, int xstride, const od_coeff y[16]);
|
||||
void od_bin_fdct32(od_coeff y[32], const od_coeff *x, int xstride);
|
||||
void od_bin_idct32(od_coeff *x, int xstride, const od_coeff y[32]);
|
||||
#if CONFIG_TX64X64
|
||||
void od_bin_fdct64(od_coeff y[64], const od_coeff *x, int xstride);
|
||||
void od_bin_idct64(od_coeff *x, int xstride, const od_coeff y[64]);
|
||||
#endif
|
||||
#endif
|
||||
|
|
|
|||
5464
third_party/aom/av1/common/entropy.c
vendored
5464
third_party/aom/av1/common/entropy.c
vendored
File diff suppressed because it is too large
Load diff
84
third_party/aom/av1/common/entropy.h
vendored
84
third_party/aom/av1/common/entropy.h
vendored
|
|
@ -28,8 +28,7 @@ extern "C" {
|
|||
#define GROUP_DIFF_UPDATE_PROB 252
|
||||
|
||||
#if CONFIG_Q_ADAPT_PROBS
|
||||
#define QCTX_BIN_BITS 2
|
||||
#define QCTX_BINS (1 << QCTX_BIN_BITS)
|
||||
#define TOKEN_CDF_Q_CTXS 4
|
||||
#endif // CONFIG_Q_ADAPT_PROBS
|
||||
|
||||
// Coefficient token alphabet
|
||||
|
|
@ -61,8 +60,25 @@ extern "C" {
|
|||
|
||||
#if CONFIG_LV_MAP
|
||||
#define TXB_SKIP_CONTEXTS 13
|
||||
#define SIG_COEF_CONTEXTS 20
|
||||
|
||||
#if CONFIG_CTX1D
|
||||
#define EOB_COEF_CONTEXTS_2D 25
|
||||
#define EOB_COEF_CONTEXTS_1D 25
|
||||
#define EOB_COEF_CONTEXTS \
|
||||
(EOB_COEF_CONTEXTS_2D + EOB_COEF_CONTEXTS_1D + EOB_COEF_CONTEXTS_1D)
|
||||
#else // CONFIG_CTX1D
|
||||
#define EOB_COEF_CONTEXTS 25
|
||||
#endif // CONFIG_CTX1D
|
||||
|
||||
#if CONFIG_EXT_TX
|
||||
#define SIG_COEF_CONTEXTS_2D 16
|
||||
#define SIG_COEF_CONTEXTS_1D 16
|
||||
#define SIG_COEF_CONTEXTS \
|
||||
(SIG_COEF_CONTEXTS_2D + SIG_COEF_CONTEXTS_1D + SIG_COEF_CONTEXTS_1D)
|
||||
#else // CONFIG_EXT_TX
|
||||
#define SIG_COEF_CONTEXTS_2D 16
|
||||
#define SIG_COEF_CONTEXTS 16
|
||||
#endif // CONFIG_EXT_TX
|
||||
#define COEFF_BASE_CONTEXTS 42
|
||||
#define DC_SIGN_CONTEXTS 3
|
||||
|
||||
|
|
@ -71,10 +87,26 @@ extern "C" {
|
|||
#define LEVEL_CONTEXTS (BR_TMP_OFFSET * BR_REF_CAT)
|
||||
|
||||
#define NUM_BASE_LEVELS 2
|
||||
#define COEFF_BASE_RANGE (15 - NUM_BASE_LEVELS)
|
||||
#define COEFF_BASE_RANGE (16 - NUM_BASE_LEVELS)
|
||||
#define BASE_RANGE_SETS 3
|
||||
|
||||
#define COEFF_CONTEXT_BITS 6
|
||||
#define COEFF_CONTEXT_MASK ((1 << COEFF_CONTEXT_BITS) - 1)
|
||||
|
||||
#define BASE_CONTEXT_POSITION_NUM 12
|
||||
|
||||
#if CONFIG_CTX1D
|
||||
#define EMPTY_LINE_CONTEXTS 5
|
||||
#define HV_EOB_CONTEXTS 24
|
||||
#endif // CONFIG_CTX1D
|
||||
|
||||
typedef enum TX_CLASS {
|
||||
TX_CLASS_2D = 0,
|
||||
TX_CLASS_HORIZ = 1,
|
||||
TX_CLASS_VERT = 2,
|
||||
TX_CLASSES = 3,
|
||||
} TX_CLASS;
|
||||
|
||||
#endif
|
||||
|
||||
DECLARE_ALIGNED(16, extern const uint8_t, av1_pt_energy_class[ENTROPY_TOKENS]);
|
||||
|
|
@ -169,26 +201,19 @@ static INLINE int av1_get_cat6_extrabits_size(TX_SIZE tx_size,
|
|||
distinct bands). */
|
||||
|
||||
#define COEFF_CONTEXTS 6
|
||||
#define BLOCKZ_CONTEXTS 3
|
||||
#define COEFF_CONTEXTS0 3 // for band 0
|
||||
#define BAND_COEFF_CONTEXTS(band) \
|
||||
((band) == 0 ? COEFF_CONTEXTS0 : COEFF_CONTEXTS)
|
||||
|
||||
// #define ENTROPY_STATS
|
||||
|
||||
typedef unsigned int av1_coeff_count[REF_TYPES][COEF_BANDS][COEFF_CONTEXTS]
|
||||
[ENTROPY_TOKENS];
|
||||
typedef unsigned int av1_coeff_stats[REF_TYPES][COEF_BANDS][COEFF_CONTEXTS]
|
||||
[ENTROPY_NODES][2];
|
||||
|
||||
#define SUBEXP_PARAM 4 /* Subexponential code parameter */
|
||||
#define MODULUS_PARAM 13 /* Modulus parameter */
|
||||
|
||||
struct AV1Common;
|
||||
struct frame_contexts;
|
||||
void av1_default_coef_probs(struct AV1Common *cm);
|
||||
#if CONFIG_LV_MAP
|
||||
void av1_adapt_coef_probs(struct AV1Common *cm);
|
||||
void av1_adapt_coef_cdfs(struct AV1Common *cm, struct frame_contexts *pre_fc);
|
||||
#endif // CONFIG_LV_MAP
|
||||
|
||||
// This is the index in the scan order beyond which all coefficients for
|
||||
// 8x8 transform and above are in the top band.
|
||||
|
|
@ -221,26 +246,13 @@ static INLINE const uint8_t *get_band_translate(TX_SIZE tx_size) {
|
|||
|
||||
#define UNCONSTRAINED_NODES 3
|
||||
|
||||
#define PIVOT_NODE 2 // which node is pivot
|
||||
|
||||
#define MODEL_NODES (ENTROPY_NODES - UNCONSTRAINED_NODES)
|
||||
#define TAIL_NODES (MODEL_NODES + 1)
|
||||
extern const aom_tree_index av1_coef_con_tree[TREE_SIZE(ENTROPY_TOKENS)];
|
||||
extern const aom_prob av1_pareto8_full[COEFF_PROB_MODELS][MODEL_NODES];
|
||||
|
||||
typedef aom_prob av1_coeff_probs_model[REF_TYPES][COEF_BANDS][COEFF_CONTEXTS]
|
||||
[UNCONSTRAINED_NODES];
|
||||
|
||||
typedef unsigned int av1_coeff_count_model[REF_TYPES][COEF_BANDS]
|
||||
[COEFF_CONTEXTS]
|
||||
[UNCONSTRAINED_NODES + 1];
|
||||
|
||||
void av1_model_to_full_probs(const aom_prob *model, aom_prob *full);
|
||||
|
||||
typedef aom_cdf_prob coeff_cdf_model[REF_TYPES][COEF_BANDS][COEFF_CONTEXTS]
|
||||
[CDF_SIZE(ENTROPY_TOKENS)];
|
||||
typedef aom_prob av1_blockz_probs_model[REF_TYPES][BLOCKZ_CONTEXTS];
|
||||
typedef unsigned int av1_blockz_count_model[REF_TYPES][BLOCKZ_CONTEXTS][2];
|
||||
extern const aom_cdf_prob av1_pareto8_token_probs[COEFF_PROB_MODELS]
|
||||
[ENTROPY_TOKENS - 2];
|
||||
extern const aom_cdf_prob av1_pareto8_tail_probs[COEFF_PROB_MODELS]
|
||||
|
|
@ -314,6 +326,16 @@ static INLINE int get_entropy_context(TX_SIZE tx_size, const ENTROPY_CONTEXT *a,
|
|||
left_ec = !!(*(const uint64_t *)l | *(const uint64_t *)(l + 8) |
|
||||
*(const uint64_t *)(l + 16) | *(const uint64_t *)(l + 24));
|
||||
break;
|
||||
case TX_32X64:
|
||||
above_ec = !!(*(const uint64_t *)a | *(const uint64_t *)(a + 8));
|
||||
left_ec = !!(*(const uint64_t *)l | *(const uint64_t *)(l + 8) |
|
||||
*(const uint64_t *)(l + 16) | *(const uint64_t *)(l + 24));
|
||||
break;
|
||||
case TX_64X32:
|
||||
above_ec = !!(*(const uint64_t *)a | *(const uint64_t *)(a + 8) |
|
||||
*(const uint64_t *)(a + 16) | *(const uint64_t *)(a + 24));
|
||||
left_ec = !!(*(const uint64_t *)l | *(const uint64_t *)(l + 8));
|
||||
break;
|
||||
#endif // CONFIG_TX64X64
|
||||
#if CONFIG_RECT_TX_EXT && (CONFIG_EXT_TX || CONFIG_VAR_TX)
|
||||
case TX_4X16:
|
||||
|
|
@ -384,6 +406,14 @@ static INLINE int get_entropy_context(TX_SIZE tx_size, const ENTROPY_CONTEXT *a,
|
|||
above_ec = !!(*(const uint64_t *)a | *(const uint64_t *)(a + 8));
|
||||
left_ec = !!(*(const uint64_t *)l | *(const uint64_t *)(l + 8));
|
||||
break;
|
||||
case TX_32X64:
|
||||
above_ec = !!*(const uint64_t *)a;
|
||||
left_ec = !!(*(const uint64_t *)l | *(const uint64_t *)(l + 8));
|
||||
break;
|
||||
case TX_64X32:
|
||||
above_ec = !!(*(const uint64_t *)a | *(const uint64_t *)(a + 8));
|
||||
left_ec = !!*(const uint64_t *)l;
|
||||
break;
|
||||
#endif // CONFIG_TX64X64
|
||||
#if CONFIG_RECT_TX_EXT && (CONFIG_EXT_TX || CONFIG_VAR_TX)
|
||||
case TX_4X16:
|
||||
|
|
@ -414,7 +444,7 @@ static INLINE int get_entropy_context(TX_SIZE tx_size, const ENTROPY_CONTEXT *a,
|
|||
#define COEF_MAX_UPDATE_FACTOR_AFTER_KEY 128
|
||||
|
||||
#if CONFIG_ADAPT_SCAN
|
||||
#define ADAPT_SCAN_PROB_PRECISION 16
|
||||
#define ADAPT_SCAN_PROB_PRECISION 10
|
||||
// 1/8 update rate
|
||||
#define ADAPT_SCAN_UPDATE_LOG_RATE 3
|
||||
#define ADAPT_SCAN_UPDATE_RATE \
|
||||
|
|
|
|||
6188
third_party/aom/av1/common/entropymode.c
vendored
6188
third_party/aom/av1/common/entropymode.c
vendored
File diff suppressed because it is too large
Load diff
295
third_party/aom/av1/common/entropymode.h
vendored
295
third_party/aom/av1/common/entropymode.h
vendored
|
|
@ -33,14 +33,11 @@ extern "C" {
|
|||
#define TX_SIZE_CONTEXTS 2
|
||||
|
||||
#define INTER_OFFSET(mode) ((mode)-NEARESTMV)
|
||||
#if CONFIG_EXT_INTER
|
||||
#if CONFIG_COMPOUND_SINGLEREF
|
||||
#define INTER_SINGLEREF_COMP_OFFSET(mode) ((mode)-SR_NEAREST_NEARMV)
|
||||
#endif // CONFIG_COMPOUND_SINGLEREF
|
||||
#define INTER_COMPOUND_OFFSET(mode) ((mode)-NEAREST_NEARESTMV)
|
||||
#endif // CONFIG_EXT_INTER
|
||||
|
||||
#if CONFIG_PALETTE
|
||||
// Number of possible contexts for a color index.
|
||||
// As can be seen from av1_get_palette_color_index_context(), the possible
|
||||
// contexts are (2,0,0), (2,2,1), (3,2,0), (4,1,0), (5,0,0). These are mapped to
|
||||
|
|
@ -70,11 +67,10 @@ extern "C" {
|
|||
#define PALETTE_UV_MODE_CONTEXTS 2
|
||||
|
||||
#define PALETTE_MAX_BLOCK_SIZE (64 * 64)
|
||||
#endif // CONFIG_PALETTE
|
||||
|
||||
#if CONFIG_INTRABC
|
||||
#define INTRABC_PROB_DEFAULT 192
|
||||
#endif // CONFIG_INTRABC
|
||||
#if CONFIG_KF_CTX
|
||||
#define KF_MODE_CONTEXTS 5
|
||||
#endif
|
||||
|
||||
struct AV1Common;
|
||||
|
||||
|
|
@ -98,12 +94,8 @@ typedef struct frame_contexts {
|
|||
#else
|
||||
aom_prob partition_prob[PARTITION_CONTEXTS][PARTITION_TYPES - 1];
|
||||
#endif
|
||||
av1_coeff_probs_model coef_probs[TX_SIZES][PLANE_TYPES];
|
||||
coeff_cdf_model coef_tail_cdfs[TX_SIZES][PLANE_TYPES];
|
||||
coeff_cdf_model coef_head_cdfs[TX_SIZES][PLANE_TYPES];
|
||||
aom_prob blockzero_probs[TX_SIZES][PLANE_TYPES][REF_TYPES][BLOCKZ_CONTEXTS];
|
||||
aom_prob switchable_interp_prob[SWITCHABLE_FILTER_CONTEXTS]
|
||||
[SWITCHABLE_FILTERS - 1];
|
||||
#if CONFIG_ADAPT_SCAN
|
||||
// TODO(angiebird): try aom_prob
|
||||
#if CONFIG_CHROMA_2X2
|
||||
|
|
@ -179,6 +171,38 @@ typedef struct frame_contexts {
|
|||
aom_prob coeff_base[TX_SIZES][PLANE_TYPES][NUM_BASE_LEVELS]
|
||||
[COEFF_BASE_CONTEXTS];
|
||||
aom_prob coeff_lps[TX_SIZES][PLANE_TYPES][LEVEL_CONTEXTS];
|
||||
#if BR_NODE
|
||||
aom_prob coeff_br[TX_SIZES][PLANE_TYPES][BASE_RANGE_SETS][LEVEL_CONTEXTS];
|
||||
#endif
|
||||
#if CONFIG_CTX1D
|
||||
aom_prob eob_mode[TX_SIZES][PLANE_TYPES][TX_CLASSES];
|
||||
aom_prob empty_line[TX_SIZES][PLANE_TYPES][TX_CLASSES][EMPTY_LINE_CONTEXTS];
|
||||
aom_prob hv_eob[TX_SIZES][PLANE_TYPES][TX_CLASSES][HV_EOB_CONTEXTS];
|
||||
#endif // CONFIG_CTX1D
|
||||
|
||||
#if LV_MAP_PROB
|
||||
aom_cdf_prob txb_skip_cdf[TX_SIZES][TXB_SKIP_CONTEXTS][CDF_SIZE(2)];
|
||||
aom_cdf_prob nz_map_cdf[TX_SIZES][PLANE_TYPES][SIG_COEF_CONTEXTS]
|
||||
[CDF_SIZE(2)];
|
||||
aom_cdf_prob eob_flag_cdf[TX_SIZES][PLANE_TYPES][EOB_COEF_CONTEXTS]
|
||||
[CDF_SIZE(2)];
|
||||
aom_cdf_prob dc_sign_cdf[PLANE_TYPES][DC_SIGN_CONTEXTS][CDF_SIZE(2)];
|
||||
aom_cdf_prob coeff_base_cdf[TX_SIZES][PLANE_TYPES][NUM_BASE_LEVELS]
|
||||
[COEFF_BASE_CONTEXTS][CDF_SIZE(2)];
|
||||
aom_cdf_prob coeff_lps_cdf[TX_SIZES][PLANE_TYPES][LEVEL_CONTEXTS]
|
||||
[CDF_SIZE(2)];
|
||||
#if BR_NODE
|
||||
aom_cdf_prob coeff_br_cdf[TX_SIZES][PLANE_TYPES][BASE_RANGE_SETS]
|
||||
[LEVEL_CONTEXTS][CDF_SIZE(2)];
|
||||
#endif
|
||||
#if CONFIG_CTX1D
|
||||
aom_cdf_prob eob_mode_cdf[TX_SIZES][PLANE_TYPES][TX_CLASSES][CDF_SIZE(2)];
|
||||
aom_cdf_prob empty_line_cdf[TX_SIZES][PLANE_TYPES][TX_CLASSES]
|
||||
[EMPTY_LINE_CONTEXTS][CDF_SIZE(2)];
|
||||
aom_cdf_prob hv_eob_cdf[TX_SIZES][PLANE_TYPES][TX_CLASSES][HV_EOB_CONTEXTS]
|
||||
[CDF_SIZE(2)];
|
||||
#endif // CONFIG_CTX1D
|
||||
#endif // LV_MAP_PROB
|
||||
#endif
|
||||
|
||||
aom_prob newmv_prob[NEWMV_MODE_CONTEXTS];
|
||||
|
|
@ -192,7 +216,6 @@ typedef struct frame_contexts {
|
|||
aom_cdf_prob drl_cdf[DRL_MODE_CONTEXTS][CDF_SIZE(2)];
|
||||
#endif
|
||||
|
||||
#if CONFIG_EXT_INTER
|
||||
aom_prob inter_compound_mode_probs[INTER_MODE_CONTEXTS]
|
||||
[INTER_COMPOUND_MODES - 1];
|
||||
aom_cdf_prob inter_compound_mode_cdf[INTER_MODE_CONTEXTS]
|
||||
|
|
@ -204,7 +227,9 @@ typedef struct frame_contexts {
|
|||
INTER_SINGLEREF_COMP_MODES)];
|
||||
#endif // CONFIG_COMPOUND_SINGLEREF
|
||||
aom_prob compound_type_prob[BLOCK_SIZES_ALL][COMPOUND_TYPES - 1];
|
||||
#if CONFIG_WEDGE || CONFIG_COMPOUND_SEGMENT
|
||||
aom_cdf_prob compound_type_cdf[BLOCK_SIZES_ALL][CDF_SIZE(COMPOUND_TYPES)];
|
||||
#endif // CONFIG_WEDGE || CONFIG_COMPOUND_SEGMENT
|
||||
#if CONFIG_INTERINTRA
|
||||
aom_prob interintra_prob[BLOCK_SIZE_GROUPS];
|
||||
aom_prob wedge_interintra_prob[BLOCK_SIZES_ALL];
|
||||
|
|
@ -216,7 +241,6 @@ typedef struct frame_contexts {
|
|||
aom_cdf_prob interintra_mode_cdf[BLOCK_SIZE_GROUPS]
|
||||
[CDF_SIZE(INTERINTRA_MODES)];
|
||||
#endif // CONFIG_INTERINTRA
|
||||
#endif // CONFIG_EXT_INTER
|
||||
#if CONFIG_MOTION_VAR || CONFIG_WARPED_MOTION
|
||||
aom_prob motion_mode_prob[BLOCK_SIZES_ALL][MOTION_MODES - 1];
|
||||
aom_cdf_prob motion_mode_cdf[BLOCK_SIZES_ALL][CDF_SIZE(MOTION_MODES)];
|
||||
|
|
@ -226,15 +250,18 @@ typedef struct frame_contexts {
|
|||
[CDF_SIZE(MAX_NCOBMC_MODES)];
|
||||
#endif
|
||||
#if CONFIG_MOTION_VAR && CONFIG_WARPED_MOTION
|
||||
#if CONFIG_NCOBMC_ADAPT_WEIGHT
|
||||
aom_prob ncobmc_prob[BLOCK_SIZES_ALL][OBMC_FAMILY_MODES - 1];
|
||||
aom_cdf_prob ncobmc_cdf[BLOCK_SIZES_ALL][CDF_SIZE(OBMC_FAMILY_MODES)];
|
||||
#endif // CONFIG_NCOBMC_ADAPT_WEIGHT
|
||||
aom_prob obmc_prob[BLOCK_SIZES_ALL];
|
||||
#if CONFIG_NEW_MULTISYMBOL
|
||||
#if CONFIG_NEW_MULTISYMBOL || CONFIG_NCOBMC_ADAPT_WEIGHT
|
||||
aom_cdf_prob obmc_cdf[BLOCK_SIZES_ALL][CDF_SIZE(2)];
|
||||
#endif // CONFIG_NEW_MULTISYMBOL
|
||||
#endif // CONFIG_MOTION_VAR && CONFIG_WARPED_MOTION
|
||||
#endif // CONFIG_MOTION_VAR || CONFIG_WARPED_MOTION
|
||||
aom_prob intra_inter_prob[INTRA_INTER_CONTEXTS];
|
||||
aom_prob comp_inter_prob[COMP_INTER_CONTEXTS];
|
||||
#if CONFIG_PALETTE
|
||||
aom_cdf_prob palette_y_size_cdf[PALETTE_BLOCK_SIZES][CDF_SIZE(PALETTE_SIZES)];
|
||||
aom_cdf_prob palette_uv_size_cdf[PALETTE_BLOCK_SIZES]
|
||||
[CDF_SIZE(PALETTE_SIZES)];
|
||||
|
|
@ -244,8 +271,16 @@ typedef struct frame_contexts {
|
|||
aom_cdf_prob palette_uv_color_index_cdf[PALETTE_SIZES]
|
||||
[PALETTE_COLOR_INDEX_CONTEXTS]
|
||||
[CDF_SIZE(PALETTE_COLORS)];
|
||||
#endif // CONFIG_PALETTE
|
||||
#if CONFIG_MRC_TX
|
||||
aom_cdf_prob mrc_mask_inter_cdf[PALETTE_SIZES][PALETTE_COLOR_INDEX_CONTEXTS]
|
||||
[CDF_SIZE(PALETTE_COLORS)];
|
||||
aom_cdf_prob mrc_mask_intra_cdf[PALETTE_SIZES][PALETTE_COLOR_INDEX_CONTEXTS]
|
||||
[CDF_SIZE(PALETTE_COLORS)];
|
||||
#endif // CONFIG_MRC_TX
|
||||
#if CONFIG_NEW_MULTISYMBOL
|
||||
aom_cdf_prob palette_y_mode_cdf[PALETTE_BLOCK_SIZES][PALETTE_Y_MODE_CONTEXTS]
|
||||
[CDF_SIZE(2)];
|
||||
aom_cdf_prob palette_uv_mode_cdf[PALETTE_UV_MODE_CONTEXTS][CDF_SIZE(2)];
|
||||
aom_cdf_prob comp_inter_cdf[COMP_INTER_CONTEXTS][CDF_SIZE(2)];
|
||||
aom_cdf_prob single_ref_cdf[REF_CONTEXTS][SINGLE_REFS - 1][CDF_SIZE(2)];
|
||||
#endif
|
||||
|
|
@ -273,12 +308,14 @@ typedef struct frame_contexts {
|
|||
aom_cdf_prob comp_ref_cdf[REF_CONTEXTS][COMP_REFS - 1][CDF_SIZE(2)];
|
||||
#endif // CONFIG_EXT_REFS
|
||||
#endif
|
||||
#if CONFIG_EXT_INTER && CONFIG_COMPOUND_SINGLEREF
|
||||
#if CONFIG_COMPOUND_SINGLEREF
|
||||
aom_prob comp_inter_mode_prob[COMP_INTER_MODE_CONTEXTS];
|
||||
#endif // CONFIG_EXT_INTER && CONFIG_COMPOUND_SINGLEREF
|
||||
aom_prob tx_size_probs[MAX_TX_DEPTH][TX_SIZE_CONTEXTS][MAX_TX_DEPTH];
|
||||
#endif // CONFIG_COMPOUND_SINGLEREF
|
||||
#if CONFIG_RECT_TX_EXT && (CONFIG_EXT_TX || CONFIG_VAR_TX)
|
||||
aom_prob quarter_tx_size_prob;
|
||||
#if CONFIG_NEW_MULTISYMBOL
|
||||
aom_cdf_prob quarter_tx_size_cdf[CDF_SIZE(2)];
|
||||
#endif
|
||||
#endif
|
||||
#if CONFIG_VAR_TX
|
||||
aom_prob txfm_partition_prob[TXFM_PARTITION_CONTEXTS];
|
||||
|
|
@ -294,17 +331,9 @@ typedef struct frame_contexts {
|
|||
nmv_context nmvc[NMV_CONTEXTS];
|
||||
#if CONFIG_INTRABC
|
||||
nmv_context ndvc;
|
||||
aom_prob intrabc_prob;
|
||||
aom_cdf_prob intrabc_cdf[CDF_SIZE(2)];
|
||||
#endif
|
||||
int initialized;
|
||||
#if CONFIG_EXT_TX
|
||||
aom_prob inter_ext_tx_prob[EXT_TX_SETS_INTER][EXT_TX_SIZES][TX_TYPES - 1];
|
||||
aom_prob intra_ext_tx_prob[EXT_TX_SETS_INTRA][EXT_TX_SIZES][INTRA_MODES]
|
||||
[TX_TYPES - 1];
|
||||
#else
|
||||
aom_prob intra_ext_tx_prob[EXT_TX_SIZES][TX_TYPES][TX_TYPES - 1];
|
||||
aom_prob inter_ext_tx_prob[EXT_TX_SIZES][TX_TYPES - 1];
|
||||
#endif // CONFIG_EXT_TX
|
||||
#if CONFIG_SUPERTX
|
||||
aom_prob supertx_prob[PARTITION_SUPERTX_CONTEXTS][TX_SIZES];
|
||||
#endif // CONFIG_SUPERTX
|
||||
|
|
@ -329,19 +358,25 @@ typedef struct frame_contexts {
|
|||
#endif
|
||||
aom_cdf_prob switchable_interp_cdf[SWITCHABLE_FILTER_CONTEXTS]
|
||||
[CDF_SIZE(SWITCHABLE_FILTERS)];
|
||||
/* kf_y_cdf is discarded after use, so does not require persistent storage.
|
||||
However, we keep it with the other CDFs in this struct since it needs to
|
||||
be copied to each tile to support parallelism just like the others.
|
||||
*/
|
||||
/* kf_y_cdf is discarded after use, so does not require persistent storage.
|
||||
However, we keep it with the other CDFs in this struct since it needs to
|
||||
be copied to each tile to support parallelism just like the others.
|
||||
*/
|
||||
#if CONFIG_KF_CTX
|
||||
aom_cdf_prob kf_y_cdf[KF_MODE_CONTEXTS][KF_MODE_CONTEXTS]
|
||||
[CDF_SIZE(INTRA_MODES)];
|
||||
#else
|
||||
aom_cdf_prob kf_y_cdf[INTRA_MODES][INTRA_MODES][CDF_SIZE(INTRA_MODES)];
|
||||
#endif
|
||||
aom_cdf_prob tx_size_cdf[MAX_TX_DEPTH][TX_SIZE_CONTEXTS]
|
||||
[CDF_SIZE(MAX_TX_DEPTH + 1)];
|
||||
#if CONFIG_DELTA_Q
|
||||
aom_cdf_prob delta_q_cdf[CDF_SIZE(DELTA_Q_PROBS + 1)];
|
||||
#if CONFIG_EXT_DELTA_Q
|
||||
#if CONFIG_LOOPFILTER_LEVEL
|
||||
aom_cdf_prob delta_lf_multi_cdf[FRAME_LF_COUNT][CDF_SIZE(DELTA_LF_PROBS + 1)];
|
||||
#endif // CONFIG_LOOPFILTER_LEVEL
|
||||
aom_cdf_prob delta_lf_cdf[CDF_SIZE(DELTA_LF_PROBS + 1)];
|
||||
#endif
|
||||
#endif // CONFIG_DELTA_Q
|
||||
#if CONFIG_EXT_TX
|
||||
aom_cdf_prob intra_ext_tx_cdf[EXT_TX_SETS_INTRA][EXT_TX_SIZES][INTRA_MODES]
|
||||
[CDF_SIZE(TX_TYPES)];
|
||||
|
|
@ -351,23 +386,34 @@ typedef struct frame_contexts {
|
|||
aom_cdf_prob intra_ext_tx_cdf[EXT_TX_SIZES][TX_TYPES][CDF_SIZE(TX_TYPES)];
|
||||
aom_cdf_prob inter_ext_tx_cdf[EXT_TX_SIZES][CDF_SIZE(TX_TYPES)];
|
||||
#endif // CONFIG_EXT_TX
|
||||
#if CONFIG_LGT_FROM_PRED
|
||||
aom_prob intra_lgt_prob[LGT_SIZES][INTRA_MODES];
|
||||
aom_prob inter_lgt_prob[LGT_SIZES];
|
||||
#endif // CONFIG_LGT_FROM_PRED
|
||||
#if CONFIG_EXT_INTRA && CONFIG_INTRA_INTERP
|
||||
aom_cdf_prob intra_filter_cdf[INTRA_FILTERS + 1][CDF_SIZE(INTRA_FILTERS)];
|
||||
#endif // CONFIG_EXT_INTRA && CONFIG_INTRA_INTERP
|
||||
#if CONFIG_DELTA_Q
|
||||
aom_prob delta_q_prob[DELTA_Q_PROBS];
|
||||
#if CONFIG_EXT_DELTA_Q
|
||||
#if CONFIG_LOOPFILTER_LEVEL
|
||||
aom_prob delta_lf_multi_prob[FRAME_LF_COUNT][DELTA_LF_PROBS];
|
||||
#endif // CONFIG_LOOPFILTER_LEVEL
|
||||
aom_prob delta_lf_prob[DELTA_LF_PROBS];
|
||||
#endif
|
||||
#endif
|
||||
#if CONFIG_PVQ
|
||||
// TODO(any): If PVQ is enabled, most of coefficient related cdf,
|
||||
// such as coef_cdfs[], coef_tail_cdfs[], and coef_heaf_cdfs[] can be removed.
|
||||
od_adapt_ctx pvq_context;
|
||||
#endif // CONFIG_PVQ
|
||||
#if CONFIG_CFL
|
||||
aom_cdf_prob cfl_alpha_cdf[CDF_SIZE(CFL_ALPHABET_SIZE)];
|
||||
aom_cdf_prob cfl_sign_cdf[CDF_SIZE(CFL_JOINT_SIGNS)];
|
||||
aom_cdf_prob cfl_alpha_cdf[CFL_ALPHA_CONTEXTS][CDF_SIZE(CFL_ALPHABET_SIZE)];
|
||||
#endif
|
||||
#if CONFIG_LPF_SB
|
||||
aom_cdf_prob lpf_reuse_cdf[LPF_REUSE_CONTEXT][CDF_SIZE(2)];
|
||||
aom_cdf_prob lpf_delta_cdf[LPF_DELTA_CONTEXT][CDF_SIZE(DELTA_RANGE)];
|
||||
aom_cdf_prob lpf_sign_cdf[LPF_REUSE_CONTEXT][LPF_SIGN_CONTEXT][CDF_SIZE(2)];
|
||||
#endif // CONFIG_LPF_SB
|
||||
} FRAME_CONTEXT;
|
||||
|
||||
typedef struct FRAME_COUNTS {
|
||||
|
|
@ -383,9 +429,6 @@ typedef struct FRAME_COUNTS {
|
|||
#else
|
||||
unsigned int partition[PARTITION_CONTEXTS][PARTITION_TYPES];
|
||||
#endif
|
||||
av1_coeff_count_model coef[TX_SIZES][PLANE_TYPES];
|
||||
unsigned int eob_branch[TX_SIZES][PLANE_TYPES][REF_TYPES][COEF_BANDS]
|
||||
[COEFF_CONTEXTS];
|
||||
unsigned int switchable_interp[SWITCHABLE_FILTER_CONTEXTS]
|
||||
[SWITCHABLE_FILTERS];
|
||||
#if CONFIG_ADAPT_SCAN
|
||||
|
|
@ -415,16 +458,26 @@ typedef struct FRAME_COUNTS {
|
|||
unsigned int coeff_base[TX_SIZES][PLANE_TYPES][NUM_BASE_LEVELS]
|
||||
[COEFF_BASE_CONTEXTS][2];
|
||||
unsigned int coeff_lps[TX_SIZES][PLANE_TYPES][LEVEL_CONTEXTS][2];
|
||||
unsigned int coeff_br[TX_SIZES][PLANE_TYPES][BASE_RANGE_SETS][LEVEL_CONTEXTS]
|
||||
[2];
|
||||
#if CONFIG_CTX1D
|
||||
unsigned int eob_mode[TX_SIZES][PLANE_TYPES][TX_CLASSES][2];
|
||||
unsigned int empty_line[TX_SIZES][PLANE_TYPES][TX_CLASSES]
|
||||
[EMPTY_LINE_CONTEXTS][2];
|
||||
unsigned int hv_eob[TX_SIZES][PLANE_TYPES][TX_CLASSES][HV_EOB_CONTEXTS][2];
|
||||
#endif // CONFIG_CTX1D
|
||||
#endif // CONFIG_LV_MAP
|
||||
|
||||
av1_blockz_count_model blockz_count[TX_SIZES][PLANE_TYPES];
|
||||
#if CONFIG_SYMBOLRATE
|
||||
unsigned int coeff_num[2]; // 0: zero coeff 1: non-zero coeff
|
||||
unsigned int symbol_num[2]; // 0: entropy symbol 1: non-entropy symbol
|
||||
#endif
|
||||
|
||||
unsigned int newmv_mode[NEWMV_MODE_CONTEXTS][2];
|
||||
unsigned int zeromv_mode[ZEROMV_MODE_CONTEXTS][2];
|
||||
unsigned int refmv_mode[REFMV_MODE_CONTEXTS][2];
|
||||
unsigned int drl_mode[DRL_MODE_CONTEXTS][2];
|
||||
|
||||
#if CONFIG_EXT_INTER
|
||||
unsigned int inter_compound_mode[INTER_MODE_CONTEXTS][INTER_COMPOUND_MODES];
|
||||
#if CONFIG_COMPOUND_SINGLEREF
|
||||
unsigned int inter_singleref_comp_mode[INTER_MODE_CONTEXTS]
|
||||
|
|
@ -436,13 +489,15 @@ typedef struct FRAME_COUNTS {
|
|||
unsigned int wedge_interintra[BLOCK_SIZES_ALL][2];
|
||||
#endif // CONFIG_INTERINTRA
|
||||
unsigned int compound_interinter[BLOCK_SIZES_ALL][COMPOUND_TYPES];
|
||||
#endif // CONFIG_EXT_INTER
|
||||
#if CONFIG_MOTION_VAR || CONFIG_WARPED_MOTION
|
||||
unsigned int motion_mode[BLOCK_SIZES_ALL][MOTION_MODES];
|
||||
#if CONFIG_NCOBMC_ADAPT_WEIGHT && CONFIG_MOTION_VAR
|
||||
unsigned int ncobmc_mode[ADAPT_OVERLAP_BLOCKS][MAX_NCOBMC_MODES];
|
||||
#endif
|
||||
#if CONFIG_MOTION_VAR && CONFIG_WARPED_MOTION
|
||||
#if CONFIG_NCOBMC_ADAPT_WEIGHT
|
||||
unsigned int ncobmc[BLOCK_SIZES_ALL][OBMC_FAMILY_MODES];
|
||||
#endif // CONFIG_NCOBMC_ADAPT_WEIGHT
|
||||
unsigned int obmc[BLOCK_SIZES_ALL][2];
|
||||
#endif // CONFIG_MOTION_VAR && CONFIG_WARPED_MOTION
|
||||
#endif // CONFIG_MOTION_VAR || CONFIG_WARPED_MOTION
|
||||
|
|
@ -459,13 +514,11 @@ typedef struct FRAME_COUNTS {
|
|||
#else
|
||||
unsigned int comp_ref[REF_CONTEXTS][COMP_REFS - 1][2];
|
||||
#endif // CONFIG_EXT_REFS
|
||||
#if CONFIG_EXT_INTER && CONFIG_COMPOUND_SINGLEREF
|
||||
#if CONFIG_COMPOUND_SINGLEREF
|
||||
unsigned int comp_inter_mode[COMP_INTER_MODE_CONTEXTS][2];
|
||||
#endif // CONFIG_EXT_INTER && CONFIG_COMPOUND_SINGLEREF
|
||||
// TODO(any): tx_size_totals is only used by the encoder to decide whether
|
||||
// to use forward updates for the coeff probs, and as such it does not really
|
||||
// belong into this structure.
|
||||
unsigned int tx_size_totals[TX_SIZES];
|
||||
#endif // CONFIG_COMPOUND_SINGLEREF
|
||||
// TODO(urvang): Only needed for !CONFIG_VAR_TX case. So can be removed when
|
||||
// CONFIG_VAR_TX flag is removed.
|
||||
unsigned int tx_size[MAX_TX_DEPTH][TX_SIZE_CONTEXTS][MAX_TX_DEPTH + 1];
|
||||
#if CONFIG_RECT_TX_EXT && (CONFIG_EXT_TX || CONFIG_VAR_TX)
|
||||
unsigned int quarter_tx_size[2];
|
||||
|
|
@ -479,16 +532,22 @@ typedef struct FRAME_COUNTS {
|
|||
unsigned int intrabc[2];
|
||||
nmv_context_counts dv;
|
||||
#endif
|
||||
#if CONFIG_DELTA_Q
|
||||
#if CONFIG_LGT_FROM_PRED
|
||||
unsigned int intra_lgt[LGT_SIZES][INTRA_MODES][2];
|
||||
unsigned int inter_lgt[LGT_SIZES][2];
|
||||
#endif // CONFIG_LGT_FROM_PRED
|
||||
unsigned int delta_q[DELTA_Q_PROBS][2];
|
||||
#if CONFIG_EXT_DELTA_Q
|
||||
#if CONFIG_LOOPFILTER_LEVEL
|
||||
unsigned int delta_lf_multi[FRAME_LF_COUNT][DELTA_LF_PROBS][2];
|
||||
#endif // CONFIG_LOOPFILTER_LEVEL
|
||||
unsigned int delta_lf[DELTA_LF_PROBS][2];
|
||||
#endif
|
||||
#endif
|
||||
#if CONFIG_EXT_TX
|
||||
#if CONFIG_RECT_TX
|
||||
#if CONFIG_EXT_TX && CONFIG_RECT_TX
|
||||
unsigned int tx_size_implied[TX_SIZES][TX_SIZES];
|
||||
#endif // CONFIG_RECT_TX
|
||||
#endif // CONFIG_EXT_TX && CONFIG_RECT_TX
|
||||
#if CONFIG_ENTROPY_STATS
|
||||
#if CONFIG_EXT_TX
|
||||
unsigned int inter_ext_tx[EXT_TX_SETS_INTER][EXT_TX_SIZES][TX_TYPES];
|
||||
unsigned int intra_ext_tx[EXT_TX_SETS_INTRA][EXT_TX_SIZES][INTRA_MODES]
|
||||
[TX_TYPES];
|
||||
|
|
@ -496,6 +555,7 @@ typedef struct FRAME_COUNTS {
|
|||
unsigned int intra_ext_tx[EXT_TX_SIZES][TX_TYPES][TX_TYPES];
|
||||
unsigned int inter_ext_tx[EXT_TX_SIZES][TX_TYPES];
|
||||
#endif // CONFIG_EXT_TX
|
||||
#endif // CONFIG_ENTROPY_STATS
|
||||
#if CONFIG_SUPERTX
|
||||
unsigned int supertx[PARTITION_SUPERTX_CONTEXTS][TX_SIZES][2];
|
||||
unsigned int supertx_size[TX_SIZES];
|
||||
|
|
@ -509,29 +569,103 @@ typedef struct FRAME_COUNTS {
|
|||
#if CONFIG_FILTER_INTRA
|
||||
unsigned int filter_intra[PLANE_TYPES][2];
|
||||
#endif // CONFIG_FILTER_INTRA
|
||||
#if CONFIG_LPF_SB
|
||||
unsigned int lpf_reuse[LPF_REUSE_CONTEXT][2];
|
||||
unsigned int lpf_delta[LPF_DELTA_CONTEXT][DELTA_RANGE];
|
||||
unsigned int lpf_sign[LPF_SIGN_CONTEXT][2];
|
||||
#endif // CONFIG_LPF_SB
|
||||
} FRAME_COUNTS;
|
||||
|
||||
// CDF version of 'av1_kf_y_mode_prob'.
|
||||
extern const aom_cdf_prob av1_kf_y_mode_cdf[INTRA_MODES][INTRA_MODES]
|
||||
[CDF_SIZE(INTRA_MODES)];
|
||||
#if CONFIG_KF_CTX
|
||||
extern const aom_cdf_prob default_kf_y_mode_cdf[KF_MODE_CONTEXTS]
|
||||
[KF_MODE_CONTEXTS]
|
||||
[CDF_SIZE(INTRA_MODES)];
|
||||
#else
|
||||
extern const aom_cdf_prob default_kf_y_mode_cdf[INTRA_MODES][INTRA_MODES]
|
||||
[CDF_SIZE(INTRA_MODES)];
|
||||
#endif
|
||||
|
||||
#if CONFIG_PALETTE
|
||||
extern const aom_prob av1_default_palette_y_mode_prob[PALETTE_BLOCK_SIZES]
|
||||
[PALETTE_Y_MODE_CONTEXTS];
|
||||
extern const aom_prob
|
||||
av1_default_palette_uv_mode_prob[PALETTE_UV_MODE_CONTEXTS];
|
||||
#endif // CONFIG_PALETTE
|
||||
|
||||
extern const int av1_intra_mode_ind[INTRA_MODES];
|
||||
extern const int av1_intra_mode_inv[INTRA_MODES];
|
||||
#if CONFIG_EXT_TX
|
||||
extern int av1_ext_tx_intra_ind[EXT_TX_SETS_INTRA][TX_TYPES];
|
||||
extern int av1_ext_tx_intra_inv[EXT_TX_SETS_INTRA][TX_TYPES];
|
||||
extern int av1_ext_tx_inter_ind[EXT_TX_SETS_INTER][TX_TYPES];
|
||||
extern int av1_ext_tx_inter_inv[EXT_TX_SETS_INTER][TX_TYPES];
|
||||
#endif
|
||||
static const int av1_ext_tx_ind[EXT_TX_SET_TYPES][TX_TYPES] = {
|
||||
{
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
},
|
||||
{
|
||||
1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
},
|
||||
#if CONFIG_MRC_TX
|
||||
{
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 1,
|
||||
},
|
||||
{
|
||||
1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 2,
|
||||
},
|
||||
#endif // CONFIG_MRC_TX
|
||||
{
|
||||
1, 3, 4, 2, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
},
|
||||
{
|
||||
1, 5, 6, 4, 0, 0, 0, 0, 0, 0, 2, 3, 0, 0, 0, 0,
|
||||
},
|
||||
{
|
||||
3, 4, 5, 8, 6, 7, 9, 10, 11, 0, 1, 2, 0, 0, 0, 0,
|
||||
},
|
||||
{
|
||||
7, 8, 9, 12, 10, 11, 13, 14, 15, 0, 1, 2, 3, 4, 5, 6,
|
||||
},
|
||||
};
|
||||
|
||||
static const int av1_ext_tx_inv[EXT_TX_SET_TYPES][TX_TYPES] = {
|
||||
{
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
},
|
||||
{
|
||||
9, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
},
|
||||
#if CONFIG_MRC_TX
|
||||
{
|
||||
0, 16, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
},
|
||||
{
|
||||
9, 0, 16, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
},
|
||||
#endif // CONFIG_MRC_TX
|
||||
{
|
||||
9, 0, 3, 1, 2, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
},
|
||||
{
|
||||
9, 0, 10, 11, 3, 1, 2, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
},
|
||||
{
|
||||
9, 10, 11, 0, 1, 2, 4, 5, 3, 6, 7, 8, 0, 0, 0, 0,
|
||||
},
|
||||
{
|
||||
9, 10, 11, 12, 13, 14, 15, 0, 1, 2, 4, 5, 3, 6, 7, 8,
|
||||
},
|
||||
};
|
||||
#else
|
||||
#if CONFIG_MRC_TX
|
||||
static const int av1_ext_tx_ind[TX_TYPES] = {
|
||||
0, 3, 4, 2, 1,
|
||||
};
|
||||
static const int av1_ext_tx_inv[TX_TYPES] = {
|
||||
0, 4, 3, 1, 2,
|
||||
};
|
||||
#else
|
||||
static const int av1_ext_tx_ind[TX_TYPES] = {
|
||||
0, 2, 3, 1,
|
||||
};
|
||||
static const int av1_ext_tx_inv[TX_TYPES] = {
|
||||
0, 3, 1, 2,
|
||||
};
|
||||
#endif // CONFIG_MRC_TX
|
||||
#endif // CONFIG_EXT_TX
|
||||
|
||||
#if CONFIG_EXT_INTER
|
||||
#if CONFIG_INTERINTRA
|
||||
extern const aom_tree_index
|
||||
av1_interintra_mode_tree[TREE_SIZE(INTERINTRA_MODES)];
|
||||
|
|
@ -543,36 +677,31 @@ extern const aom_tree_index
|
|||
av1_inter_singleref_comp_mode_tree[TREE_SIZE(INTER_SINGLEREF_COMP_MODES)];
|
||||
#endif // CONFIG_COMPOUND_SINGLEREF
|
||||
extern const aom_tree_index av1_compound_type_tree[TREE_SIZE(COMPOUND_TYPES)];
|
||||
#endif // CONFIG_EXT_INTER
|
||||
extern const aom_tree_index av1_partition_tree[TREE_SIZE(PARTITION_TYPES)];
|
||||
#if CONFIG_EXT_PARTITION_TYPES
|
||||
extern const aom_tree_index
|
||||
av1_ext_partition_tree[TREE_SIZE(EXT_PARTITION_TYPES)];
|
||||
#endif
|
||||
extern const aom_tree_index
|
||||
av1_switchable_interp_tree[TREE_SIZE(SWITCHABLE_FILTERS)];
|
||||
#if CONFIG_PALETTE
|
||||
extern const aom_tree_index
|
||||
av1_palette_color_index_tree[PALETTE_SIZES][TREE_SIZE(PALETTE_COLORS)];
|
||||
#endif // CONFIG_PALETTE
|
||||
extern const aom_tree_index av1_tx_size_tree[MAX_TX_DEPTH][TREE_SIZE(TX_SIZES)];
|
||||
#if CONFIG_EXT_INTRA && CONFIG_INTRA_INTERP
|
||||
extern const aom_tree_index av1_intra_filter_tree[TREE_SIZE(INTRA_FILTERS)];
|
||||
#endif // CONFIG_EXT_INTRA && CONFIG_INTRA_INTERP
|
||||
#if CONFIG_EXT_TX
|
||||
extern const aom_tree_index av1_ext_tx_inter_tree[EXT_TX_SETS_INTER]
|
||||
[TREE_SIZE(TX_TYPES)];
|
||||
extern const aom_tree_index av1_ext_tx_intra_tree[EXT_TX_SETS_INTRA]
|
||||
[TREE_SIZE(TX_TYPES)];
|
||||
extern const aom_tree_index av1_ext_tx_tree[EXT_TX_SET_TYPES]
|
||||
[TREE_SIZE(TX_TYPES)];
|
||||
#else
|
||||
extern const aom_tree_index av1_ext_tx_tree[TREE_SIZE(TX_TYPES)];
|
||||
#endif // CONFIG_EXT_TX
|
||||
#if CONFIG_MOTION_VAR || CONFIG_WARPED_MOTION
|
||||
extern const aom_tree_index av1_motion_mode_tree[TREE_SIZE(MOTION_MODES)];
|
||||
#endif // CONFIG_MOTION_VAR || CONFIG_WARPED_MOTION
|
||||
#if CONFIG_NCOBMC_ADAPT_WEIGHT && CONFIG_MOTION_VAR
|
||||
#if CONFIG_NCOBMC_ADAPT_WEIGHT
|
||||
extern const aom_tree_index av1_ncobmc_mode_tree[TREE_SIZE(MAX_NCOBMC_MODES)];
|
||||
#endif
|
||||
#if CONFIG_WARPED_MOTION
|
||||
extern const aom_tree_index av1_ncobmc_tree[TREE_SIZE(OBMC_FAMILY_MODES)];
|
||||
#endif // CONFIG_WARPED_MOTION
|
||||
#endif // CONFIG_NCOBMC_ADAPT_WEIGHT
|
||||
#if CONFIG_LOOP_RESTORATION
|
||||
#define RESTORE_NONE_SGRPROJ_PROB 64
|
||||
#define RESTORE_NONE_BILATERAL_PROB 16
|
||||
|
|
@ -581,17 +710,11 @@ extern const aom_tree_index av1_ncobmc_mode_tree[TREE_SIZE(MAX_NCOBMC_MODES)];
|
|||
extern const aom_tree_index
|
||||
av1_switchable_restore_tree[TREE_SIZE(RESTORE_SWITCHABLE_TYPES)];
|
||||
#endif // CONFIG_LOOP_RESTORATION
|
||||
extern int av1_switchable_interp_ind[SWITCHABLE_FILTERS];
|
||||
extern int av1_switchable_interp_inv[SWITCHABLE_FILTERS];
|
||||
|
||||
void av1_setup_past_independence(struct AV1Common *cm);
|
||||
|
||||
void av1_adapt_intra_frame_probs(struct AV1Common *cm);
|
||||
void av1_adapt_inter_frame_probs(struct AV1Common *cm);
|
||||
#if !CONFIG_EXT_TX
|
||||
extern int av1_ext_tx_ind[TX_TYPES];
|
||||
extern int av1_ext_tx_inv[TX_TYPES];
|
||||
#endif
|
||||
|
||||
static INLINE int av1_ceil_log2(int n) {
|
||||
int i = 1, p = 2;
|
||||
|
|
@ -602,14 +725,12 @@ static INLINE int av1_ceil_log2(int n) {
|
|||
return i;
|
||||
}
|
||||
|
||||
#if CONFIG_PALETTE
|
||||
// Returns the context for palette color index at row 'r' and column 'c',
|
||||
// along with the 'color_order' of neighbors and the 'color_idx'.
|
||||
// The 'color_map' is a 2D array with the given 'stride'.
|
||||
int av1_get_palette_color_index_context(const uint8_t *color_map, int stride,
|
||||
int r, int c, int palette_size,
|
||||
uint8_t *color_order, int *color_idx);
|
||||
#endif // CONFIG_PALETTE
|
||||
|
||||
#ifdef __cplusplus
|
||||
} // extern "C"
|
||||
|
|
|
|||
39
third_party/aom/av1/common/entropymv.c
vendored
39
third_party/aom/av1/common/entropymv.c
vendored
|
|
@ -68,6 +68,12 @@ static const nmv_context default_nmv_context = {
|
|||
{ AOM_ICDF(160 * 128), AOM_ICDF(32768), 0 },
|
||||
{ AOM_ICDF(128 * 128), AOM_ICDF(32768), 0 },
|
||||
{ AOM_ICDF(216 * 128), AOM_ICDF(32768), 0 },
|
||||
{ { AOM_ICDF(128 * 196), AOM_ICDF(32768), 0 },
|
||||
{ AOM_ICDF(128 * 198), AOM_ICDF(32768), 0 },
|
||||
{ AOM_ICDF(128 * 208), AOM_ICDF(32768), 0 },
|
||||
{ AOM_ICDF(128 * 224), AOM_ICDF(32768), 0 },
|
||||
{ AOM_ICDF(128 * 245), AOM_ICDF(32768), 0 },
|
||||
{ AOM_ICDF(128 * 240), AOM_ICDF(32768), 0 } }, // bits_cdf
|
||||
#endif
|
||||
},
|
||||
{
|
||||
|
|
@ -93,6 +99,12 @@ static const nmv_context default_nmv_context = {
|
|||
{ AOM_ICDF(160 * 128), AOM_ICDF(32768), 0 },
|
||||
{ AOM_ICDF(128 * 128), AOM_ICDF(32768), 0 },
|
||||
{ AOM_ICDF(216 * 128), AOM_ICDF(32768), 0 },
|
||||
{ { AOM_ICDF(128 * 196), AOM_ICDF(32768), 0 },
|
||||
{ AOM_ICDF(128 * 198), AOM_ICDF(32768), 0 },
|
||||
{ AOM_ICDF(128 * 208), AOM_ICDF(32768), 0 },
|
||||
{ AOM_ICDF(128 * 224), AOM_ICDF(32768), 0 },
|
||||
{ AOM_ICDF(128 * 245), AOM_ICDF(32768), 0 },
|
||||
{ AOM_ICDF(128 * 240), AOM_ICDF(32768), 0 } }, // bits_cdf
|
||||
#endif
|
||||
} },
|
||||
};
|
||||
|
|
@ -169,7 +181,7 @@ static void inc_mv_component(int v, nmv_component_counts *comp_counts, int incr,
|
|||
|
||||
if (c == MV_CLASS_0) {
|
||||
comp_counts->class0[d] += incr;
|
||||
#if CONFIG_INTRABC
|
||||
#if CONFIG_INTRABC || CONFIG_AMVR
|
||||
if (precision > MV_SUBPEL_NONE)
|
||||
#endif
|
||||
comp_counts->class0_fp[d][f] += incr;
|
||||
|
|
@ -178,7 +190,7 @@ static void inc_mv_component(int v, nmv_component_counts *comp_counts, int incr,
|
|||
int i;
|
||||
int b = c + CLASS0_BITS - 1; // number of bits
|
||||
for (i = 0; i < b; ++i) comp_counts->bits[i][((d >> i) & 1)] += incr;
|
||||
#if CONFIG_INTRABC
|
||||
#if CONFIG_INTRABC || CONFIG_AMVR
|
||||
if (precision > MV_SUBPEL_NONE)
|
||||
#endif
|
||||
comp_counts->fp[f] += incr;
|
||||
|
|
@ -222,18 +234,23 @@ void av1_adapt_mv_probs(AV1_COMMON *cm, int allow_hp) {
|
|||
|
||||
for (j = 0; j < MV_OFFSET_BITS; ++j)
|
||||
comp->bits[j] = av1_mode_mv_merge_probs(pre_comp->bits[j], c->bits[j]);
|
||||
#if CONFIG_AMVR
|
||||
if (cm->cur_frame_mv_precision_level == 0) {
|
||||
#endif
|
||||
for (j = 0; j < CLASS0_SIZE; ++j)
|
||||
aom_tree_merge_probs(av1_mv_fp_tree, pre_comp->class0_fp[j],
|
||||
c->class0_fp[j], comp->class0_fp[j]);
|
||||
|
||||
for (j = 0; j < CLASS0_SIZE; ++j)
|
||||
aom_tree_merge_probs(av1_mv_fp_tree, pre_comp->class0_fp[j],
|
||||
c->class0_fp[j], comp->class0_fp[j]);
|
||||
aom_tree_merge_probs(av1_mv_fp_tree, pre_comp->fp, c->fp, comp->fp);
|
||||
|
||||
aom_tree_merge_probs(av1_mv_fp_tree, pre_comp->fp, c->fp, comp->fp);
|
||||
|
||||
if (allow_hp) {
|
||||
comp->class0_hp =
|
||||
av1_mode_mv_merge_probs(pre_comp->class0_hp, c->class0_hp);
|
||||
comp->hp = av1_mode_mv_merge_probs(pre_comp->hp, c->hp);
|
||||
if (allow_hp) {
|
||||
comp->class0_hp =
|
||||
av1_mode_mv_merge_probs(pre_comp->class0_hp, c->class0_hp);
|
||||
comp->hp = av1_mode_mv_merge_probs(pre_comp->hp, c->hp);
|
||||
}
|
||||
#if CONFIG_AMVR
|
||||
}
|
||||
#endif
|
||||
}
|
||||
}
|
||||
}
|
||||
|
|
|
|||
6
third_party/aom/av1/common/entropymv.h
vendored
6
third_party/aom/av1/common/entropymv.h
vendored
|
|
@ -66,6 +66,9 @@ typedef enum {
|
|||
#define CLASS0_BITS 1 /* bits at integer precision for class 0 */
|
||||
#define CLASS0_SIZE (1 << CLASS0_BITS)
|
||||
#define MV_OFFSET_BITS (MV_CLASSES + CLASS0_BITS - 2)
|
||||
#if CONFIG_NEW_MULTISYMBOL
|
||||
#define MV_BITS_CONTEXTS 6
|
||||
#endif
|
||||
#define MV_FP_SIZE 4
|
||||
|
||||
#define MV_MAX_BITS (MV_CLASSES + CLASS0_BITS + 2)
|
||||
|
|
@ -97,6 +100,7 @@ typedef struct {
|
|||
aom_cdf_prob class0_hp_cdf[CDF_SIZE(2)];
|
||||
aom_cdf_prob hp_cdf[CDF_SIZE(2)];
|
||||
aom_cdf_prob class0_cdf[CDF_SIZE(CLASS0_SIZE)];
|
||||
aom_cdf_prob bits_cdf[MV_BITS_CONTEXTS][CDF_SIZE(2)];
|
||||
#endif
|
||||
} nmv_component;
|
||||
|
||||
|
|
@ -133,7 +137,7 @@ typedef struct {
|
|||
} nmv_context_counts;
|
||||
|
||||
typedef enum {
|
||||
#if CONFIG_INTRABC
|
||||
#if CONFIG_INTRABC || CONFIG_AMVR
|
||||
MV_SUBPEL_NONE = -1,
|
||||
#endif
|
||||
MV_SUBPEL_LOW_PRECISION = 0,
|
||||
|
|
|
|||
356
third_party/aom/av1/common/enums.h
vendored
356
third_party/aom/av1/common/enums.h
vendored
|
|
@ -22,6 +22,16 @@ extern "C" {
|
|||
|
||||
#undef MAX_SB_SIZE
|
||||
|
||||
#if CONFIG_NCOBMC_ADAPT_WEIGHT
|
||||
#define TWO_MODE
|
||||
#endif
|
||||
|
||||
#if CONFIG_NCOBMC || CONFIG_NCOBMC_ADAPT_WEIGHT
|
||||
#define NC_MODE_INFO 1
|
||||
#else
|
||||
#define NC_MODE_INFO 0
|
||||
#endif
|
||||
|
||||
// Max superblock size
|
||||
#if CONFIG_EXT_PARTITION
|
||||
#define MAX_SB_SIZE_LOG2 7
|
||||
|
|
@ -57,16 +67,45 @@ extern "C" {
|
|||
#define MAX_TILE_ROWS 1024
|
||||
#define MAX_TILE_COLS 1024
|
||||
#else
|
||||
#if CONFIG_MAX_TILE
|
||||
#define MAX_TILE_ROWS 64
|
||||
#define MAX_TILE_COLS 64
|
||||
#else
|
||||
#define MAX_TILE_ROWS 4
|
||||
#define MAX_TILE_COLS 64
|
||||
#endif
|
||||
#endif // CONFIG_EXT_TILE
|
||||
|
||||
#if CONFIG_VAR_TX
|
||||
#define MAX_VARTX_DEPTH 2
|
||||
#define SQR_VARTX_DEPTH_INIT 0
|
||||
#define RECT_VARTX_DEPTH_INIT 0
|
||||
#endif
|
||||
|
||||
#define MI_SIZE_64X64 (64 >> MI_SIZE_LOG2)
|
||||
|
||||
#if CONFIG_LOOPFILTER_LEVEL
|
||||
// 4 frame filter levels: y plane vertical, y plane horizontal,
|
||||
// u plane, and v plane
|
||||
#define FRAME_LF_COUNT 4
|
||||
#define DEFAULT_DELTA_LF_MULTI 0
|
||||
#endif // CONFIG_LOOPFILTER_LEVEL
|
||||
|
||||
#if CONFIG_LPF_SB
|
||||
#define LPF_DELTA_BITS 3
|
||||
#define LPF_STEP 2
|
||||
#define DELTA_RANGE (1 << LPF_DELTA_BITS)
|
||||
#define MAX_LPF_OFFSET (LPF_STEP * ((1 << LPF_DELTA_BITS) - 1))
|
||||
|
||||
#define LPF_REUSE_CONTEXT 2
|
||||
#define LPF_DELTA_CONTEXT DELTA_RANGE
|
||||
#define LPF_SIGN_CONTEXT 2
|
||||
|
||||
// Half of maximum loop filter length (15-tap)
|
||||
#define FILT_BOUNDARY_OFFSET 8
|
||||
#define FILT_BOUNDARY_MI_OFFSET (FILT_BOUNDARY_OFFSET >> MI_SIZE_LOG2)
|
||||
#endif // CONFIG_LPF_SB
|
||||
|
||||
// Bitstream profiles indicated by 2-3 bits in the uncompressed header.
|
||||
// 00: Profile 0. 8-bit 4:2:0 only.
|
||||
// 10: Profile 1. 8-bit 4:4:4, 4:2:2, and 4:4:0.
|
||||
|
|
@ -113,6 +152,12 @@ typedef enum ATTRIBUTE_PACKED {
|
|||
BLOCK_16X4,
|
||||
BLOCK_8X32,
|
||||
BLOCK_32X8,
|
||||
BLOCK_16X64,
|
||||
BLOCK_64X16,
|
||||
#if CONFIG_EXT_PARTITION
|
||||
BLOCK_32X128,
|
||||
BLOCK_128X32,
|
||||
#endif // CONFIG_EXT_PARTITION
|
||||
BLOCK_SIZES_ALL,
|
||||
BLOCK_SIZES = BLOCK_4X16,
|
||||
BLOCK_INVALID = 255,
|
||||
|
|
@ -125,10 +170,10 @@ typedef enum {
|
|||
PARTITION_VERT,
|
||||
PARTITION_SPLIT,
|
||||
#if CONFIG_EXT_PARTITION_TYPES
|
||||
PARTITION_HORZ_A, // HORZ split and the left partition is split again
|
||||
PARTITION_HORZ_B, // HORZ split and the right partition is split again
|
||||
PARTITION_VERT_A, // VERT split and the top partition is split again
|
||||
PARTITION_VERT_B, // VERT split and the bottom partition is split again
|
||||
PARTITION_HORZ_A, // HORZ split and the top partition is split again
|
||||
PARTITION_HORZ_B, // HORZ split and the bottom partition is split again
|
||||
PARTITION_VERT_A, // VERT split and the left partition is split again
|
||||
PARTITION_VERT_B, // VERT split and the right partition is split again
|
||||
PARTITION_HORZ_4, // 4:1 horizontal partition
|
||||
PARTITION_VERT_4, // 4:1 vertical partition
|
||||
EXT_PARTITION_TYPES,
|
||||
|
|
@ -142,6 +187,7 @@ typedef char PARTITION_CONTEXT;
|
|||
#define PARTITION_BLOCK_SIZES (4 + CONFIG_EXT_PARTITION)
|
||||
#define PARTITION_CONTEXTS_PRIMARY (PARTITION_BLOCK_SIZES * PARTITION_PLOFFSET)
|
||||
#if CONFIG_UNPOISON_PARTITION_CTX
|
||||
#define INVALID_PARTITION_CTX (-1)
|
||||
#define PARTITION_CONTEXTS \
|
||||
(PARTITION_CONTEXTS_PRIMARY + 2 * PARTITION_BLOCK_SIZES)
|
||||
#else
|
||||
|
|
@ -158,14 +204,18 @@ typedef enum ATTRIBUTE_PACKED {
|
|||
TX_16X16, // 16x16 transform
|
||||
TX_32X32, // 32x32 transform
|
||||
#if CONFIG_TX64X64
|
||||
TX_64X64, // 64x64 transform
|
||||
TX_64X64, // 64x64 transform
|
||||
#endif // CONFIG_TX64X64
|
||||
TX_4X8, // 4x8 transform
|
||||
TX_8X4, // 8x4 transform
|
||||
TX_8X16, // 8x16 transform
|
||||
TX_16X8, // 16x8 transform
|
||||
TX_16X32, // 16x32 transform
|
||||
TX_32X16, // 32x16 transform
|
||||
#if CONFIG_TX64X64
|
||||
TX_32X64, // 32x64 transform
|
||||
TX_64X32, // 64x32 transform
|
||||
#endif // CONFIG_TX64X64
|
||||
TX_4X8, // 4x8 transform
|
||||
TX_8X4, // 8x4 transform
|
||||
TX_8X16, // 8x16 transform
|
||||
TX_16X8, // 16x8 transform
|
||||
TX_16X32, // 16x32 transform
|
||||
TX_32X16, // 32x16 transform
|
||||
TX_4X16, // 4x16 transform
|
||||
TX_16X4, // 16x4 transform
|
||||
TX_8X32, // 8x32 transform
|
||||
|
|
@ -182,6 +232,10 @@ typedef enum ATTRIBUTE_PACKED {
|
|||
|
||||
#define MAX_TX_DEPTH (TX_SIZES - TX_SIZE_CTX_MIN)
|
||||
|
||||
#if CONFIG_CTX1D
|
||||
#define MAX_HVTX_SIZE (1 << 5)
|
||||
#endif // CONFIG_CTX1D
|
||||
|
||||
#define MAX_TX_SIZE_LOG2 (5 + CONFIG_TX64X64)
|
||||
#define MAX_TX_SIZE (1 << MAX_TX_SIZE_LOG2)
|
||||
#define MIN_TX_SIZE_LOG2 2
|
||||
|
|
@ -192,11 +246,9 @@ typedef enum ATTRIBUTE_PACKED {
|
|||
#define MAX_TX_BLOCKS_IN_MAX_SB_LOG2 ((MAX_SB_SIZE_LOG2 - MAX_TX_SIZE_LOG2) * 2)
|
||||
#define MAX_TX_BLOCKS_IN_MAX_SB (1 << MAX_TX_BLOCKS_IN_MAX_SB_LOG2)
|
||||
|
||||
#define MAX_NUM_TXB (1 << (MAX_SB_SIZE_LOG2 - MIN_TX_SIZE_LOG2))
|
||||
|
||||
#if CONFIG_NCOBMC_ADAPT_WEIGHT && CONFIG_MOTION_VAR
|
||||
#if CONFIG_NCOBMC_ADAPT_WEIGHT
|
||||
typedef enum ATTRIBUTE_PACKED {
|
||||
NO_OVERLAP,
|
||||
NCOBMC_MODE_0,
|
||||
NCOBMC_MODE_1,
|
||||
NCOBMC_MODE_2,
|
||||
NCOBMC_MODE_3,
|
||||
|
|
@ -204,20 +256,33 @@ typedef enum ATTRIBUTE_PACKED {
|
|||
NCOBMC_MODE_5,
|
||||
NCOBMC_MODE_6,
|
||||
NCOBMC_MODE_7,
|
||||
NCOBMC_MODE_8,
|
||||
MAX_NCOBMC_MODES
|
||||
} NCOBMC_MODE;
|
||||
// #define MAX_INTRPL_MODES 9
|
||||
ALL_NCOBMC_MODES,
|
||||
#ifdef TWO_MODE
|
||||
MAX_NCOBMC_MODES = NCOBMC_MODE_1 + 1,
|
||||
#else
|
||||
MAX_NCOBMC_MODES = ALL_NCOBMC_MODES,
|
||||
#endif
|
||||
NO_OVERLAP = MAX_NCOBMC_MODES + 1
|
||||
} NCOBMC_MODE;
|
||||
|
||||
typedef enum {
|
||||
ADAPT_OVERLAP_BLOCK_8X8,
|
||||
ADAPT_OVERLAP_BLOCK_16X16,
|
||||
ADAPT_OVERLAP_BLOCK_32X32,
|
||||
ADAPT_OVERLAP_BLOCK_64X64,
|
||||
ADAPT_OVERLAP_BLOCKS,
|
||||
ADAPT_OVERLAP_BLOCK_INVALID = 255
|
||||
} ADAPT_OVERLAP_BLOCK;
|
||||
#endif // CONFIG_NCOBMC_ADAPT_WEIGHT
|
||||
|
||||
// frame transform mode
|
||||
typedef enum {
|
||||
ONLY_4X4 = 0, // only 4x4 transform used
|
||||
ALLOW_8X8 = 1, // allow block transform size up to 8x8
|
||||
ALLOW_16X16 = 2, // allow block transform size up to 16x16
|
||||
ALLOW_32X32 = 3, // allow block transform size up to 32x32
|
||||
ONLY_4X4, // only 4x4 transform used
|
||||
ALLOW_8X8, // allow block transform size up to 8x8
|
||||
ALLOW_16X16, // allow block transform size up to 16x16
|
||||
ALLOW_32X32, // allow block transform size up to 32x32
|
||||
#if CONFIG_TX64X64
|
||||
ALLOW_64X64 = 4, // allow block transform size up to 64x64
|
||||
ALLOW_64X64, // allow block transform size up to 64x64
|
||||
#endif
|
||||
TX_MODE_SELECT, // transform specified for each block
|
||||
TX_MODES,
|
||||
|
|
@ -225,33 +290,33 @@ typedef enum {
|
|||
|
||||
// 1D tx types
|
||||
typedef enum {
|
||||
DCT_1D = 0,
|
||||
ADST_1D = 1,
|
||||
FLIPADST_1D = 2,
|
||||
IDTX_1D = 3,
|
||||
DCT_1D,
|
||||
ADST_1D,
|
||||
FLIPADST_1D,
|
||||
IDTX_1D,
|
||||
// TODO(sarahparker) need to eventually put something here for the
|
||||
// mrc experiment to make this work with the ext-tx pruning functions
|
||||
TX_TYPES_1D = 4,
|
||||
TX_TYPES_1D,
|
||||
} TX_TYPE_1D;
|
||||
|
||||
typedef enum {
|
||||
DCT_DCT = 0, // DCT in both horizontal and vertical
|
||||
ADST_DCT = 1, // ADST in vertical, DCT in horizontal
|
||||
DCT_ADST = 2, // DCT in vertical, ADST in horizontal
|
||||
ADST_ADST = 3, // ADST in both directions
|
||||
DCT_DCT, // DCT in both horizontal and vertical
|
||||
ADST_DCT, // ADST in vertical, DCT in horizontal
|
||||
DCT_ADST, // DCT in vertical, ADST in horizontal
|
||||
ADST_ADST, // ADST in both directions
|
||||
#if CONFIG_EXT_TX
|
||||
FLIPADST_DCT = 4,
|
||||
DCT_FLIPADST = 5,
|
||||
FLIPADST_FLIPADST = 6,
|
||||
ADST_FLIPADST = 7,
|
||||
FLIPADST_ADST = 8,
|
||||
IDTX = 9,
|
||||
V_DCT = 10,
|
||||
H_DCT = 11,
|
||||
V_ADST = 12,
|
||||
H_ADST = 13,
|
||||
V_FLIPADST = 14,
|
||||
H_FLIPADST = 15,
|
||||
FLIPADST_DCT,
|
||||
DCT_FLIPADST,
|
||||
FLIPADST_FLIPADST,
|
||||
ADST_FLIPADST,
|
||||
FLIPADST_ADST,
|
||||
IDTX,
|
||||
V_DCT,
|
||||
H_DCT,
|
||||
V_ADST,
|
||||
H_ADST,
|
||||
V_FLIPADST,
|
||||
H_FLIPADST,
|
||||
#endif // CONFIG_EXT_TX
|
||||
#if CONFIG_MRC_TX
|
||||
MRC_DCT, // DCT in both directions with mrc based bitmask
|
||||
|
|
@ -260,6 +325,28 @@ typedef enum {
|
|||
} TX_TYPE;
|
||||
|
||||
#if CONFIG_EXT_TX
|
||||
typedef enum {
|
||||
// DCT only
|
||||
EXT_TX_SET_DCTONLY,
|
||||
// DCT + Identity only
|
||||
EXT_TX_SET_DCT_IDTX,
|
||||
#if CONFIG_MRC_TX
|
||||
// DCT + MRC_DCT
|
||||
EXT_TX_SET_MRC_DCT,
|
||||
// DCT + MRC_DCT + IDTX
|
||||
EXT_TX_SET_MRC_DCT_IDTX,
|
||||
#endif // CONFIG_MRC_TX
|
||||
// Discrete Trig transforms w/o flip (4) + Identity (1)
|
||||
EXT_TX_SET_DTT4_IDTX,
|
||||
// Discrete Trig transforms w/o flip (4) + Identity (1) + 1D Hor/vert DCT (2)
|
||||
EXT_TX_SET_DTT4_IDTX_1DDCT,
|
||||
// Discrete Trig transforms w/ flip (9) + Identity (1) + 1D Hor/Ver DCT (2)
|
||||
EXT_TX_SET_DTT9_IDTX_1DDCT,
|
||||
// Discrete Trig transforms w/ flip (9) + Identity (1) + 1D Hor/Ver (6)
|
||||
EXT_TX_SET_ALL16,
|
||||
EXT_TX_SET_TYPES
|
||||
} TxSetType;
|
||||
|
||||
#define IS_2D_TRANSFORM(tx_type) (tx_type < IDTX)
|
||||
#else
|
||||
#define IS_2D_TRANSFORM(tx_type) 1
|
||||
|
|
@ -304,14 +391,9 @@ typedef enum {
|
|||
AOM_LAST3_FLAG = 1 << 2,
|
||||
AOM_GOLD_FLAG = 1 << 3,
|
||||
AOM_BWD_FLAG = 1 << 4,
|
||||
#if CONFIG_ALTREF2
|
||||
AOM_ALT2_FLAG = 1 << 5,
|
||||
AOM_ALT_FLAG = 1 << 6,
|
||||
AOM_REFFRAME_ALL = (1 << 7) - 1
|
||||
#else // !CONFIG_ALTREF2
|
||||
AOM_ALT_FLAG = 1 << 5,
|
||||
AOM_REFFRAME_ALL = (1 << 6) - 1
|
||||
#endif // CONFIG_ALTREF2
|
||||
#else // !CONFIG_EXT_REFS
|
||||
AOM_GOLD_FLAG = 1 << 1,
|
||||
AOM_ALT_FLAG = 1 << 2,
|
||||
|
|
@ -323,28 +405,56 @@ typedef enum {
|
|||
#define USE_UNI_COMP_REFS 1
|
||||
|
||||
typedef enum {
|
||||
UNIDIR_COMP_REFERENCE = 0,
|
||||
BIDIR_COMP_REFERENCE = 1,
|
||||
COMP_REFERENCE_TYPES = 2,
|
||||
UNIDIR_COMP_REFERENCE,
|
||||
BIDIR_COMP_REFERENCE,
|
||||
COMP_REFERENCE_TYPES,
|
||||
} COMP_REFERENCE_TYPE;
|
||||
#else // !CONFIG_EXT_COMP_REFS
|
||||
#define USE_UNI_COMP_REFS 0
|
||||
#endif // CONFIG_EXT_COMP_REFS
|
||||
|
||||
typedef enum { PLANE_TYPE_Y = 0, PLANE_TYPE_UV = 1, PLANE_TYPES } PLANE_TYPE;
|
||||
typedef enum { PLANE_TYPE_Y, PLANE_TYPE_UV, PLANE_TYPES } PLANE_TYPE;
|
||||
|
||||
#if CONFIG_CFL
|
||||
// TODO(ltrudeau) this should change based on QP size
|
||||
#define CB_ALPHABET_SIZE 4
|
||||
#define CR_ALPHABET_SIZE 4
|
||||
#define CFL_ALPHABET_SIZE (CB_ALPHABET_SIZE * CR_ALPHABET_SIZE)
|
||||
#define CFL_MAGS_SIZE 7
|
||||
#define CFL_ALPHABET_SIZE_LOG2 4
|
||||
#define CFL_ALPHABET_SIZE (1 << CFL_ALPHABET_SIZE_LOG2)
|
||||
#define CFL_MAGS_SIZE ((2 << CFL_ALPHABET_SIZE_LOG2) + 1)
|
||||
#define CFL_IDX_U(idx) (idx >> CFL_ALPHABET_SIZE_LOG2)
|
||||
#define CFL_IDX_V(idx) (idx & (CFL_ALPHABET_SIZE - 1))
|
||||
|
||||
typedef enum { CFL_PRED_U = 0, CFL_PRED_V = 1, CFL_PRED_PLANES } CFL_PRED_TYPE;
|
||||
typedef enum { CFL_SIGN_NEG = 0, CFL_SIGN_POS = 1, CFL_SIGNS } CFL_SIGN_TYPE;
|
||||
typedef enum { CFL_PRED_U, CFL_PRED_V, CFL_PRED_PLANES } CFL_PRED_TYPE;
|
||||
|
||||
typedef enum {
|
||||
CFL_SIGN_ZERO,
|
||||
CFL_SIGN_NEG,
|
||||
CFL_SIGN_POS,
|
||||
CFL_SIGNS
|
||||
} CFL_SIGN_TYPE;
|
||||
|
||||
// CFL_SIGN_ZERO,CFL_SIGN_ZERO is invalid
|
||||
#define CFL_JOINT_SIGNS (CFL_SIGNS * CFL_SIGNS - 1)
|
||||
// CFL_SIGN_U is equivalent to (js + 1) / 3 for js in 0 to 8
|
||||
#define CFL_SIGN_U(js) (((js + 1) * 11) >> 5)
|
||||
// CFL_SIGN_V is equivalent to (js + 1) % 3 for js in 0 to 8
|
||||
#define CFL_SIGN_V(js) ((js + 1) - CFL_SIGNS * CFL_SIGN_U(js))
|
||||
|
||||
// There is no context when the alpha for a given plane is zero.
|
||||
// So there are 2 fewer contexts than joint signs.
|
||||
#define CFL_ALPHA_CONTEXTS (CFL_JOINT_SIGNS + 1 - CFL_SIGNS)
|
||||
#define CFL_CONTEXT_U(js) (js + 1 - CFL_SIGNS)
|
||||
// Also, the contexts are symmetric under swapping the planes.
|
||||
#define CFL_CONTEXT_V(js) \
|
||||
(CFL_SIGN_V(js) * CFL_SIGNS + CFL_SIGN_U(js) - CFL_SIGNS)
|
||||
#endif
|
||||
|
||||
#if CONFIG_PALETTE
|
||||
typedef enum {
|
||||
PALETTE_MAP,
|
||||
#if CONFIG_MRC_TX
|
||||
MRC_MAP,
|
||||
#endif // CONFIG_MRC_TX
|
||||
COLOR_MAP_TYPES,
|
||||
} COLOR_MAP_TYPE;
|
||||
|
||||
typedef enum {
|
||||
TWO_COLORS,
|
||||
THREE_COLORS,
|
||||
|
|
@ -367,33 +477,29 @@ typedef enum {
|
|||
PALETTE_COLOR_EIGHT,
|
||||
PALETTE_COLORS
|
||||
} PALETTE_COLOR;
|
||||
#endif // CONFIG_PALETTE
|
||||
|
||||
// Note: All directional predictors must be between V_PRED and D63_PRED (both
|
||||
// inclusive).
|
||||
typedef enum ATTRIBUTE_PACKED {
|
||||
DC_PRED, // Average of above and left pixels
|
||||
V_PRED, // Vertical
|
||||
H_PRED, // Horizontal
|
||||
D45_PRED, // Directional 45 deg = round(arctan(1/1) * 180/pi)
|
||||
D135_PRED, // Directional 135 deg = 180 - 45
|
||||
D117_PRED, // Directional 117 deg = 180 - 63
|
||||
D153_PRED, // Directional 153 deg = 180 - 27
|
||||
D207_PRED, // Directional 207 deg = 180 + 27
|
||||
D63_PRED, // Directional 63 deg = round(arctan(2/1) * 180/pi)
|
||||
#if CONFIG_ALT_INTRA
|
||||
DC_PRED, // Average of above and left pixels
|
||||
V_PRED, // Vertical
|
||||
H_PRED, // Horizontal
|
||||
D45_PRED, // Directional 45 deg = round(arctan(1/1) * 180/pi)
|
||||
D135_PRED, // Directional 135 deg = 180 - 45
|
||||
D117_PRED, // Directional 117 deg = 180 - 63
|
||||
D153_PRED, // Directional 153 deg = 180 - 27
|
||||
D207_PRED, // Directional 207 deg = 180 + 27
|
||||
D63_PRED, // Directional 63 deg = round(arctan(2/1) * 180/pi)
|
||||
SMOOTH_PRED, // Combination of horizontal and vertical interpolation
|
||||
#if CONFIG_SMOOTH_HV
|
||||
SMOOTH_V_PRED, // Vertical interpolation
|
||||
SMOOTH_H_PRED, // Horizontal interpolation
|
||||
#endif // CONFIG_SMOOTH_HV
|
||||
#endif // CONFIG_ALT_INTRA
|
||||
TM_PRED, // True-motion
|
||||
NEARESTMV,
|
||||
NEARMV,
|
||||
ZEROMV,
|
||||
NEWMV,
|
||||
#if CONFIG_EXT_INTER
|
||||
#if CONFIG_COMPOUND_SINGLEREF
|
||||
// Single ref compound modes
|
||||
SR_NEAREST_NEARMV,
|
||||
|
|
@ -411,7 +517,6 @@ typedef enum ATTRIBUTE_PACKED {
|
|||
NEW_NEARMV,
|
||||
ZERO_ZEROMV,
|
||||
NEW_NEWMV,
|
||||
#endif // CONFIG_EXT_INTER
|
||||
MB_MODE_COUNT,
|
||||
INTRA_MODES = TM_PRED + 1, // TM_PRED has to be the last intra mode.
|
||||
INTRA_INVALID = MB_MODE_COUNT // For uv_mode in inter blocks
|
||||
|
|
@ -421,23 +526,22 @@ typedef enum ATTRIBUTE_PACKED {
|
|||
// TODO(ltrudeau) Do we really want to pack this?
|
||||
// TODO(ltrudeau) Do we match with PREDICTION_MODE?
|
||||
typedef enum ATTRIBUTE_PACKED {
|
||||
UV_DC_PRED, // Average of above and left pixels
|
||||
UV_V_PRED, // Vertical
|
||||
UV_H_PRED, // Horizontal
|
||||
UV_D45_PRED, // Directional 45 deg = round(arctan(1/1) * 180/pi)
|
||||
UV_D135_PRED, // Directional 135 deg = 180 - 45
|
||||
UV_D117_PRED, // Directional 117 deg = 180 - 63
|
||||
UV_D153_PRED, // Directional 153 deg = 180 - 27
|
||||
UV_D207_PRED, // Directional 207 deg = 180 + 27
|
||||
UV_D63_PRED, // Directional 63 deg = round(arctan(2/1) * 180/pi)
|
||||
#if CONFIG_ALT_INTRA
|
||||
UV_DC_PRED, // Average of above and left pixels
|
||||
UV_V_PRED, // Vertical
|
||||
UV_H_PRED, // Horizontal
|
||||
UV_D45_PRED, // Directional 45 deg = round(arctan(1/1) * 180/pi)
|
||||
UV_D135_PRED, // Directional 135 deg = 180 - 45
|
||||
UV_D117_PRED, // Directional 117 deg = 180 - 63
|
||||
UV_D153_PRED, // Directional 153 deg = 180 - 27
|
||||
UV_D207_PRED, // Directional 207 deg = 180 + 27
|
||||
UV_D63_PRED, // Directional 63 deg = round(arctan(2/1) * 180/pi)
|
||||
UV_SMOOTH_PRED, // Combination of horizontal and vertical interpolation
|
||||
#if CONFIG_SMOOTH_HV
|
||||
UV_SMOOTH_V_PRED, // Vertical interpolation
|
||||
UV_SMOOTH_H_PRED, // Horizontal interpolation
|
||||
#endif // CONFIG_SMOOTH_HV
|
||||
#endif // CONFIG_ALT_INTRA
|
||||
UV_TM_PRED, // True-motion
|
||||
UV_CFL_PRED, // Chroma-from-Luma
|
||||
UV_INTRA_MODES,
|
||||
UV_MODE_INVALID, // For uv_mode in inter blocks
|
||||
} UV_PREDICTION_MODE;
|
||||
|
|
@ -449,47 +553,35 @@ typedef enum ATTRIBUTE_PACKED {
|
|||
#endif // CONFIG_CFL
|
||||
|
||||
typedef enum {
|
||||
SIMPLE_TRANSLATION = 0,
|
||||
SIMPLE_TRANSLATION,
|
||||
#if CONFIG_MOTION_VAR
|
||||
OBMC_CAUSAL, // 2-sided OBMC
|
||||
#endif // CONFIG_MOTION_VAR
|
||||
#if CONFIG_NCOBMC_ADAPT_WEIGHT
|
||||
NCOBMC_ADAPT_WEIGHT,
|
||||
#endif // CONFIG_NCOBMC_ADAPT_WEIGHT
|
||||
#endif // CONFIG_MOTION_VAR
|
||||
#if CONFIG_WARPED_MOTION
|
||||
WARPED_CAUSAL, // 2-sided WARPED
|
||||
#endif // CONFIG_WARPED_MOTION
|
||||
#if CONFIG_NCOBMC_ADAPT_WEIGHT
|
||||
NCOBMC_ADAPT_WEIGHT,
|
||||
#endif
|
||||
MOTION_MODES
|
||||
#if CONFIG_NCOBMC_ADAPT_WEIGHT && CONFIG_WARPED_MOTION
|
||||
,
|
||||
OBMC_FAMILY_MODES = NCOBMC_ADAPT_WEIGHT + 1
|
||||
#endif
|
||||
} MOTION_MODE;
|
||||
|
||||
#if CONFIG_NCOBMC_ADAPT_WEIGHT
|
||||
typedef enum {
|
||||
ADAPT_OVERLAP_BLOCK_8X8,
|
||||
ADAPT_OVERLAP_BLOCK_16X16,
|
||||
ADAPT_OVERLAP_BLOCK_32X32,
|
||||
ADAPT_OVERLAP_BLOCK_64X64,
|
||||
ADAPT_OVERLAP_BLOCKS,
|
||||
ADAPT_OVERLAP_BLOCK_INVALID = 255
|
||||
} ADAPT_OVERLAP_BLOCK;
|
||||
#endif
|
||||
|
||||
#if CONFIG_EXT_INTER
|
||||
#if CONFIG_INTERINTRA
|
||||
typedef enum {
|
||||
II_DC_PRED = 0,
|
||||
II_DC_PRED,
|
||||
II_V_PRED,
|
||||
II_H_PRED,
|
||||
#if CONFIG_ALT_INTRA
|
||||
II_SMOOTH_PRED,
|
||||
#else
|
||||
II_TM_PRED,
|
||||
#endif // CONFIG_ALT_INTRA
|
||||
INTERINTRA_MODES
|
||||
} INTERINTRA_MODE;
|
||||
#endif
|
||||
|
||||
typedef enum {
|
||||
COMPOUND_AVERAGE = 0,
|
||||
COMPOUND_AVERAGE,
|
||||
#if CONFIG_WEDGE
|
||||
COMPOUND_WEDGE,
|
||||
#endif // CONFIG_WEDGE
|
||||
|
|
@ -498,7 +590,6 @@ typedef enum {
|
|||
#endif // CONFIG_COMPOUND_SEGMENT
|
||||
COMPOUND_TYPES,
|
||||
} COMPOUND_TYPE;
|
||||
#endif // CONFIG_EXT_INTER
|
||||
|
||||
// TODO(huisu): Consider adding FILTER_SMOOTH_PRED to "FILTER_INTRA_MODE".
|
||||
#if CONFIG_FILTER_INTRA
|
||||
|
|
@ -523,13 +614,11 @@ typedef enum {
|
|||
|
||||
#define INTER_MODES (1 + NEWMV - NEARESTMV)
|
||||
|
||||
#if CONFIG_EXT_INTER
|
||||
#if CONFIG_COMPOUND_SINGLEREF
|
||||
#define INTER_SINGLEREF_COMP_MODES (1 + SR_NEW_NEWMV - SR_NEAREST_NEARMV)
|
||||
#endif // CONFIG_COMPOUND_SINGLEREF
|
||||
|
||||
#define INTER_COMPOUND_MODES (1 + NEW_NEWMV - NEAREST_NEARESTMV)
|
||||
#endif // CONFIG_EXT_INTER
|
||||
|
||||
#define SKIP_CONTEXTS 3
|
||||
|
||||
|
|
@ -553,7 +642,6 @@ typedef enum {
|
|||
#define SKIP_NEARESTMV_SUB8X8_OFFSET 11
|
||||
|
||||
#define INTER_MODE_CONTEXTS 7
|
||||
#if CONFIG_DELTA_Q
|
||||
#define DELTA_Q_SMALL 3
|
||||
#define DELTA_Q_PROBS (DELTA_Q_SMALL)
|
||||
#define DEFAULT_DELTA_Q_RES 4
|
||||
|
|
@ -562,7 +650,6 @@ typedef enum {
|
|||
#define DELTA_LF_PROBS (DELTA_LF_SMALL)
|
||||
#define DEFAULT_DELTA_LF_RES 2
|
||||
#endif
|
||||
#endif
|
||||
|
||||
/* Segment Feature Masks */
|
||||
#define MAX_MV_REF_CANDIDATES 2
|
||||
|
|
@ -583,9 +670,9 @@ typedef enum {
|
|||
#define UNI_COMP_REF_CONTEXTS 3
|
||||
#endif // CONFIG_EXT_COMP_REFS
|
||||
|
||||
#if CONFIG_EXT_INTER && CONFIG_COMPOUND_SINGLEREF
|
||||
#if CONFIG_COMPOUND_SINGLEREF
|
||||
#define COMP_INTER_MODE_CONTEXTS 4
|
||||
#endif // CONFIG_EXT_INTER && CONFIG_COMPOUND_SINGLEREF
|
||||
#endif // CONFIG_COMPOUND_SINGLEREF
|
||||
|
||||
#if CONFIG_VAR_TX
|
||||
#define TXFM_PARTITION_CONTEXTS ((TX_SIZES - TX_8X8) * 6 - 2)
|
||||
|
|
@ -601,14 +688,8 @@ typedef uint8_t TXFM_CONTEXT;
|
|||
#define LAST3_FRAME 3
|
||||
#define GOLDEN_FRAME 4
|
||||
#define BWDREF_FRAME 5
|
||||
|
||||
#if CONFIG_ALTREF2
|
||||
#define ALTREF2_FRAME 6
|
||||
#define ALTREF_FRAME 7
|
||||
#else // !CONFIG_ALTREF2
|
||||
#define ALTREF_FRAME 6
|
||||
#endif // CONFIG_ALTREF2
|
||||
|
||||
#define LAST_REF_FRAMES (LAST3_FRAME - LAST_FRAME + 1)
|
||||
#else // !CONFIG_EXT_REFS
|
||||
#define GOLDEN_FRAME 2
|
||||
|
|
@ -651,9 +732,9 @@ typedef enum {
|
|||
|
||||
#if CONFIG_LOOP_RESTORATION
|
||||
typedef enum {
|
||||
RESTORE_NONE = 0,
|
||||
RESTORE_WIENER = 1,
|
||||
RESTORE_SGRPROJ = 2,
|
||||
RESTORE_NONE,
|
||||
RESTORE_WIENER,
|
||||
RESTORE_SGRPROJ,
|
||||
RESTORE_SWITCHABLE,
|
||||
RESTORE_SWITCHABLE_TYPES = RESTORE_SWITCHABLE,
|
||||
RESTORE_TYPES,
|
||||
|
|
@ -662,7 +743,7 @@ typedef enum {
|
|||
|
||||
#if CONFIG_FRAME_SUPERRES
|
||||
#define SUPERRES_SCALE_BITS 3
|
||||
#define SUPERRES_SCALE_NUMERATOR_MIN 8
|
||||
#define SUPERRES_SCALE_DENOMINATOR_MIN 8
|
||||
#endif // CONFIG_FRAME_SUPERRES
|
||||
|
||||
#if CONFIG_LPF_DIRECT
|
||||
|
|
@ -678,6 +759,27 @@ typedef enum {
|
|||
} FILTER_DEGREE;
|
||||
#endif // CONFIG_LPF_DIRECT
|
||||
|
||||
#if CONFIG_OBU
|
||||
// R19
|
||||
typedef enum {
|
||||
OBU_SEQUENCE_HEADER = 1,
|
||||
OBU_TD = 2,
|
||||
OBU_FRAME_HEADER = 3,
|
||||
OBU_TILE_GROUP = 4,
|
||||
OBU_METADATA = 5,
|
||||
OBU_PADDING = 15,
|
||||
} OBU_TYPE;
|
||||
#endif
|
||||
|
||||
#if CONFIG_LGT_FROM_PRED
|
||||
#define LGT_SIZES 2
|
||||
// Note: at least one of LGT_FROM_PRED_INTRA and LGT_FROM_PRED_INTER must be 1
|
||||
#define LGT_FROM_PRED_INTRA 1
|
||||
#define LGT_FROM_PRED_INTER 1
|
||||
// LGT_SL_INTRA: LGTs with a mode-dependent first self-loop and a break point
|
||||
#define LGT_SL_INTRA 0
|
||||
#endif // CONFIG_LGT_FROM_PRED
|
||||
|
||||
#ifdef __cplusplus
|
||||
} // extern "C"
|
||||
#endif
|
||||
|
|
|
|||
94
third_party/aom/av1/common/filter.c
vendored
94
third_party/aom/av1/common/filter.c
vendored
|
|
@ -51,7 +51,6 @@ DECLARE_ALIGNED(16, static const int16_t,
|
|||
#if USE_EXTRA_FILTER
|
||||
DECLARE_ALIGNED(256, static const InterpKernel,
|
||||
sub_pel_filters_8[SUBPEL_SHIFTS]) = {
|
||||
#if CONFIG_FILTER_7BIT
|
||||
{ 0, 0, 0, 128, 0, 0, 0, 0 }, { 0, 2, -6, 126, 8, -2, 0, 0 },
|
||||
{ 0, 2, -10, 122, 18, -4, 0, 0 }, { 0, 2, -12, 116, 28, -8, 2, 0 },
|
||||
{ 0, 2, -14, 110, 38, -10, 2, 0 }, { 0, 2, -14, 102, 48, -12, 2, 0 },
|
||||
|
|
@ -60,22 +59,10 @@ DECLARE_ALIGNED(256, static const InterpKernel,
|
|||
{ 0, 2, -12, 58, 94, -16, 2, 0 }, { 0, 2, -12, 48, 102, -14, 2, 0 },
|
||||
{ 0, 2, -10, 38, 110, -14, 2, 0 }, { 0, 2, -8, 28, 116, -12, 2, 0 },
|
||||
{ 0, 0, -4, 18, 122, -10, 2, 0 }, { 0, 0, -2, 8, 126, -6, 2, 0 }
|
||||
#else
|
||||
// intfilt 0.575
|
||||
{ 0, 0, 0, 128, 0, 0, 0, 0 }, { 0, 1, -5, 126, 8, -3, 1, 0 },
|
||||
{ -1, 3, -10, 123, 18, -6, 2, -1 }, { -1, 4, -14, 118, 27, -9, 3, 0 },
|
||||
{ -1, 5, -16, 112, 37, -12, 4, -1 }, { -1, 5, -18, 105, 48, -14, 4, -1 },
|
||||
{ -1, 6, -19, 97, 58, -17, 5, -1 }, { -1, 6, -20, 88, 68, -18, 6, -1 },
|
||||
{ -1, 6, -19, 78, 78, -19, 6, -1 }, { -1, 6, -18, 68, 88, -20, 6, -1 },
|
||||
{ -1, 5, -17, 58, 97, -19, 6, -1 }, { -1, 4, -14, 48, 105, -18, 5, -1 },
|
||||
{ -1, 4, -12, 37, 112, -16, 5, -1 }, { 0, 3, -9, 27, 118, -14, 4, -1 },
|
||||
{ -1, 2, -6, 18, 123, -10, 3, -1 }, { 0, 1, -3, 8, 126, -5, 1, 0 },
|
||||
#endif
|
||||
};
|
||||
|
||||
DECLARE_ALIGNED(256, static const InterpKernel,
|
||||
sub_pel_filters_regular_uv[SUBPEL_SHIFTS]) = {
|
||||
#if CONFIG_FILTER_7BIT
|
||||
{ 0, 0, 0, 128, 0, 0, 0, 0 }, { 0, 2, -6, 126, 8, -2, 0, 0 },
|
||||
{ 0, 2, -10, 122, 18, -4, 0, 0 }, { 0, 2, -12, 116, 28, -8, 2, 0 },
|
||||
{ 0, 2, -14, 110, 38, -10, 2, 0 }, { 0, 2, -14, 102, 48, -12, 2, 0 },
|
||||
|
|
@ -84,17 +71,6 @@ DECLARE_ALIGNED(256, static const InterpKernel,
|
|||
{ 0, 2, -12, 58, 94, -16, 2, 0 }, { 0, 2, -12, 48, 102, -14, 2, 0 },
|
||||
{ 0, 2, -10, 38, 110, -14, 2, 0 }, { 0, 2, -8, 28, 116, -12, 2, 0 },
|
||||
{ 0, 0, -4, 18, 122, -10, 2, 0 }, { 0, 0, -2, 8, 126, -6, 2, 0 }
|
||||
#else
|
||||
// intfilt 0.575
|
||||
{ 0, 0, 0, 128, 0, 0, 0, 0 }, { 0, 1, -5, 126, 8, -3, 1, 0 },
|
||||
{ -1, 3, -10, 123, 18, -6, 2, -1 }, { -1, 4, -14, 118, 27, -9, 3, 0 },
|
||||
{ -1, 5, -16, 112, 37, -12, 4, -1 }, { -1, 5, -18, 105, 48, -14, 4, -1 },
|
||||
{ -1, 6, -19, 97, 58, -17, 5, -1 }, { -1, 6, -20, 88, 68, -18, 6, -1 },
|
||||
{ -1, 6, -19, 78, 78, -19, 6, -1 }, { -1, 6, -18, 68, 88, -20, 6, -1 },
|
||||
{ -1, 5, -17, 58, 97, -19, 6, -1 }, { -1, 4, -14, 48, 105, -18, 5, -1 },
|
||||
{ -1, 4, -12, 37, 112, -16, 5, -1 }, { 0, 3, -9, 27, 118, -14, 4, -1 },
|
||||
{ -1, 2, -6, 18, 123, -10, 3, -1 }, { 0, 1, -3, 8, 126, -5, 1, 0 },
|
||||
#endif
|
||||
};
|
||||
|
||||
#if USE_12TAP_FILTER
|
||||
|
|
@ -134,7 +110,6 @@ DECLARE_ALIGNED(256, static const int16_t,
|
|||
#else
|
||||
DECLARE_ALIGNED(256, static const InterpKernel,
|
||||
sub_pel_filters_8sharp[SUBPEL_SHIFTS]) = {
|
||||
#if CONFIG_FILTER_7BIT
|
||||
{ 0, 0, 0, 128, 0, 0, 0, 0 }, { -2, 2, -6, 126, 8, -2, 2, 0 },
|
||||
{ -2, 6, -12, 124, 16, -6, 4, -2 }, { -2, 8, -18, 120, 26, -10, 6, -2 },
|
||||
{ -4, 10, -22, 116, 38, -14, 6, -2 }, { -4, 10, -22, 108, 48, -18, 8, -2 },
|
||||
|
|
@ -143,16 +118,6 @@ DECLARE_ALIGNED(256, static const InterpKernel,
|
|||
{ -2, 8, -20, 60, 100, -24, 10, -4 }, { -2, 8, -18, 48, 108, -22, 10, -4 },
|
||||
{ -2, 6, -14, 38, 116, -22, 10, -4 }, { -2, 6, -10, 26, 120, -18, 8, -2 },
|
||||
{ -2, 4, -6, 16, 124, -12, 6, -2 }, { 0, 2, -2, 8, 126, -6, 2, -2 }
|
||||
#else
|
||||
{ 0, 0, 0, 128, 0, 0, 0, 0 }, { -1, 3, -7, 127, 8, -3, 1, 0 },
|
||||
{ -2, 5, -13, 125, 17, -6, 3, -1 }, { -3, 7, -17, 121, 27, -10, 5, -2 },
|
||||
{ -4, 9, -20, 115, 37, -13, 6, -2 }, { -4, 10, -23, 108, 48, -16, 8, -3 },
|
||||
{ -4, 10, -24, 100, 59, -19, 9, -3 }, { -4, 11, -24, 90, 70, -21, 10, -4 },
|
||||
{ -4, 11, -23, 80, 80, -23, 11, -4 }, { -4, 10, -21, 70, 90, -24, 11, -4 },
|
||||
{ -3, 9, -19, 59, 100, -24, 10, -4 }, { -3, 8, -16, 48, 108, -23, 10, -4 },
|
||||
{ -2, 6, -13, 37, 115, -20, 9, -4 }, { -2, 5, -10, 27, 121, -17, 7, -3 },
|
||||
{ -1, 3, -6, 17, 125, -13, 5, -2 }, { 0, 1, -3, 8, 127, -7, 3, -1 }
|
||||
#endif
|
||||
};
|
||||
#endif
|
||||
|
||||
|
|
@ -184,7 +149,6 @@ DECLARE_ALIGNED(256, static const InterpKernel,
|
|||
|
||||
DECLARE_ALIGNED(256, static const InterpKernel,
|
||||
sub_pel_filters_8smooth[SUBPEL_SHIFTS]) = {
|
||||
#if CONFIG_FILTER_7BIT
|
||||
{ 0, 0, 0, 128, 0, 0, 0, 0 }, { 0, 2, 28, 62, 34, 2, 0, 0 },
|
||||
{ 0, 0, 26, 62, 36, 4, 0, 0 }, { 0, 0, 22, 62, 40, 4, 0, 0 },
|
||||
{ 0, 0, 20, 60, 42, 6, 0, 0 }, { 0, 0, 18, 58, 44, 8, 0, 0 },
|
||||
|
|
@ -193,22 +157,10 @@ DECLARE_ALIGNED(256, static const InterpKernel,
|
|||
{ 0, 0, 10, 46, 56, 16, 0, 0 }, { 0, 0, 8, 44, 58, 18, 0, 0 },
|
||||
{ 0, 0, 6, 42, 60, 20, 0, 0 }, { 0, 0, 4, 40, 62, 22, 0, 0 },
|
||||
{ 0, 0, 4, 36, 62, 26, 0, 0 }, { 0, 0, 2, 34, 62, 28, 2, 0 }
|
||||
#else
|
||||
// freqmultiplier = 0.8
|
||||
{ 0, 0, 0, 128, 0, 0, 0, 0 }, { 0, -5, 13, 102, 24, -7, 1, 0 },
|
||||
{ 0, -4, 8, 100, 31, -8, 1, 0 }, { 0, -3, 4, 97, 37, -8, 1, 0 },
|
||||
{ 0, -2, 0, 94, 44, -9, 1, 0 }, { 0, -2, -3, 90, 51, -9, 1, 0 },
|
||||
{ 0, -1, -5, 84, 59, -9, 0, 0 }, { 0, 0, -7, 79, 65, -9, 0, 0 },
|
||||
{ 0, 0, -8, 72, 72, -8, 0, 0 }, { 0, 0, -9, 65, 79, -7, 0, 0 },
|
||||
{ 0, 0, -9, 59, 84, -5, -1, 0 }, { 0, 1, -9, 51, 90, -3, -2, 0 },
|
||||
{ 0, 1, -9, 44, 94, 0, -2, 0 }, { 0, 1, -8, 37, 97, 4, -3, 0 },
|
||||
{ 0, 1, -8, 31, 100, 8, -4, 0 }, { 0, 1, -7, 24, 102, 13, -5, 0 },
|
||||
#endif
|
||||
};
|
||||
|
||||
DECLARE_ALIGNED(256, static const InterpKernel,
|
||||
sub_pel_filters_smooth_uv[SUBPEL_SHIFTS]) = {
|
||||
#if CONFIG_FILTER_7BIT
|
||||
{ 0, 0, 0, 128, 0, 0, 0, 0 }, { 0, 2, 28, 62, 34, 2, 0, 0 },
|
||||
{ 0, 0, 26, 62, 36, 4, 0, 0 }, { 0, 0, 22, 62, 40, 4, 0, 0 },
|
||||
{ 0, 0, 20, 60, 42, 6, 0, 0 }, { 0, 0, 18, 58, 44, 8, 0, 0 },
|
||||
|
|
@ -217,23 +169,11 @@ DECLARE_ALIGNED(256, static const InterpKernel,
|
|||
{ 0, 0, 10, 46, 56, 16, 0, 0 }, { 0, 0, 8, 44, 58, 18, 0, 0 },
|
||||
{ 0, 0, 6, 42, 60, 20, 0, 0 }, { 0, 0, 4, 40, 62, 22, 0, 0 },
|
||||
{ 0, 0, 4, 36, 62, 26, 0, 0 }, { 0, 0, 2, 34, 62, 28, 2, 0 }
|
||||
#else
|
||||
// freqmultiplier = 0.8
|
||||
{ 0, 0, 0, 128, 0, 0, 0, 0 }, { 0, -5, 13, 102, 24, -7, 1, 0 },
|
||||
{ 0, -4, 8, 100, 31, -8, 1, 0 }, { 0, -3, 4, 97, 37, -8, 1, 0 },
|
||||
{ 0, -2, 0, 94, 44, -9, 1, 0 }, { 0, -2, -3, 90, 51, -9, 1, 0 },
|
||||
{ 0, -1, -5, 84, 59, -9, 0, 0 }, { 0, 0, -7, 79, 65, -9, 0, 0 },
|
||||
{ 0, 0, -8, 72, 72, -8, 0, 0 }, { 0, 0, -9, 65, 79, -7, 0, 0 },
|
||||
{ 0, 0, -9, 59, 84, -5, -1, 0 }, { 0, 1, -9, 51, 90, -3, -2, 0 },
|
||||
{ 0, 1, -9, 44, 94, 0, -2, 0 }, { 0, 1, -8, 37, 97, 4, -3, 0 },
|
||||
{ 0, 1, -8, 31, 100, 8, -4, 0 }, { 0, 1, -7, 24, 102, 13, -5, 0 },
|
||||
#endif
|
||||
};
|
||||
#else // USE_EXTRA_FILTER
|
||||
#else // USE_EXTRA_FILTER
|
||||
|
||||
DECLARE_ALIGNED(256, static const InterpKernel,
|
||||
sub_pel_filters_8[SUBPEL_SHIFTS]) = {
|
||||
#if CONFIG_FILTER_7BIT
|
||||
{ 0, 0, 0, 128, 0, 0, 0, 0 }, { 0, 2, -6, 126, 8, -2, 0, 0 },
|
||||
{ 0, 2, -10, 122, 18, -4, 0, 0 }, { 0, 2, -12, 116, 28, -8, 2, 0 },
|
||||
{ 0, 2, -14, 110, 38, -10, 2, 0 }, { 0, 2, -14, 102, 48, -12, 2, 0 },
|
||||
|
|
@ -242,21 +182,10 @@ DECLARE_ALIGNED(256, static const InterpKernel,
|
|||
{ 0, 2, -12, 58, 94, -16, 2, 0 }, { 0, 2, -12, 48, 102, -14, 2, 0 },
|
||||
{ 0, 2, -10, 38, 110, -14, 2, 0 }, { 0, 2, -8, 28, 116, -12, 2, 0 },
|
||||
{ 0, 0, -4, 18, 122, -10, 2, 0 }, { 0, 0, -2, 8, 126, -6, 2, 0 }
|
||||
#else
|
||||
{ 0, 0, 0, 128, 0, 0, 0, 0 }, { 0, 1, -5, 126, 8, -3, 1, 0 },
|
||||
{ -1, 3, -10, 122, 18, -6, 2, 0 }, { -1, 4, -13, 118, 27, -9, 3, -1 },
|
||||
{ -1, 4, -16, 112, 37, -11, 4, -1 }, { -1, 5, -18, 105, 48, -14, 4, -1 },
|
||||
{ -1, 5, -19, 97, 58, -16, 5, -1 }, { -1, 6, -19, 88, 68, -18, 5, -1 },
|
||||
{ -1, 6, -19, 78, 78, -19, 6, -1 }, { -1, 5, -18, 68, 88, -19, 6, -1 },
|
||||
{ -1, 5, -16, 58, 97, -19, 5, -1 }, { -1, 4, -14, 48, 105, -18, 5, -1 },
|
||||
{ -1, 4, -11, 37, 112, -16, 4, -1 }, { -1, 3, -9, 27, 118, -13, 4, -1 },
|
||||
{ 0, 2, -6, 18, 122, -10, 3, -1 }, { 0, 1, -3, 8, 126, -5, 1, 0 }
|
||||
#endif
|
||||
};
|
||||
|
||||
DECLARE_ALIGNED(256, static const InterpKernel,
|
||||
sub_pel_filters_8sharp[SUBPEL_SHIFTS]) = {
|
||||
#if CONFIG_FILTER_7BIT
|
||||
{ 0, 0, 0, 128, 0, 0, 0, 0 }, { -2, 2, -6, 126, 8, -2, 2, 0 },
|
||||
{ -2, 6, -12, 124, 16, -6, 4, -2 }, { -2, 8, -18, 120, 26, -10, 6, -2 },
|
||||
{ -4, 10, -22, 116, 38, -14, 6, -2 }, { -4, 10, -22, 108, 48, -18, 8, -2 },
|
||||
|
|
@ -265,21 +194,10 @@ DECLARE_ALIGNED(256, static const InterpKernel,
|
|||
{ -2, 8, -20, 60, 100, -24, 10, -4 }, { -2, 8, -18, 48, 108, -22, 10, -4 },
|
||||
{ -2, 6, -14, 38, 116, -22, 10, -4 }, { -2, 6, -10, 26, 120, -18, 8, -2 },
|
||||
{ -2, 4, -6, 16, 124, -12, 6, -2 }, { 0, 2, -2, 8, 126, -6, 2, -2 }
|
||||
#else
|
||||
{ 0, 0, 0, 128, 0, 0, 0, 0 }, { -1, 3, -7, 127, 8, -3, 1, 0 },
|
||||
{ -2, 5, -13, 125, 17, -6, 3, -1 }, { -3, 7, -17, 121, 27, -10, 5, -2 },
|
||||
{ -4, 9, -20, 115, 37, -13, 6, -2 }, { -4, 10, -23, 108, 48, -16, 8, -3 },
|
||||
{ -4, 10, -24, 100, 59, -19, 9, -3 }, { -4, 11, -24, 90, 70, -21, 10, -4 },
|
||||
{ -4, 11, -23, 80, 80, -23, 11, -4 }, { -4, 10, -21, 70, 90, -24, 11, -4 },
|
||||
{ -3, 9, -19, 59, 100, -24, 10, -4 }, { -3, 8, -16, 48, 108, -23, 10, -4 },
|
||||
{ -2, 6, -13, 37, 115, -20, 9, -4 }, { -2, 5, -10, 27, 121, -17, 7, -3 },
|
||||
{ -1, 3, -6, 17, 125, -13, 5, -2 }, { 0, 1, -3, 8, 127, -7, 3, -1 }
|
||||
#endif
|
||||
};
|
||||
|
||||
DECLARE_ALIGNED(256, static const InterpKernel,
|
||||
sub_pel_filters_8smooth[SUBPEL_SHIFTS]) = {
|
||||
#if CONFIG_FILTER_7BIT
|
||||
{ 0, 0, 0, 128, 0, 0, 0, 0 }, { 0, 2, 28, 62, 34, 2, 0, 0 },
|
||||
{ 0, 0, 26, 62, 36, 4, 0, 0 }, { 0, 0, 22, 62, 40, 4, 0, 0 },
|
||||
{ 0, 0, 20, 60, 42, 6, 0, 0 }, { 0, 0, 18, 58, 44, 8, 0, 0 },
|
||||
|
|
@ -288,16 +206,6 @@ DECLARE_ALIGNED(256, static const InterpKernel,
|
|||
{ 0, 0, 10, 46, 56, 16, 0, 0 }, { 0, 0, 8, 44, 58, 18, 0, 0 },
|
||||
{ 0, 0, 6, 42, 60, 20, 0, 0 }, { 0, 0, 4, 40, 62, 22, 0, 0 },
|
||||
{ 0, 0, 4, 36, 62, 26, 0, 0 }, { 0, 0, 2, 34, 62, 28, 2, 0 }
|
||||
#else
|
||||
{ 0, 0, 0, 128, 0, 0, 0, 0 }, { -3, -1, 32, 64, 38, 1, -3, 0 },
|
||||
{ -2, -2, 29, 63, 41, 2, -3, 0 }, { -2, -2, 26, 63, 43, 4, -4, 0 },
|
||||
{ -2, -3, 24, 62, 46, 5, -4, 0 }, { -2, -3, 21, 60, 49, 7, -4, 0 },
|
||||
{ -1, -4, 18, 59, 51, 9, -4, 0 }, { -1, -4, 16, 57, 53, 12, -4, -1 },
|
||||
{ -1, -4, 14, 55, 55, 14, -4, -1 }, { -1, -4, 12, 53, 57, 16, -4, -1 },
|
||||
{ 0, -4, 9, 51, 59, 18, -4, -1 }, { 0, -4, 7, 49, 60, 21, -3, -2 },
|
||||
{ 0, -4, 5, 46, 62, 24, -3, -2 }, { 0, -4, 4, 43, 63, 26, -2, -2 },
|
||||
{ 0, -3, 2, 41, 63, 29, -2, -2 }, { 0, -3, 1, 38, 64, 32, -1, -3 }
|
||||
#endif
|
||||
};
|
||||
#endif // USE_EXTRA_FILTER
|
||||
|
||||
|
|
|
|||
47
third_party/aom/av1/common/filter.h
vendored
47
third_party/aom/av1/common/filter.h
vendored
|
|
@ -12,6 +12,8 @@
|
|||
#ifndef AV1_COMMON_FILTER_H_
|
||||
#define AV1_COMMON_FILTER_H_
|
||||
|
||||
#include <assert.h>
|
||||
|
||||
#include "./aom_config.h"
|
||||
#include "aom/aom_integer.h"
|
||||
#include "aom_dsp/aom_filter.h"
|
||||
|
|
@ -30,10 +32,10 @@ extern "C" {
|
|||
typedef enum {
|
||||
EIGHTTAP_REGULAR,
|
||||
EIGHTTAP_SMOOTH,
|
||||
MULTITAP_SHARP,
|
||||
#if USE_EXTRA_FILTER
|
||||
EIGHTTAP_SMOOTH2,
|
||||
#endif // USE_EXTRA_FILTER
|
||||
MULTITAP_SHARP,
|
||||
BILINEAR,
|
||||
#if USE_EXTRA_FILTER
|
||||
EIGHTTAP_SHARP,
|
||||
|
|
@ -51,6 +53,49 @@ typedef enum {
|
|||
#endif
|
||||
} InterpFilter;
|
||||
|
||||
// With CONFIG_DUAL_FILTER, pack two InterpFilter's into a uint32_t: since
|
||||
// there are at most 10 filters, we can use 16 bits for each and have more than
|
||||
// enough space. This reduces argument passing and unifies the operation of
|
||||
// setting a (pair of) filters.
|
||||
//
|
||||
// Without CONFIG_DUAL_FILTER,
|
||||
#if CONFIG_DUAL_FILTER
|
||||
typedef uint32_t InterpFilters;
|
||||
static INLINE InterpFilter av1_extract_interp_filter(InterpFilters filters,
|
||||
int x_filter) {
|
||||
return (InterpFilter)((filters >> (x_filter ? 16 : 0)) & 0xffff);
|
||||
}
|
||||
|
||||
static INLINE InterpFilters av1_make_interp_filters(InterpFilter y_filter,
|
||||
InterpFilter x_filter) {
|
||||
uint16_t y16 = y_filter & 0xffff;
|
||||
uint16_t x16 = x_filter & 0xffff;
|
||||
return y16 | ((uint32_t)x16 << 16);
|
||||
}
|
||||
|
||||
static INLINE InterpFilters av1_broadcast_interp_filter(InterpFilter filter) {
|
||||
return av1_make_interp_filters(filter, filter);
|
||||
}
|
||||
#else
|
||||
typedef InterpFilter InterpFilters;
|
||||
static INLINE InterpFilter av1_extract_interp_filter(InterpFilters filters,
|
||||
int x_filter) {
|
||||
#ifdef NDEBUG
|
||||
(void)x_filter;
|
||||
#endif
|
||||
assert(!x_filter);
|
||||
return filters;
|
||||
}
|
||||
|
||||
static INLINE InterpFilters av1_broadcast_interp_filter(InterpFilter filter) {
|
||||
return filter;
|
||||
}
|
||||
#endif
|
||||
|
||||
static INLINE InterpFilter av1_unswitchable_filter(InterpFilter filter) {
|
||||
return filter == SWITCHABLE ? EIGHTTAP_REGULAR : filter;
|
||||
}
|
||||
|
||||
#if USE_EXTRA_FILTER
|
||||
#define LOG_SWITCHABLE_FILTERS \
|
||||
3 /* (1 << LOG_SWITCHABLE_FILTERS) > SWITCHABLE_FILTERS */
|
||||
|
|
|
|||
1395
third_party/aom/av1/common/idct.c
vendored
1395
third_party/aom/av1/common/idct.c
vendored
File diff suppressed because it is too large
Load diff
53
third_party/aom/av1/common/idct.h
vendored
53
third_party/aom/av1/common/idct.h
vendored
|
|
@ -26,13 +26,28 @@
|
|||
extern "C" {
|
||||
#endif
|
||||
|
||||
// TODO(kslu) move the common stuff in idct.h to av1_txfm.h or txfm_common.h
|
||||
typedef void (*transform_1d)(const tran_low_t *, tran_low_t *);
|
||||
|
||||
typedef struct {
|
||||
transform_1d cols, rows; // vertical and horizontal
|
||||
} transform_2d;
|
||||
|
||||
#if CONFIG_LGT
|
||||
int get_lgt4(const TxfmParam *txfm_param, int is_col,
|
||||
const tran_high_t **lgtmtx);
|
||||
int get_lgt8(const TxfmParam *txfm_param, int is_col,
|
||||
const tran_high_t **lgtmtx);
|
||||
#endif // CONFIG_LGT
|
||||
|
||||
#if CONFIG_LGT_FROM_PRED
|
||||
void get_lgt4_from_pred(const TxfmParam *txfm_param, int is_col,
|
||||
const tran_high_t **lgtmtx, int ntx);
|
||||
void get_lgt8_from_pred(const TxfmParam *txfm_param, int is_col,
|
||||
const tran_high_t **lgtmtx, int ntx);
|
||||
void get_lgt16up_from_pred(const TxfmParam *txfm_param, int is_col,
|
||||
const tran_high_t **lgtmtx, int ntx);
|
||||
#endif // CONFIG_LGT_FROM_PRED
|
||||
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
typedef void (*highbd_transform_1d)(const tran_low_t *, tran_low_t *, int bd);
|
||||
|
||||
|
|
@ -53,9 +68,12 @@ void av1_inv_txfm_add(const tran_low_t *input, uint8_t *dest, int stride,
|
|||
TxfmParam *txfm_param);
|
||||
void av1_inverse_transform_block(const MACROBLOCKD *xd,
|
||||
const tran_low_t *dqcoeff,
|
||||
#if CONFIG_LGT
|
||||
#if CONFIG_LGT_FROM_PRED
|
||||
PREDICTION_MODE mode,
|
||||
#endif
|
||||
#if CONFIG_MRC_TX && SIGNAL_ANY_MRC_MASK
|
||||
uint8_t *mrc_mask,
|
||||
#endif // CONFIG_MRC_TX && SIGNAL_ANY_MRC_MASK
|
||||
TX_TYPE tx_type, TX_SIZE tx_size, uint8_t *dst,
|
||||
int stride, int eob);
|
||||
void av1_inverse_transform_block_facade(MACROBLOCKD *xd, int plane, int block,
|
||||
|
|
@ -72,37 +90,6 @@ void av1_highbd_inv_txfm_add_8x4(const tran_low_t *input, uint8_t *dest,
|
|||
void av1_highbd_inv_txfm_add(const tran_low_t *input, uint8_t *dest, int stride,
|
||||
TxfmParam *txfm_param);
|
||||
|
||||
#if CONFIG_DPCM_INTRA
|
||||
void av1_dpcm_inv_txfm_add_4_c(const tran_low_t *input, int stride,
|
||||
TX_TYPE_1D tx_type, uint8_t *dest);
|
||||
void av1_dpcm_inv_txfm_add_8_c(const tran_low_t *input, int stride,
|
||||
TX_TYPE_1D tx_type, uint8_t *dest);
|
||||
void av1_dpcm_inv_txfm_add_16_c(const tran_low_t *input, int stride,
|
||||
TX_TYPE_1D tx_type, uint8_t *dest);
|
||||
void av1_dpcm_inv_txfm_add_32_c(const tran_low_t *input, int stride,
|
||||
TX_TYPE_1D tx_type, uint8_t *dest);
|
||||
typedef void (*dpcm_inv_txfm_add_func)(const tran_low_t *input, int stride,
|
||||
TX_TYPE_1D tx_type, uint8_t *dest);
|
||||
dpcm_inv_txfm_add_func av1_get_dpcm_inv_txfm_add_func(int tx_length);
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
void av1_hbd_dpcm_inv_txfm_add_4_c(const tran_low_t *input, int stride,
|
||||
TX_TYPE_1D tx_type, int bd, uint16_t *dest,
|
||||
int dir);
|
||||
void av1_hbd_dpcm_inv_txfm_add_8_c(const tran_low_t *input, int stride,
|
||||
TX_TYPE_1D tx_type, int bd, uint16_t *dest,
|
||||
int dir);
|
||||
void av1_hbd_dpcm_inv_txfm_add_16_c(const tran_low_t *input, int stride,
|
||||
TX_TYPE_1D tx_type, int bd, uint16_t *dest,
|
||||
int dir);
|
||||
void av1_hbd_dpcm_inv_txfm_add_32_c(const tran_low_t *input, int stride,
|
||||
TX_TYPE_1D tx_type, int bd, uint16_t *dest,
|
||||
int dir);
|
||||
typedef void (*hbd_dpcm_inv_txfm_add_func)(const tran_low_t *input, int stride,
|
||||
TX_TYPE_1D tx_type, int bd,
|
||||
uint16_t *dest, int dir);
|
||||
hbd_dpcm_inv_txfm_add_func av1_get_hbd_dpcm_inv_txfm_add_func(int tx_length);
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
#endif // CONFIG_DPCM_INTRA
|
||||
#ifdef __cplusplus
|
||||
} // extern "C"
|
||||
#endif
|
||||
|
|
|
|||
|
|
@ -1,97 +0,0 @@
|
|||
/*
|
||||
* Copyright (c) 2016, Alliance for Open Media. All rights reserved
|
||||
*
|
||||
* This source code is subject to the terms of the BSD 2 Clause License and
|
||||
* the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
|
||||
* was not distributed with this source code in the LICENSE file, you can
|
||||
* obtain it at www.aomedia.org/license/software. If the Alliance for Open
|
||||
* Media Patent License 1.0 was not distributed with this source code in the
|
||||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#include <assert.h>
|
||||
#include <stdio.h>
|
||||
|
||||
#include "./aom_config.h"
|
||||
#include "./av1_rtcd.h"
|
||||
#include "av1/common/common.h"
|
||||
#include "av1/common/blockd.h"
|
||||
#include "aom_dsp/mips/inv_txfm_dspr2.h"
|
||||
#include "aom_dsp/txfm_common.h"
|
||||
#include "aom_ports/mem.h"
|
||||
|
||||
#if HAVE_DSPR2
|
||||
void av1_iht16x16_256_add_dspr2(const int16_t *input, uint8_t *dest, int pitch,
|
||||
TxfmParam *txfm_param) {
|
||||
int i, j;
|
||||
DECLARE_ALIGNED(32, int16_t, out[16 * 16]);
|
||||
int16_t *outptr = out;
|
||||
int16_t temp_out[16];
|
||||
uint32_t pos = 45;
|
||||
int tx_type = txfm_param->tx_type;
|
||||
|
||||
/* bit positon for extract from acc */
|
||||
__asm__ __volatile__("wrdsp %[pos], 1 \n\t" : : [pos] "r"(pos));
|
||||
|
||||
switch (tx_type) {
|
||||
case DCT_DCT: // DCT in both horizontal and vertical
|
||||
idct16_rows_dspr2(input, outptr, 16);
|
||||
idct16_cols_add_blk_dspr2(out, dest, pitch);
|
||||
break;
|
||||
case ADST_DCT: // ADST in vertical, DCT in horizontal
|
||||
idct16_rows_dspr2(input, outptr, 16);
|
||||
|
||||
outptr = out;
|
||||
|
||||
for (i = 0; i < 16; ++i) {
|
||||
iadst16_dspr2(outptr, temp_out);
|
||||
|
||||
for (j = 0; j < 16; ++j)
|
||||
dest[j * pitch + i] = clip_pixel(ROUND_POWER_OF_TWO(temp_out[j], 6) +
|
||||
dest[j * pitch + i]);
|
||||
outptr += 16;
|
||||
}
|
||||
break;
|
||||
case DCT_ADST: // DCT in vertical, ADST in horizontal
|
||||
{
|
||||
int16_t temp_in[16 * 16];
|
||||
|
||||
for (i = 0; i < 16; ++i) {
|
||||
/* prefetch row */
|
||||
prefetch_load((const uint8_t *)(input + 16));
|
||||
|
||||
iadst16_dspr2(input, outptr);
|
||||
input += 16;
|
||||
outptr += 16;
|
||||
}
|
||||
|
||||
for (i = 0; i < 16; ++i)
|
||||
for (j = 0; j < 16; ++j) temp_in[j * 16 + i] = out[i * 16 + j];
|
||||
|
||||
idct16_cols_add_blk_dspr2(temp_in, dest, pitch);
|
||||
} break;
|
||||
case ADST_ADST: // ADST in both directions
|
||||
{
|
||||
int16_t temp_in[16];
|
||||
|
||||
for (i = 0; i < 16; ++i) {
|
||||
/* prefetch row */
|
||||
prefetch_load((const uint8_t *)(input + 16));
|
||||
|
||||
iadst16_dspr2(input, outptr);
|
||||
input += 16;
|
||||
outptr += 16;
|
||||
}
|
||||
|
||||
for (i = 0; i < 16; ++i) {
|
||||
for (j = 0; j < 16; ++j) temp_in[j] = out[j * 16 + i];
|
||||
iadst16_dspr2(temp_in, temp_out);
|
||||
for (j = 0; j < 16; ++j)
|
||||
dest[j * pitch + i] = clip_pixel(ROUND_POWER_OF_TWO(temp_out[j], 6) +
|
||||
dest[j * pitch + i]);
|
||||
}
|
||||
} break;
|
||||
default: printf("av1_short_iht16x16_add_dspr2 : Invalid tx_type\n"); break;
|
||||
}
|
||||
}
|
||||
#endif // #if HAVE_DSPR2
|
||||
|
|
@ -1,91 +0,0 @@
|
|||
/*
|
||||
* Copyright (c) 2016, Alliance for Open Media. All rights reserved
|
||||
*
|
||||
* This source code is subject to the terms of the BSD 2 Clause License and
|
||||
* the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
|
||||
* was not distributed with this source code in the LICENSE file, you can
|
||||
* obtain it at www.aomedia.org/license/software. If the Alliance for Open
|
||||
* Media Patent License 1.0 was not distributed with this source code in the
|
||||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#include <assert.h>
|
||||
#include <stdio.h>
|
||||
|
||||
#include "./aom_config.h"
|
||||
#include "./av1_rtcd.h"
|
||||
#include "av1/common/common.h"
|
||||
#include "av1/common/blockd.h"
|
||||
#include "aom_dsp/mips/inv_txfm_dspr2.h"
|
||||
#include "aom_dsp/txfm_common.h"
|
||||
#include "aom_ports/mem.h"
|
||||
|
||||
#if HAVE_DSPR2
|
||||
void av1_iht4x4_16_add_dspr2(const int16_t *input, uint8_t *dest,
|
||||
int dest_stride, TxfmParam *txfm_param) {
|
||||
int i, j;
|
||||
DECLARE_ALIGNED(32, int16_t, out[4 * 4]);
|
||||
int16_t *outptr = out;
|
||||
int16_t temp_in[4 * 4], temp_out[4];
|
||||
uint32_t pos = 45;
|
||||
int tx_type = txfm_param->tx_type;
|
||||
|
||||
/* bit positon for extract from acc */
|
||||
__asm__ __volatile__("wrdsp %[pos], 1 \n\t"
|
||||
:
|
||||
: [pos] "r"(pos));
|
||||
|
||||
switch (tx_type) {
|
||||
case DCT_DCT: // DCT in both horizontal and vertical
|
||||
aom_idct4_rows_dspr2(input, outptr);
|
||||
aom_idct4_columns_add_blk_dspr2(&out[0], dest, dest_stride);
|
||||
break;
|
||||
case ADST_DCT: // ADST in vertical, DCT in horizontal
|
||||
aom_idct4_rows_dspr2(input, outptr);
|
||||
|
||||
outptr = out;
|
||||
|
||||
for (i = 0; i < 4; ++i) {
|
||||
iadst4_dspr2(outptr, temp_out);
|
||||
|
||||
for (j = 0; j < 4; ++j)
|
||||
dest[j * dest_stride + i] = clip_pixel(
|
||||
ROUND_POWER_OF_TWO(temp_out[j], 4) + dest[j * dest_stride + i]);
|
||||
|
||||
outptr += 4;
|
||||
}
|
||||
break;
|
||||
case DCT_ADST: // DCT in vertical, ADST in horizontal
|
||||
for (i = 0; i < 4; ++i) {
|
||||
iadst4_dspr2(input, outptr);
|
||||
input += 4;
|
||||
outptr += 4;
|
||||
}
|
||||
|
||||
for (i = 0; i < 4; ++i) {
|
||||
for (j = 0; j < 4; ++j) {
|
||||
temp_in[i * 4 + j] = out[j * 4 + i];
|
||||
}
|
||||
}
|
||||
aom_idct4_columns_add_blk_dspr2(&temp_in[0], dest, dest_stride);
|
||||
break;
|
||||
case ADST_ADST: // ADST in both directions
|
||||
for (i = 0; i < 4; ++i) {
|
||||
iadst4_dspr2(input, outptr);
|
||||
input += 4;
|
||||
outptr += 4;
|
||||
}
|
||||
|
||||
for (i = 0; i < 4; ++i) {
|
||||
for (j = 0; j < 4; ++j) temp_in[j] = out[j * 4 + i];
|
||||
iadst4_dspr2(temp_in, temp_out);
|
||||
|
||||
for (j = 0; j < 4; ++j)
|
||||
dest[j * dest_stride + i] = clip_pixel(
|
||||
ROUND_POWER_OF_TWO(temp_out[j], 4) + dest[j * dest_stride + i]);
|
||||
}
|
||||
break;
|
||||
default: printf("av1_short_iht4x4_add_dspr2 : Invalid tx_type\n"); break;
|
||||
}
|
||||
}
|
||||
#endif // #if HAVE_DSPR2
|
||||
|
|
@ -1,86 +0,0 @@
|
|||
/*
|
||||
* Copyright (c) 2016, Alliance for Open Media. All rights reserved
|
||||
*
|
||||
* This source code is subject to the terms of the BSD 2 Clause License and
|
||||
* the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
|
||||
* was not distributed with this source code in the LICENSE file, you can
|
||||
* obtain it at www.aomedia.org/license/software. If the Alliance for Open
|
||||
* Media Patent License 1.0 was not distributed with this source code in the
|
||||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#include <assert.h>
|
||||
#include <stdio.h>
|
||||
|
||||
#include "./aom_config.h"
|
||||
#include "./av1_rtcd.h"
|
||||
#include "av1/common/common.h"
|
||||
#include "av1/common/blockd.h"
|
||||
#include "aom_dsp/mips/inv_txfm_dspr2.h"
|
||||
#include "aom_dsp/txfm_common.h"
|
||||
#include "aom_ports/mem.h"
|
||||
|
||||
#if HAVE_DSPR2
|
||||
void av1_iht8x8_64_add_dspr2(const int16_t *input, uint8_t *dest,
|
||||
int dest_stride, TxfmParam *txfm_param) {
|
||||
int i, j;
|
||||
DECLARE_ALIGNED(32, int16_t, out[8 * 8]);
|
||||
int16_t *outptr = out;
|
||||
int16_t temp_in[8 * 8], temp_out[8];
|
||||
uint32_t pos = 45;
|
||||
int tx_type = txfm_param->tx_type;
|
||||
|
||||
/* bit positon for extract from acc */
|
||||
__asm__ __volatile__("wrdsp %[pos], 1 \n\t" : : [pos] "r"(pos));
|
||||
|
||||
switch (tx_type) {
|
||||
case DCT_DCT: // DCT in both horizontal and vertical
|
||||
idct8_rows_dspr2(input, outptr, 8);
|
||||
idct8_columns_add_blk_dspr2(&out[0], dest, dest_stride);
|
||||
break;
|
||||
case ADST_DCT: // ADST in vertical, DCT in horizontal
|
||||
idct8_rows_dspr2(input, outptr, 8);
|
||||
|
||||
for (i = 0; i < 8; ++i) {
|
||||
iadst8_dspr2(&out[i * 8], temp_out);
|
||||
|
||||
for (j = 0; j < 8; ++j)
|
||||
dest[j * dest_stride + i] = clip_pixel(
|
||||
ROUND_POWER_OF_TWO(temp_out[j], 5) + dest[j * dest_stride + i]);
|
||||
}
|
||||
break;
|
||||
case DCT_ADST: // DCT in vertical, ADST in horizontal
|
||||
for (i = 0; i < 8; ++i) {
|
||||
iadst8_dspr2(input, outptr);
|
||||
input += 8;
|
||||
outptr += 8;
|
||||
}
|
||||
|
||||
for (i = 0; i < 8; ++i) {
|
||||
for (j = 0; j < 8; ++j) {
|
||||
temp_in[i * 8 + j] = out[j * 8 + i];
|
||||
}
|
||||
}
|
||||
idct8_columns_add_blk_dspr2(&temp_in[0], dest, dest_stride);
|
||||
break;
|
||||
case ADST_ADST: // ADST in both directions
|
||||
for (i = 0; i < 8; ++i) {
|
||||
iadst8_dspr2(input, outptr);
|
||||
input += 8;
|
||||
outptr += 8;
|
||||
}
|
||||
|
||||
for (i = 0; i < 8; ++i) {
|
||||
for (j = 0; j < 8; ++j) temp_in[j] = out[j * 8 + i];
|
||||
|
||||
iadst8_dspr2(temp_in, temp_out);
|
||||
|
||||
for (j = 0; j < 8; ++j)
|
||||
dest[j * dest_stride + i] = clip_pixel(
|
||||
ROUND_POWER_OF_TWO(temp_out[j], 5) + dest[j * dest_stride + i]);
|
||||
}
|
||||
break;
|
||||
default: printf("av1_short_iht8x8_add_dspr2 : Invalid tx_type\n"); break;
|
||||
}
|
||||
}
|
||||
#endif // #if HAVE_DSPR2
|
||||
|
|
@ -19,7 +19,7 @@ void av1_iht16x16_256_add_msa(const int16_t *input, uint8_t *dst,
|
|||
int32_t i;
|
||||
DECLARE_ALIGNED(32, int16_t, out[16 * 16]);
|
||||
int16_t *out_ptr = &out[0];
|
||||
int32_t tx_type = txfm_param->tx_type;
|
||||
const TX_TYPE tx_type = txfm_param->tx_type;
|
||||
|
||||
switch (tx_type) {
|
||||
case DCT_DCT:
|
||||
|
|
|
|||
|
|
@ -17,7 +17,7 @@
|
|||
void av1_iht4x4_16_add_msa(const int16_t *input, uint8_t *dst,
|
||||
int32_t dst_stride, TxfmParam *txfm_param) {
|
||||
v8i16 in0, in1, in2, in3;
|
||||
int32_t tx_type = txfm_param->tx_type;
|
||||
const TX_TYPE tx_type = txfm_param->tx_type;
|
||||
|
||||
/* load vector elements of 4x4 block */
|
||||
LD4x4_SH(input, in0, in1, in2, in3);
|
||||
|
|
|
|||
|
|
@ -17,7 +17,7 @@
|
|||
void av1_iht8x8_64_add_msa(const int16_t *input, uint8_t *dst,
|
||||
int32_t dst_stride, TxfmParam *txfm_param) {
|
||||
v8i16 in0, in1, in2, in3, in4, in5, in6, in7;
|
||||
int32_t tx_type = txfm_param->tx_type;
|
||||
const TX_TYPE tx_type = txfm_param->tx_type;
|
||||
|
||||
/* load vector elements of 8x8 block */
|
||||
LD_SH8(input, 8, in0, in1, in2, in3, in4, in5, in6, in7);
|
||||
|
|
|
|||
83
third_party/aom/av1/common/mv.h
vendored
83
third_party/aom/av1/common/mv.h
vendored
|
|
@ -20,6 +20,8 @@
|
|||
extern "C" {
|
||||
#endif
|
||||
|
||||
#define INVALID_MV 0x80008000
|
||||
|
||||
typedef struct mv {
|
||||
int16_t row;
|
||||
int16_t col;
|
||||
|
|
@ -88,10 +90,12 @@ typedef enum {
|
|||
// GLOBAL_TRANS_TYPES 7 - up to full homography
|
||||
#define GLOBAL_TRANS_TYPES 4
|
||||
|
||||
#if GLOBAL_TRANS_TYPES > 4
|
||||
// First bit indicates whether using identity or not
|
||||
// GLOBAL_TYPE_BITS=ceiling(log2(GLOBAL_TRANS_TYPES-1)) is the
|
||||
// number of bits needed to cover the remaining possibilities
|
||||
#define GLOBAL_TYPE_BITS (get_msb(2 * GLOBAL_TRANS_TYPES - 3))
|
||||
#endif // GLOBAL_TRANS_TYPES > 4
|
||||
|
||||
typedef struct {
|
||||
#if CONFIG_GLOBAL_MOTION
|
||||
|
|
@ -116,14 +120,14 @@ typedef struct {
|
|||
int16_t alpha, beta, gamma, delta;
|
||||
} WarpedMotionParams;
|
||||
|
||||
static INLINE void set_default_warp_params(WarpedMotionParams *wm) {
|
||||
static const int32_t default_wm_mat[8] = {
|
||||
0, 0, (1 << WARPEDMODEL_PREC_BITS), 0, 0, (1 << WARPEDMODEL_PREC_BITS), 0, 0
|
||||
};
|
||||
memset(wm, 0, sizeof(*wm));
|
||||
memcpy(wm->wmmat, default_wm_mat, sizeof(wm->wmmat));
|
||||
wm->wmtype = IDENTITY;
|
||||
}
|
||||
/* clang-format off */
|
||||
static const WarpedMotionParams default_warp_params = {
|
||||
IDENTITY,
|
||||
{ 0, 0, (1 << WARPEDMODEL_PREC_BITS), 0, 0, (1 << WARPEDMODEL_PREC_BITS), 0,
|
||||
0 },
|
||||
0, 0, 0, 0
|
||||
};
|
||||
/* clang-format on */
|
||||
#endif // CONFIG_GLOBAL_MOTION || CONFIG_WARPED_MOTION
|
||||
|
||||
#if CONFIG_GLOBAL_MOTION
|
||||
|
|
@ -202,21 +206,70 @@ static INLINE int convert_to_trans_prec(int allow_hp, int coor) {
|
|||
else
|
||||
return ROUND_POWER_OF_TWO_SIGNED(coor, WARPEDMODEL_PREC_BITS - 2) * 2;
|
||||
}
|
||||
#if CONFIG_AMVR
|
||||
static INLINE void integer_mv_precision(MV *mv) {
|
||||
int mod = (mv->row % 8);
|
||||
if (mod != 0) {
|
||||
mv->row -= mod;
|
||||
if (abs(mod) > 4) {
|
||||
if (mod > 0) {
|
||||
mv->row += 8;
|
||||
} else {
|
||||
mv->row -= 8;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Convert a global motion translation vector (which may have more bits than a
|
||||
// regular motion vector) into a motion vector
|
||||
mod = (mv->col % 8);
|
||||
if (mod != 0) {
|
||||
mv->col -= mod;
|
||||
if (abs(mod) > 4) {
|
||||
if (mod > 0) {
|
||||
mv->col += 8;
|
||||
} else {
|
||||
mv->col -= 8;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
#endif
|
||||
// Convert a global motion vector into a motion vector at the centre of the
|
||||
// given block.
|
||||
//
|
||||
// The resulting motion vector will have three fractional bits of precision. If
|
||||
// allow_hp is zero, the bottom bit will always be zero. If CONFIG_AMVR and
|
||||
// is_integer is true, the bottom three bits will be zero (so the motion vector
|
||||
// represents an integer)
|
||||
static INLINE int_mv gm_get_motion_vector(const WarpedMotionParams *gm,
|
||||
int allow_hp, BLOCK_SIZE bsize,
|
||||
int mi_col, int mi_row,
|
||||
int block_idx) {
|
||||
int mi_col, int mi_row, int block_idx
|
||||
#if CONFIG_AMVR
|
||||
,
|
||||
int is_integer
|
||||
#endif
|
||||
) {
|
||||
const int unify_bsize = CONFIG_CB4X4;
|
||||
int_mv res;
|
||||
const int32_t *mat = gm->wmmat;
|
||||
int x, y, tx, ty;
|
||||
|
||||
if (gm->wmtype == TRANSLATION) {
|
||||
// All global motion vectors are stored with WARPEDMODEL_PREC_BITS (16)
|
||||
// bits of fractional precision. The offset for a translation is stored in
|
||||
// entries 0 and 1. For translations, all but the top three (two if
|
||||
// cm->allow_high_precision_mv is false) fractional bits are always zero.
|
||||
//
|
||||
// After the right shifts, there are 3 fractional bits of precision. If
|
||||
// allow_hp is false, the bottom bit is always zero (so we don't need a
|
||||
// call to convert_to_trans_prec here)
|
||||
res.as_mv.row = gm->wmmat[0] >> GM_TRANS_ONLY_PREC_DIFF;
|
||||
res.as_mv.col = gm->wmmat[1] >> GM_TRANS_ONLY_PREC_DIFF;
|
||||
assert(IMPLIES(1 & (res.as_mv.row | res.as_mv.col), allow_hp));
|
||||
#if CONFIG_AMVR
|
||||
if (is_integer) {
|
||||
integer_mv_precision(&res.as_mv);
|
||||
}
|
||||
#endif
|
||||
return res;
|
||||
}
|
||||
|
||||
|
|
@ -256,6 +309,12 @@ static INLINE int_mv gm_get_motion_vector(const WarpedMotionParams *gm,
|
|||
|
||||
res.as_mv.row = ty;
|
||||
res.as_mv.col = tx;
|
||||
|
||||
#if CONFIG_AMVR
|
||||
if (is_integer) {
|
||||
integer_mv_precision(&res.as_mv);
|
||||
}
|
||||
#endif
|
||||
return res;
|
||||
}
|
||||
|
||||
|
|
|
|||
1473
third_party/aom/av1/common/mvref_common.c
vendored
1473
third_party/aom/av1/common/mvref_common.c
vendored
File diff suppressed because it is too large
Load diff
85
third_party/aom/av1/common/mvref_common.h
vendored
85
third_party/aom/av1/common/mvref_common.h
vendored
|
|
@ -19,6 +19,8 @@ extern "C" {
|
|||
#endif
|
||||
|
||||
#define MVREF_NEIGHBOURS 9
|
||||
#define MVREF_ROWS 3
|
||||
#define MVREF_COLS 4
|
||||
|
||||
typedef struct position {
|
||||
int row;
|
||||
|
|
@ -51,19 +53,16 @@ static const int mode_2_counter[] = {
|
|||
9, // D153_PRED
|
||||
9, // D207_PRED
|
||||
9, // D63_PRED
|
||||
#if CONFIG_ALT_INTRA
|
||||
9, // SMOOTH_PRED
|
||||
#if CONFIG_SMOOTH_HV
|
||||
9, // SMOOTH_V_PRED
|
||||
9, // SMOOTH_H_PRED
|
||||
#endif // CONFIG_SMOOTH_HV
|
||||
#endif // CONFIG_ALT_INTRA
|
||||
9, // TM_PRED
|
||||
0, // NEARESTMV
|
||||
0, // NEARMV
|
||||
3, // ZEROMV
|
||||
1, // NEWMV
|
||||
#if CONFIG_EXT_INTER
|
||||
#if CONFIG_COMPOUND_SINGLEREF
|
||||
0, // SR_NEAREST_NEARMV
|
||||
// 1, // SR_NEAREST_NEWMV
|
||||
|
|
@ -79,7 +78,6 @@ static const int mode_2_counter[] = {
|
|||
1, // NEW_NEARMV
|
||||
3, // ZERO_ZEROMV
|
||||
1, // NEW_NEWMV
|
||||
#endif // CONFIG_EXT_INTER
|
||||
};
|
||||
|
||||
// There are 3^3 different combinations of 3 counts that can be either 0,1 or
|
||||
|
|
@ -209,11 +207,46 @@ static INLINE int is_inside(const TileInfo *const tile, int mi_col, int mi_row,
|
|||
}
|
||||
}
|
||||
|
||||
static INLINE void lower_mv_precision(MV *mv, int allow_hp) {
|
||||
if (!allow_hp) {
|
||||
if (mv->row & 1) mv->row += (mv->row > 0 ? -1 : 1);
|
||||
if (mv->col & 1) mv->col += (mv->col > 0 ? -1 : 1);
|
||||
static INLINE int find_valid_row_offset(const TileInfo *const tile, int mi_row,
|
||||
int mi_rows, const AV1_COMMON *cm,
|
||||
int row_offset) {
|
||||
#if CONFIG_DEPENDENT_HORZTILES
|
||||
const int dependent_horz_tile_flag = cm->dependent_horz_tiles;
|
||||
#else
|
||||
const int dependent_horz_tile_flag = 0;
|
||||
(void)cm;
|
||||
#endif
|
||||
if (dependent_horz_tile_flag && !tile->tg_horz_boundary)
|
||||
return clamp(row_offset, -mi_row, mi_rows - mi_row - 1);
|
||||
else
|
||||
return clamp(row_offset, tile->mi_row_start - mi_row,
|
||||
tile->mi_row_end - mi_row - 1);
|
||||
}
|
||||
|
||||
static INLINE int find_valid_col_offset(const TileInfo *const tile, int mi_col,
|
||||
int col_offset) {
|
||||
return clamp(col_offset, tile->mi_col_start - mi_col,
|
||||
tile->mi_col_end - mi_col - 1);
|
||||
}
|
||||
|
||||
static INLINE void lower_mv_precision(MV *mv, int allow_hp
|
||||
#if CONFIG_AMVR
|
||||
,
|
||||
int is_integer
|
||||
#endif
|
||||
) {
|
||||
#if CONFIG_AMVR
|
||||
if (is_integer) {
|
||||
integer_mv_precision(mv);
|
||||
} else {
|
||||
#endif
|
||||
if (!allow_hp) {
|
||||
if (mv->row & 1) mv->row += (mv->row > 0 ? -1 : 1);
|
||||
if (mv->col & 1) mv->col += (mv->col > 0 ? -1 : 1);
|
||||
}
|
||||
#if CONFIG_AMVR
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
static INLINE uint8_t av1_get_pred_diff_ctx(const int_mv pred_mv,
|
||||
|
|
@ -280,10 +313,8 @@ static MV_REFERENCE_FRAME ref_frame_map[COMP_REFS][2] = {
|
|||
{ LAST_FRAME, BWDREF_FRAME }, { LAST2_FRAME, BWDREF_FRAME },
|
||||
{ LAST3_FRAME, BWDREF_FRAME }, { GOLDEN_FRAME, BWDREF_FRAME },
|
||||
|
||||
#if CONFIG_ALTREF2
|
||||
{ LAST_FRAME, ALTREF2_FRAME }, { LAST2_FRAME, ALTREF2_FRAME },
|
||||
{ LAST3_FRAME, ALTREF2_FRAME }, { GOLDEN_FRAME, ALTREF2_FRAME },
|
||||
#endif // CONFIG_ALTREF2
|
||||
|
||||
{ LAST_FRAME, ALTREF_FRAME }, { LAST2_FRAME, ALTREF_FRAME },
|
||||
{ LAST3_FRAME, ALTREF_FRAME }, { GOLDEN_FRAME, ALTREF_FRAME }
|
||||
|
|
@ -357,39 +388,49 @@ static INLINE uint8_t av1_drl_ctx(const CANDIDATE_MV *ref_mv_stack,
|
|||
return 0;
|
||||
}
|
||||
|
||||
#if CONFIG_FRAME_MARKER
|
||||
void av1_setup_frame_buf_refs(AV1_COMMON *cm);
|
||||
#if CONFIG_FRAME_SIGN_BIAS
|
||||
void av1_setup_frame_sign_bias(AV1_COMMON *cm);
|
||||
#endif // CONFIG_FRAME_SIGN_BIAS
|
||||
#if CONFIG_MFMV
|
||||
void av1_setup_motion_field(AV1_COMMON *cm);
|
||||
#endif // CONFIG_MFMV
|
||||
#endif // CONFIG_FRAME_MARKER
|
||||
|
||||
void av1_copy_frame_mvs(const AV1_COMMON *const cm, MODE_INFO *mi, int mi_row,
|
||||
int mi_col, int x_mis, int y_mis);
|
||||
|
||||
typedef void (*find_mv_refs_sync)(void *const data, int mi_row);
|
||||
void av1_find_mv_refs(const AV1_COMMON *cm, const MACROBLOCKD *xd,
|
||||
MODE_INFO *mi, MV_REFERENCE_FRAME ref_frame,
|
||||
uint8_t *ref_mv_count, CANDIDATE_MV *ref_mv_stack,
|
||||
#if CONFIG_EXT_INTER
|
||||
int16_t *compound_mode_context,
|
||||
#endif // CONFIG_EXT_INTER
|
||||
int_mv *mv_ref_list, int mi_row, int mi_col,
|
||||
find_mv_refs_sync sync, void *const data,
|
||||
int16_t *mode_context);
|
||||
int16_t *compound_mode_context, int_mv *mv_ref_list,
|
||||
int mi_row, int mi_col, find_mv_refs_sync sync,
|
||||
void *const data, int16_t *mode_context);
|
||||
|
||||
// check a list of motion vectors by sad score using a number rows of pixels
|
||||
// above and a number cols of pixels in the left to select the one with best
|
||||
// score to use as ref motion vector
|
||||
#if CONFIG_AMVR
|
||||
void av1_find_best_ref_mvs(int allow_hp, int_mv *mvlist, int_mv *nearest_mv,
|
||||
int_mv *near_mv, int is_integer);
|
||||
#else
|
||||
void av1_find_best_ref_mvs(int allow_hp, int_mv *mvlist, int_mv *nearest_mv,
|
||||
int_mv *near_mv);
|
||||
#endif
|
||||
|
||||
void av1_append_sub8x8_mvs_for_idx(const AV1_COMMON *cm, MACROBLOCKD *xd,
|
||||
int block, int ref, int mi_row, int mi_col,
|
||||
CANDIDATE_MV *ref_mv_stack,
|
||||
uint8_t *ref_mv_count,
|
||||
#if CONFIG_EXT_INTER
|
||||
int_mv *mv_list,
|
||||
#endif // CONFIG_EXT_INTER
|
||||
uint8_t *ref_mv_count, int_mv *mv_list,
|
||||
int_mv *nearest_mv, int_mv *near_mv);
|
||||
|
||||
#if CONFIG_EXT_INTER
|
||||
// This function keeps a mode count for a given MB/SB
|
||||
void av1_update_mv_context(const AV1_COMMON *cm, const MACROBLOCKD *xd,
|
||||
MODE_INFO *mi, MV_REFERENCE_FRAME ref_frame,
|
||||
int_mv *mv_ref_list, int block, int mi_row,
|
||||
int mi_col, int16_t *mode_context);
|
||||
#endif // CONFIG_EXT_INTER
|
||||
|
||||
#if CONFIG_WARPED_MOTION
|
||||
#if WARPED_MOTION_SORT_SAMPLES
|
||||
|
|
|
|||
1181
third_party/aom/av1/common/ncobmc_kernels.c
vendored
Normal file
1181
third_party/aom/av1/common/ncobmc_kernels.c
vendored
Normal file
File diff suppressed because it is too large
Load diff
22
third_party/aom/av1/common/ncobmc_kernels.h
vendored
Normal file
22
third_party/aom/av1/common/ncobmc_kernels.h
vendored
Normal file
|
|
@ -0,0 +1,22 @@
|
|||
/*
|
||||
* Copyright (c) 2016, Alliance for Open Media. All rights reserved
|
||||
*
|
||||
* This source code is subject to the terms of the BSD 2 Clause License and
|
||||
* the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
|
||||
* was not distributed with this source code in the LICENSE file, you can
|
||||
* obtain it at www.aomedia.org/license/software. If the Alliance for Open
|
||||
* Media Patent License 1.0 was not distributed with this source code in the
|
||||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#include <stdio.h>
|
||||
#include "av1/common/enums.h"
|
||||
#include "av1/common/onyxc_int.h"
|
||||
#include "av1/common/common.h"
|
||||
|
||||
#ifndef AV1_COMMON_NCOBMC_KERNELS_H_
|
||||
#define AV1_COMMON_NCOBMC_KERNELS_H_
|
||||
|
||||
void get_default_ncobmc_kernels(AV1_COMMON *cm);
|
||||
|
||||
#endif // AV1_COMMON_NCOBMC_KERNELS_H_
|
||||
96
third_party/aom/av1/common/obmc.h
vendored
Normal file
96
third_party/aom/av1/common/obmc.h
vendored
Normal file
|
|
@ -0,0 +1,96 @@
|
|||
/*
|
||||
* Copyright (c) 2017, Alliance for Open Media. All rights reserved
|
||||
*
|
||||
* This source code is subject to the terms of the BSD 2 Clause License and
|
||||
* the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
|
||||
* was not distributed with this source code in the LICENSE file, you can
|
||||
* obtain it at www.aomedia.org/license/software. If the Alliance for Open
|
||||
* Media Patent License 1.0 was not distributed with this source code in the
|
||||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#ifndef AV1_COMMON_OBMC_H_
|
||||
#define AV1_COMMON_OBMC_H_
|
||||
|
||||
#if CONFIG_MOTION_VAR
|
||||
typedef void (*overlappable_nb_visitor_t)(MACROBLOCKD *xd, int rel_mi_pos,
|
||||
uint8_t nb_mi_size, MODE_INFO *nb_mi,
|
||||
void *fun_ctxt);
|
||||
|
||||
static INLINE void foreach_overlappable_nb_above(const AV1_COMMON *cm,
|
||||
MACROBLOCKD *xd, int mi_col,
|
||||
int nb_max,
|
||||
overlappable_nb_visitor_t fun,
|
||||
void *fun_ctxt) {
|
||||
if (!xd->up_available) return;
|
||||
|
||||
int nb_count = 0;
|
||||
|
||||
// prev_row_mi points into the mi array, starting at the beginning of the
|
||||
// previous row.
|
||||
MODE_INFO **prev_row_mi = xd->mi - mi_col - 1 * xd->mi_stride;
|
||||
const int end_col = AOMMIN(mi_col + xd->n8_w, cm->mi_cols);
|
||||
uint8_t mi_step;
|
||||
for (int above_mi_col = mi_col; above_mi_col < end_col && nb_count < nb_max;
|
||||
above_mi_col += mi_step) {
|
||||
MODE_INFO **above_mi = prev_row_mi + above_mi_col;
|
||||
mi_step = AOMMIN(mi_size_wide[above_mi[0]->mbmi.sb_type],
|
||||
mi_size_wide[BLOCK_64X64]);
|
||||
#if CONFIG_CHROMA_SUB8X8
|
||||
// If we're considering a block with width 4, it should be treated as
|
||||
// half of a pair of blocks with chroma information in the second. Move
|
||||
// above_mi_col back to the start of the pair if needed, set above_mbmi
|
||||
// to point at the block with chroma information, and set mi_step to 2 to
|
||||
// step over the entire pair at the end of the iteration.
|
||||
if (mi_step == 1) {
|
||||
above_mi_col &= ~1;
|
||||
above_mi = prev_row_mi + above_mi_col + 1;
|
||||
mi_step = 2;
|
||||
}
|
||||
#endif // CONFIG_CHROMA_SUB8X8
|
||||
MB_MODE_INFO *above_mbmi = &above_mi[0]->mbmi;
|
||||
if (is_neighbor_overlappable(above_mbmi)) {
|
||||
++nb_count;
|
||||
fun(xd, above_mi_col - mi_col, AOMMIN(xd->n8_w, mi_step), *above_mi,
|
||||
fun_ctxt);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static INLINE void foreach_overlappable_nb_left(const AV1_COMMON *cm,
|
||||
MACROBLOCKD *xd, int mi_row,
|
||||
int nb_max,
|
||||
overlappable_nb_visitor_t fun,
|
||||
void *fun_ctxt) {
|
||||
if (!xd->left_available) return;
|
||||
|
||||
int nb_count = 0;
|
||||
|
||||
// prev_col_mi points into the mi array, starting at the top of the
|
||||
// previous column
|
||||
MODE_INFO **prev_col_mi = xd->mi - 1 - mi_row * xd->mi_stride;
|
||||
const int end_row = AOMMIN(mi_row + xd->n8_h, cm->mi_rows);
|
||||
uint8_t mi_step;
|
||||
for (int left_mi_row = mi_row; left_mi_row < end_row && nb_count < nb_max;
|
||||
left_mi_row += mi_step) {
|
||||
MODE_INFO **left_mi = prev_col_mi + left_mi_row * xd->mi_stride;
|
||||
mi_step = AOMMIN(mi_size_high[left_mi[0]->mbmi.sb_type],
|
||||
mi_size_high[BLOCK_64X64]);
|
||||
#if CONFIG_CHROMA_SUB8X8
|
||||
if (mi_step == 1) {
|
||||
left_mi_row &= ~1;
|
||||
left_mi = prev_col_mi + (left_mi_row + 1) * xd->mi_stride;
|
||||
mi_step = 2;
|
||||
}
|
||||
#endif // CONFIG_CHROMA_SUB8X8
|
||||
MB_MODE_INFO *left_mbmi = &left_mi[0]->mbmi;
|
||||
if (is_neighbor_overlappable(left_mbmi)) {
|
||||
++nb_count;
|
||||
fun(xd, left_mi_row - mi_row, AOMMIN(xd->n8_h, mi_step), *left_mi,
|
||||
fun_ctxt);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#endif // CONFIG_MOTION_VAR
|
||||
#endif // AV1_COMMON_OBMC_H_
|
||||
416
third_party/aom/av1/common/od_dering.c
vendored
416
third_party/aom/av1/common/od_dering.c
vendored
|
|
@ -1,416 +0,0 @@
|
|||
/*
|
||||
* Copyright (c) 2016, Alliance for Open Media. All rights reserved
|
||||
*
|
||||
* This source code is subject to the terms of the BSD 2 Clause License and
|
||||
* the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
|
||||
* was not distributed with this source code in the LICENSE file, you can
|
||||
* obtain it at www.aomedia.org/license/software. If the Alliance for Open
|
||||
* Media Patent License 1.0 was not distributed with this source code in the
|
||||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#include <math.h>
|
||||
#include <stdlib.h>
|
||||
|
||||
#ifdef HAVE_CONFIG_H
|
||||
#include "./config.h"
|
||||
#endif
|
||||
|
||||
#include "./aom_dsp_rtcd.h"
|
||||
#include "./av1_rtcd.h"
|
||||
#include "./cdef.h"
|
||||
|
||||
/* Generated from gen_filter_tables.c. */
|
||||
const int OD_DIRECTION_OFFSETS_TABLE[8][3] = {
|
||||
{ -1 * OD_FILT_BSTRIDE + 1, -2 * OD_FILT_BSTRIDE + 2,
|
||||
-3 * OD_FILT_BSTRIDE + 3 },
|
||||
{ 0 * OD_FILT_BSTRIDE + 1, -1 * OD_FILT_BSTRIDE + 2,
|
||||
-1 * OD_FILT_BSTRIDE + 3 },
|
||||
{ 0 * OD_FILT_BSTRIDE + 1, 0 * OD_FILT_BSTRIDE + 2, 0 * OD_FILT_BSTRIDE + 3 },
|
||||
{ 0 * OD_FILT_BSTRIDE + 1, 1 * OD_FILT_BSTRIDE + 2, 1 * OD_FILT_BSTRIDE + 3 },
|
||||
{ 1 * OD_FILT_BSTRIDE + 1, 2 * OD_FILT_BSTRIDE + 2, 3 * OD_FILT_BSTRIDE + 3 },
|
||||
{ 1 * OD_FILT_BSTRIDE + 0, 2 * OD_FILT_BSTRIDE + 1, 3 * OD_FILT_BSTRIDE + 1 },
|
||||
{ 1 * OD_FILT_BSTRIDE + 0, 2 * OD_FILT_BSTRIDE + 0, 3 * OD_FILT_BSTRIDE + 0 },
|
||||
{ 1 * OD_FILT_BSTRIDE + 0, 2 * OD_FILT_BSTRIDE - 1, 3 * OD_FILT_BSTRIDE - 1 },
|
||||
};
|
||||
|
||||
/* Detect direction. 0 means 45-degree up-right, 2 is horizontal, and so on.
|
||||
The search minimizes the weighted variance along all the lines in a
|
||||
particular direction, i.e. the squared error between the input and a
|
||||
"predicted" block where each pixel is replaced by the average along a line
|
||||
in a particular direction. Since each direction have the same sum(x^2) term,
|
||||
that term is never computed. See Section 2, step 2, of:
|
||||
http://jmvalin.ca/notes/intra_paint.pdf */
|
||||
int od_dir_find8_c(const uint16_t *img, int stride, int32_t *var,
|
||||
int coeff_shift) {
|
||||
int i;
|
||||
int32_t cost[8] = { 0 };
|
||||
int partial[8][15] = { { 0 } };
|
||||
int32_t best_cost = 0;
|
||||
int best_dir = 0;
|
||||
/* Instead of dividing by n between 2 and 8, we multiply by 3*5*7*8/n.
|
||||
The output is then 840 times larger, but we don't care for finding
|
||||
the max. */
|
||||
static const int div_table[] = { 0, 840, 420, 280, 210, 168, 140, 120, 105 };
|
||||
for (i = 0; i < 8; i++) {
|
||||
int j;
|
||||
for (j = 0; j < 8; j++) {
|
||||
int x;
|
||||
/* We subtract 128 here to reduce the maximum range of the squared
|
||||
partial sums. */
|
||||
x = (img[i * stride + j] >> coeff_shift) - 128;
|
||||
partial[0][i + j] += x;
|
||||
partial[1][i + j / 2] += x;
|
||||
partial[2][i] += x;
|
||||
partial[3][3 + i - j / 2] += x;
|
||||
partial[4][7 + i - j] += x;
|
||||
partial[5][3 - i / 2 + j] += x;
|
||||
partial[6][j] += x;
|
||||
partial[7][i / 2 + j] += x;
|
||||
}
|
||||
}
|
||||
for (i = 0; i < 8; i++) {
|
||||
cost[2] += partial[2][i] * partial[2][i];
|
||||
cost[6] += partial[6][i] * partial[6][i];
|
||||
}
|
||||
cost[2] *= div_table[8];
|
||||
cost[6] *= div_table[8];
|
||||
for (i = 0; i < 7; i++) {
|
||||
cost[0] += (partial[0][i] * partial[0][i] +
|
||||
partial[0][14 - i] * partial[0][14 - i]) *
|
||||
div_table[i + 1];
|
||||
cost[4] += (partial[4][i] * partial[4][i] +
|
||||
partial[4][14 - i] * partial[4][14 - i]) *
|
||||
div_table[i + 1];
|
||||
}
|
||||
cost[0] += partial[0][7] * partial[0][7] * div_table[8];
|
||||
cost[4] += partial[4][7] * partial[4][7] * div_table[8];
|
||||
for (i = 1; i < 8; i += 2) {
|
||||
int j;
|
||||
for (j = 0; j < 4 + 1; j++) {
|
||||
cost[i] += partial[i][3 + j] * partial[i][3 + j];
|
||||
}
|
||||
cost[i] *= div_table[8];
|
||||
for (j = 0; j < 4 - 1; j++) {
|
||||
cost[i] += (partial[i][j] * partial[i][j] +
|
||||
partial[i][10 - j] * partial[i][10 - j]) *
|
||||
div_table[2 * j + 2];
|
||||
}
|
||||
}
|
||||
for (i = 0; i < 8; i++) {
|
||||
if (cost[i] > best_cost) {
|
||||
best_cost = cost[i];
|
||||
best_dir = i;
|
||||
}
|
||||
}
|
||||
/* Difference between the optimal variance and the variance along the
|
||||
orthogonal direction. Again, the sum(x^2) terms cancel out. */
|
||||
*var = best_cost - cost[(best_dir + 4) & 7];
|
||||
/* We'd normally divide by 840, but dividing by 1024 is close enough
|
||||
for what we're going to do with this. */
|
||||
*var >>= 10;
|
||||
return best_dir;
|
||||
}
|
||||
|
||||
/* Smooth in the direction detected. */
|
||||
void od_filter_dering_direction_8x8_c(uint16_t *y, int ystride,
|
||||
const uint16_t *in, int threshold,
|
||||
int dir, int damping) {
|
||||
int i;
|
||||
int j;
|
||||
int k;
|
||||
static const int taps[3] = { 3, 2, 1 };
|
||||
for (i = 0; i < 8; i++) {
|
||||
for (j = 0; j < 8; j++) {
|
||||
int16_t sum;
|
||||
int16_t xx;
|
||||
int16_t yy;
|
||||
xx = in[i * OD_FILT_BSTRIDE + j];
|
||||
sum = 0;
|
||||
for (k = 0; k < 3; k++) {
|
||||
int16_t p0;
|
||||
int16_t p1;
|
||||
p0 = in[i * OD_FILT_BSTRIDE + j + OD_DIRECTION_OFFSETS_TABLE[dir][k]] -
|
||||
xx;
|
||||
p1 = in[i * OD_FILT_BSTRIDE + j - OD_DIRECTION_OFFSETS_TABLE[dir][k]] -
|
||||
xx;
|
||||
sum += taps[k] * constrain(p0, threshold, damping);
|
||||
sum += taps[k] * constrain(p1, threshold, damping);
|
||||
}
|
||||
sum = (sum + 8) >> 4;
|
||||
yy = xx + sum;
|
||||
y[i * ystride + j] = yy;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/* Smooth in the direction detected. */
|
||||
void od_filter_dering_direction_4x4_c(uint16_t *y, int ystride,
|
||||
const uint16_t *in, int threshold,
|
||||
int dir, int damping) {
|
||||
int i;
|
||||
int j;
|
||||
int k;
|
||||
static const int taps[2] = { 4, 1 };
|
||||
for (i = 0; i < 4; i++) {
|
||||
for (j = 0; j < 4; j++) {
|
||||
int16_t sum;
|
||||
int16_t xx;
|
||||
int16_t yy;
|
||||
xx = in[i * OD_FILT_BSTRIDE + j];
|
||||
sum = 0;
|
||||
for (k = 0; k < 2; k++) {
|
||||
int16_t p0;
|
||||
int16_t p1;
|
||||
p0 = in[i * OD_FILT_BSTRIDE + j + OD_DIRECTION_OFFSETS_TABLE[dir][k]] -
|
||||
xx;
|
||||
p1 = in[i * OD_FILT_BSTRIDE + j - OD_DIRECTION_OFFSETS_TABLE[dir][k]] -
|
||||
xx;
|
||||
sum += taps[k] * constrain(p0, threshold, damping);
|
||||
sum += taps[k] * constrain(p1, threshold, damping);
|
||||
}
|
||||
sum = (sum + 8) >> 4;
|
||||
yy = xx + sum;
|
||||
y[i * ystride + j] = yy;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/* Compute deringing filter threshold for an 8x8 block based on the
|
||||
directional variance difference. A high variance difference means that we
|
||||
have a highly directional pattern (e.g. a high contrast edge), so we can
|
||||
apply more deringing. A low variance means that we either have a low
|
||||
contrast edge, or a non-directional texture, so we want to be careful not
|
||||
to blur. */
|
||||
static INLINE int od_adjust_thresh(int threshold, int32_t var) {
|
||||
const int i = var >> 6 ? AOMMIN(get_msb(var >> 6), 12) : 0;
|
||||
/* We use the variance of 8x8 blocks to adjust the threshold. */
|
||||
return var ? (threshold * (4 + i) + 8) >> 4 : 0;
|
||||
}
|
||||
|
||||
void copy_8x8_16bit_to_16bit_c(uint16_t *dst, int dstride, const uint16_t *src,
|
||||
int sstride) {
|
||||
int i, j;
|
||||
for (i = 0; i < 8; i++)
|
||||
for (j = 0; j < 8; j++) dst[i * dstride + j] = src[i * sstride + j];
|
||||
}
|
||||
|
||||
void copy_4x4_16bit_to_16bit_c(uint16_t *dst, int dstride, const uint16_t *src,
|
||||
int sstride) {
|
||||
int i, j;
|
||||
for (i = 0; i < 4; i++)
|
||||
for (j = 0; j < 4; j++) dst[i * dstride + j] = src[i * sstride + j];
|
||||
}
|
||||
|
||||
static void copy_dering_16bit_to_16bit(uint16_t *dst, int dstride,
|
||||
uint16_t *src, dering_list *dlist,
|
||||
int dering_count, int bsize) {
|
||||
int bi, bx, by;
|
||||
|
||||
if (bsize == BLOCK_8X8) {
|
||||
for (bi = 0; bi < dering_count; bi++) {
|
||||
by = dlist[bi].by;
|
||||
bx = dlist[bi].bx;
|
||||
copy_8x8_16bit_to_16bit(&dst[(by << 3) * dstride + (bx << 3)], dstride,
|
||||
&src[bi << (3 + 3)], 8);
|
||||
}
|
||||
} else if (bsize == BLOCK_4X8) {
|
||||
for (bi = 0; bi < dering_count; bi++) {
|
||||
by = dlist[bi].by;
|
||||
bx = dlist[bi].bx;
|
||||
copy_4x4_16bit_to_16bit(&dst[(by << 3) * dstride + (bx << 2)], dstride,
|
||||
&src[bi << (3 + 2)], 4);
|
||||
copy_4x4_16bit_to_16bit(&dst[((by << 3) + 4) * dstride + (bx << 2)],
|
||||
dstride, &src[(bi << (3 + 2)) + 4 * 4], 4);
|
||||
}
|
||||
} else if (bsize == BLOCK_8X4) {
|
||||
for (bi = 0; bi < dering_count; bi++) {
|
||||
by = dlist[bi].by;
|
||||
bx = dlist[bi].bx;
|
||||
copy_4x4_16bit_to_16bit(&dst[(by << 2) * dstride + (bx << 3)], dstride,
|
||||
&src[bi << (2 + 3)], 8);
|
||||
copy_4x4_16bit_to_16bit(&dst[(by << 2) * dstride + (bx << 3) + 4],
|
||||
dstride, &src[(bi << (2 + 3)) + 4], 8);
|
||||
}
|
||||
} else {
|
||||
assert(bsize == BLOCK_4X4);
|
||||
for (bi = 0; bi < dering_count; bi++) {
|
||||
by = dlist[bi].by;
|
||||
bx = dlist[bi].bx;
|
||||
copy_4x4_16bit_to_16bit(&dst[(by << 2) * dstride + (bx << 2)], dstride,
|
||||
&src[bi << (2 + 2)], 4);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void copy_8x8_16bit_to_8bit_c(uint8_t *dst, int dstride, const uint16_t *src,
|
||||
int sstride) {
|
||||
int i, j;
|
||||
for (i = 0; i < 8; i++)
|
||||
for (j = 0; j < 8; j++)
|
||||
dst[i * dstride + j] = (uint8_t)src[i * sstride + j];
|
||||
}
|
||||
|
||||
void copy_4x4_16bit_to_8bit_c(uint8_t *dst, int dstride, const uint16_t *src,
|
||||
int sstride) {
|
||||
int i, j;
|
||||
for (i = 0; i < 4; i++)
|
||||
for (j = 0; j < 4; j++)
|
||||
dst[i * dstride + j] = (uint8_t)src[i * sstride + j];
|
||||
}
|
||||
|
||||
static void copy_dering_16bit_to_8bit(uint8_t *dst, int dstride,
|
||||
const uint16_t *src, dering_list *dlist,
|
||||
int dering_count, int bsize) {
|
||||
int bi, bx, by;
|
||||
if (bsize == BLOCK_8X8) {
|
||||
for (bi = 0; bi < dering_count; bi++) {
|
||||
by = dlist[bi].by;
|
||||
bx = dlist[bi].bx;
|
||||
copy_8x8_16bit_to_8bit(&dst[(by << 3) * dstride + (bx << 3)], dstride,
|
||||
&src[bi << (3 + 3)], 8);
|
||||
}
|
||||
} else if (bsize == BLOCK_4X8) {
|
||||
for (bi = 0; bi < dering_count; bi++) {
|
||||
by = dlist[bi].by;
|
||||
bx = dlist[bi].bx;
|
||||
copy_4x4_16bit_to_8bit(&dst[(by << 3) * dstride + (bx << 2)], dstride,
|
||||
&src[bi << (3 + 2)], 4);
|
||||
copy_4x4_16bit_to_8bit(&dst[((by << 3) + 4) * dstride + (bx << 2)],
|
||||
dstride, &src[(bi << (3 + 2)) + 4 * 4], 4);
|
||||
}
|
||||
} else if (bsize == BLOCK_8X4) {
|
||||
for (bi = 0; bi < dering_count; bi++) {
|
||||
by = dlist[bi].by;
|
||||
bx = dlist[bi].bx;
|
||||
copy_4x4_16bit_to_8bit(&dst[(by << 2) * dstride + (bx << 3)], dstride,
|
||||
&src[bi << (2 + 3)], 8);
|
||||
copy_4x4_16bit_to_8bit(&dst[(by << 2) * dstride + (bx << 3) + 4], dstride,
|
||||
&src[(bi << (2 + 3)) + 4], 8);
|
||||
}
|
||||
} else {
|
||||
assert(bsize == BLOCK_4X4);
|
||||
for (bi = 0; bi < dering_count; bi++) {
|
||||
by = dlist[bi].by;
|
||||
bx = dlist[bi].bx;
|
||||
copy_4x4_16bit_to_8bit(&dst[(by << 2) * dstride + (bx << 2)], dstride,
|
||||
&src[bi << (2 * 2)], 4);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
int get_filter_skip(int level) {
|
||||
int filter_skip = level & 1;
|
||||
if (level == 1) filter_skip = 0;
|
||||
return filter_skip;
|
||||
}
|
||||
|
||||
void od_dering(uint8_t *dst, int dstride, uint16_t *y, uint16_t *in, int xdec,
|
||||
int ydec, int dir[OD_DERING_NBLOCKS][OD_DERING_NBLOCKS],
|
||||
int *dirinit, int var[OD_DERING_NBLOCKS][OD_DERING_NBLOCKS],
|
||||
int pli, dering_list *dlist, int dering_count, int level,
|
||||
int clpf_strength, int clpf_damping, int dering_damping,
|
||||
int coeff_shift, int skip_dering, int hbd) {
|
||||
int bi;
|
||||
int bx;
|
||||
int by;
|
||||
int bsize, bsizex, bsizey;
|
||||
|
||||
int threshold = (level >> 1) << coeff_shift;
|
||||
int filter_skip = get_filter_skip(level);
|
||||
if (level == 1) threshold = 31 << coeff_shift;
|
||||
|
||||
od_filter_dering_direction_func filter_dering_direction[] = {
|
||||
od_filter_dering_direction_4x4, od_filter_dering_direction_8x8
|
||||
};
|
||||
clpf_damping += coeff_shift - (pli != AOM_PLANE_Y);
|
||||
dering_damping += coeff_shift - (pli != AOM_PLANE_Y);
|
||||
bsize =
|
||||
ydec ? (xdec ? BLOCK_4X4 : BLOCK_8X4) : (xdec ? BLOCK_4X8 : BLOCK_8X8);
|
||||
bsizex = 3 - xdec;
|
||||
bsizey = 3 - ydec;
|
||||
|
||||
if (!skip_dering) {
|
||||
if (pli == 0) {
|
||||
if (!dirinit || !*dirinit) {
|
||||
for (bi = 0; bi < dering_count; bi++) {
|
||||
by = dlist[bi].by;
|
||||
bx = dlist[bi].bx;
|
||||
dir[by][bx] =
|
||||
od_dir_find8(&in[8 * by * OD_FILT_BSTRIDE + 8 * bx],
|
||||
OD_FILT_BSTRIDE, &var[by][bx], coeff_shift);
|
||||
}
|
||||
if (dirinit) *dirinit = 1;
|
||||
}
|
||||
}
|
||||
// Only run dering for non-zero threshold (which is always the case for
|
||||
// 4:2:2 or 4:4:0). If we don't dering, we still need to eventually write
|
||||
// something out in y[] later.
|
||||
if (threshold != 0) {
|
||||
assert(bsize == BLOCK_8X8 || bsize == BLOCK_4X4);
|
||||
for (bi = 0; bi < dering_count; bi++) {
|
||||
int t = !filter_skip && dlist[bi].skip ? 0 : threshold;
|
||||
by = dlist[bi].by;
|
||||
bx = dlist[bi].bx;
|
||||
(filter_dering_direction[bsize == BLOCK_8X8])(
|
||||
&y[bi << (bsizex + bsizey)], 1 << bsizex,
|
||||
&in[(by * OD_FILT_BSTRIDE << bsizey) + (bx << bsizex)],
|
||||
pli ? t : od_adjust_thresh(t, var[by][bx]), dir[by][bx],
|
||||
dering_damping);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (clpf_strength) {
|
||||
if (threshold && !skip_dering)
|
||||
copy_dering_16bit_to_16bit(in, OD_FILT_BSTRIDE, y, dlist, dering_count,
|
||||
bsize);
|
||||
for (bi = 0; bi < dering_count; bi++) {
|
||||
by = dlist[bi].by;
|
||||
bx = dlist[bi].bx;
|
||||
int py = by << bsizey;
|
||||
int px = bx << bsizex;
|
||||
|
||||
if (!filter_skip && dlist[bi].skip) continue;
|
||||
if (!dst || hbd) {
|
||||
// 16 bit destination if high bitdepth or 8 bit destination not given
|
||||
(!threshold || (dir[by][bx] < 4 && dir[by][bx]) ? aom_clpf_block_hbd
|
||||
: aom_clpf_hblock_hbd)(
|
||||
dst ? (uint16_t *)dst + py * dstride + px
|
||||
: &y[bi << (bsizex + bsizey)],
|
||||
in + py * OD_FILT_BSTRIDE + px, dst && hbd ? dstride : 1 << bsizex,
|
||||
OD_FILT_BSTRIDE, 1 << bsizex, 1 << bsizey,
|
||||
clpf_strength << coeff_shift, clpf_damping);
|
||||
} else {
|
||||
// Do clpf and write the result to an 8 bit destination
|
||||
(!threshold || (dir[by][bx] < 4 && dir[by][bx]) ? aom_clpf_block
|
||||
: aom_clpf_hblock)(
|
||||
dst + py * dstride + px, in + py * OD_FILT_BSTRIDE + px, dstride,
|
||||
OD_FILT_BSTRIDE, 1 << bsizex, 1 << bsizey,
|
||||
clpf_strength << coeff_shift, clpf_damping);
|
||||
}
|
||||
}
|
||||
} else if (threshold != 0) {
|
||||
// No clpf, so copy instead
|
||||
if (hbd) {
|
||||
copy_dering_16bit_to_16bit((uint16_t *)dst, dstride, y, dlist,
|
||||
dering_count, bsize);
|
||||
} else {
|
||||
copy_dering_16bit_to_8bit(dst, dstride, y, dlist, dering_count, bsize);
|
||||
}
|
||||
} else if (dirinit) {
|
||||
// If we're here, both dering and clpf are off, and we still haven't written
|
||||
// anything to y[] yet, so we just copy the input to y[]. This is necessary
|
||||
// only for av1_cdef_search() and only av1_cdef_search() sets dirinit.
|
||||
for (bi = 0; bi < dering_count; bi++) {
|
||||
by = dlist[bi].by;
|
||||
bx = dlist[bi].bx;
|
||||
int iy, ix;
|
||||
// TODO(stemidts/jmvalin): SIMD optimisations
|
||||
for (iy = 0; iy < 1 << bsizey; iy++)
|
||||
for (ix = 0; ix < 1 << bsizex; ix++)
|
||||
y[(bi << (bsizex + bsizey)) + (iy << bsizex) + ix] =
|
||||
in[((by << bsizey) + iy) * OD_FILT_BSTRIDE + (bx << bsizex) + ix];
|
||||
}
|
||||
}
|
||||
}
|
||||
51
third_party/aom/av1/common/od_dering.h
vendored
51
third_party/aom/av1/common/od_dering.h
vendored
|
|
@ -1,51 +0,0 @@
|
|||
/*
|
||||
* Copyright (c) 2016, Alliance for Open Media. All rights reserved
|
||||
*
|
||||
* This source code is subject to the terms of the BSD 2 Clause License and
|
||||
* the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
|
||||
* was not distributed with this source code in the LICENSE file, you can
|
||||
* obtain it at www.aomedia.org/license/software. If the Alliance for Open
|
||||
* Media Patent License 1.0 was not distributed with this source code in the
|
||||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#if !defined(_dering_H)
|
||||
#define _dering_H (1)
|
||||
|
||||
#include "odintrin.h"
|
||||
|
||||
#define OD_DERING_NBLOCKS (MAX_SB_SIZE / 8)
|
||||
|
||||
/* We need to buffer three vertical lines. */
|
||||
#define OD_FILT_VBORDER (3)
|
||||
/* We only need to buffer three horizontal pixels too, but let's align to
|
||||
16 bytes (8 x 16 bits) to make vectorization easier. */
|
||||
#define OD_FILT_HBORDER (8)
|
||||
#define OD_FILT_BSTRIDE ALIGN_POWER_OF_TWO(MAX_SB_SIZE + 2 * OD_FILT_HBORDER, 3)
|
||||
|
||||
#define OD_DERING_VERY_LARGE (30000)
|
||||
#define OD_DERING_INBUF_SIZE \
|
||||
(OD_FILT_BSTRIDE * (MAX_SB_SIZE + 2 * OD_FILT_VBORDER))
|
||||
|
||||
extern const int OD_DIRECTION_OFFSETS_TABLE[8][3];
|
||||
|
||||
typedef struct {
|
||||
uint8_t by;
|
||||
uint8_t bx;
|
||||
uint8_t skip;
|
||||
} dering_list;
|
||||
|
||||
typedef void (*od_filter_dering_direction_func)(uint16_t *y, int ystride,
|
||||
const uint16_t *in,
|
||||
int threshold, int dir,
|
||||
int damping);
|
||||
|
||||
int get_filter_skip(int level);
|
||||
|
||||
void od_dering(uint8_t *dst, int dstride, uint16_t *y, uint16_t *in, int xdec,
|
||||
int ydec, int dir[OD_DERING_NBLOCKS][OD_DERING_NBLOCKS],
|
||||
int *dirinit, int var[OD_DERING_NBLOCKS][OD_DERING_NBLOCKS],
|
||||
int pli, dering_list *dlist, int dering_count, int level,
|
||||
int clpf_strength, int clpf_damping, int dering_damping,
|
||||
int coeff_shift, int skip_dering, int hbd);
|
||||
#endif
|
||||
390
third_party/aom/av1/common/od_dering_simd.h
vendored
390
third_party/aom/av1/common/od_dering_simd.h
vendored
|
|
@ -1,390 +0,0 @@
|
|||
/*
|
||||
* Copyright (c) 2016, Alliance for Open Media. All rights reserved
|
||||
*
|
||||
* This source code is subject to the terms of the BSD 2 Clause License and
|
||||
* the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
|
||||
* was not distributed with this source code in the LICENSE file, you can
|
||||
* obtain it at www.aomedia.org/license/software. If the Alliance for Open
|
||||
* Media Patent License 1.0 was not distributed with this source code in the
|
||||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#include "./av1_rtcd.h"
|
||||
#include "./cdef_simd.h"
|
||||
#include "./od_dering.h"
|
||||
|
||||
/* partial A is a 16-bit vector of the form:
|
||||
[x8 x7 x6 x5 x4 x3 x2 x1] and partial B has the form:
|
||||
[0 y1 y2 y3 y4 y5 y6 y7].
|
||||
This function computes (x1^2+y1^2)*C1 + (x2^2+y2^2)*C2 + ...
|
||||
(x7^2+y2^7)*C7 + (x8^2+0^2)*C8 where the C1..C8 constants are in const1
|
||||
and const2. */
|
||||
static INLINE v128 fold_mul_and_sum(v128 partiala, v128 partialb, v128 const1,
|
||||
v128 const2) {
|
||||
v128 tmp;
|
||||
/* Reverse partial B. */
|
||||
partialb = v128_shuffle_8(
|
||||
partialb, v128_from_32(0x0f0e0100, 0x03020504, 0x07060908, 0x0b0a0d0c));
|
||||
/* Interleave the x and y values of identical indices and pair x8 with 0. */
|
||||
tmp = partiala;
|
||||
partiala = v128_ziplo_16(partialb, partiala);
|
||||
partialb = v128_ziphi_16(partialb, tmp);
|
||||
/* Square and add the corresponding x and y values. */
|
||||
partiala = v128_madd_s16(partiala, partiala);
|
||||
partialb = v128_madd_s16(partialb, partialb);
|
||||
/* Multiply by constant. */
|
||||
partiala = v128_mullo_s32(partiala, const1);
|
||||
partialb = v128_mullo_s32(partialb, const2);
|
||||
/* Sum all results. */
|
||||
partiala = v128_add_32(partiala, partialb);
|
||||
return partiala;
|
||||
}
|
||||
|
||||
static INLINE v128 hsum4(v128 x0, v128 x1, v128 x2, v128 x3) {
|
||||
v128 t0, t1, t2, t3;
|
||||
t0 = v128_ziplo_32(x1, x0);
|
||||
t1 = v128_ziplo_32(x3, x2);
|
||||
t2 = v128_ziphi_32(x1, x0);
|
||||
t3 = v128_ziphi_32(x3, x2);
|
||||
x0 = v128_ziplo_64(t1, t0);
|
||||
x1 = v128_ziphi_64(t1, t0);
|
||||
x2 = v128_ziplo_64(t3, t2);
|
||||
x3 = v128_ziphi_64(t3, t2);
|
||||
return v128_add_32(v128_add_32(x0, x1), v128_add_32(x2, x3));
|
||||
}
|
||||
|
||||
/* Computes cost for directions 0, 5, 6 and 7. We can call this function again
|
||||
to compute the remaining directions. */
|
||||
static INLINE v128 compute_directions(v128 lines[8], int32_t tmp_cost1[4]) {
|
||||
v128 partial4a, partial4b, partial5a, partial5b, partial7a, partial7b;
|
||||
v128 partial6;
|
||||
v128 tmp;
|
||||
/* Partial sums for lines 0 and 1. */
|
||||
partial4a = v128_shl_n_byte(lines[0], 14);
|
||||
partial4b = v128_shr_n_byte(lines[0], 2);
|
||||
partial4a = v128_add_16(partial4a, v128_shl_n_byte(lines[1], 12));
|
||||
partial4b = v128_add_16(partial4b, v128_shr_n_byte(lines[1], 4));
|
||||
tmp = v128_add_16(lines[0], lines[1]);
|
||||
partial5a = v128_shl_n_byte(tmp, 10);
|
||||
partial5b = v128_shr_n_byte(tmp, 6);
|
||||
partial7a = v128_shl_n_byte(tmp, 4);
|
||||
partial7b = v128_shr_n_byte(tmp, 12);
|
||||
partial6 = tmp;
|
||||
|
||||
/* Partial sums for lines 2 and 3. */
|
||||
partial4a = v128_add_16(partial4a, v128_shl_n_byte(lines[2], 10));
|
||||
partial4b = v128_add_16(partial4b, v128_shr_n_byte(lines[2], 6));
|
||||
partial4a = v128_add_16(partial4a, v128_shl_n_byte(lines[3], 8));
|
||||
partial4b = v128_add_16(partial4b, v128_shr_n_byte(lines[3], 8));
|
||||
tmp = v128_add_16(lines[2], lines[3]);
|
||||
partial5a = v128_add_16(partial5a, v128_shl_n_byte(tmp, 8));
|
||||
partial5b = v128_add_16(partial5b, v128_shr_n_byte(tmp, 8));
|
||||
partial7a = v128_add_16(partial7a, v128_shl_n_byte(tmp, 6));
|
||||
partial7b = v128_add_16(partial7b, v128_shr_n_byte(tmp, 10));
|
||||
partial6 = v128_add_16(partial6, tmp);
|
||||
|
||||
/* Partial sums for lines 4 and 5. */
|
||||
partial4a = v128_add_16(partial4a, v128_shl_n_byte(lines[4], 6));
|
||||
partial4b = v128_add_16(partial4b, v128_shr_n_byte(lines[4], 10));
|
||||
partial4a = v128_add_16(partial4a, v128_shl_n_byte(lines[5], 4));
|
||||
partial4b = v128_add_16(partial4b, v128_shr_n_byte(lines[5], 12));
|
||||
tmp = v128_add_16(lines[4], lines[5]);
|
||||
partial5a = v128_add_16(partial5a, v128_shl_n_byte(tmp, 6));
|
||||
partial5b = v128_add_16(partial5b, v128_shr_n_byte(tmp, 10));
|
||||
partial7a = v128_add_16(partial7a, v128_shl_n_byte(tmp, 8));
|
||||
partial7b = v128_add_16(partial7b, v128_shr_n_byte(tmp, 8));
|
||||
partial6 = v128_add_16(partial6, tmp);
|
||||
|
||||
/* Partial sums for lines 6 and 7. */
|
||||
partial4a = v128_add_16(partial4a, v128_shl_n_byte(lines[6], 2));
|
||||
partial4b = v128_add_16(partial4b, v128_shr_n_byte(lines[6], 14));
|
||||
partial4a = v128_add_16(partial4a, lines[7]);
|
||||
tmp = v128_add_16(lines[6], lines[7]);
|
||||
partial5a = v128_add_16(partial5a, v128_shl_n_byte(tmp, 4));
|
||||
partial5b = v128_add_16(partial5b, v128_shr_n_byte(tmp, 12));
|
||||
partial7a = v128_add_16(partial7a, v128_shl_n_byte(tmp, 10));
|
||||
partial7b = v128_add_16(partial7b, v128_shr_n_byte(tmp, 6));
|
||||
partial6 = v128_add_16(partial6, tmp);
|
||||
|
||||
/* Compute costs in terms of partial sums. */
|
||||
partial4a =
|
||||
fold_mul_and_sum(partial4a, partial4b, v128_from_32(210, 280, 420, 840),
|
||||
v128_from_32(105, 120, 140, 168));
|
||||
partial7a =
|
||||
fold_mul_and_sum(partial7a, partial7b, v128_from_32(210, 420, 0, 0),
|
||||
v128_from_32(105, 105, 105, 140));
|
||||
partial5a =
|
||||
fold_mul_and_sum(partial5a, partial5b, v128_from_32(210, 420, 0, 0),
|
||||
v128_from_32(105, 105, 105, 140));
|
||||
partial6 = v128_madd_s16(partial6, partial6);
|
||||
partial6 = v128_mullo_s32(partial6, v128_dup_32(105));
|
||||
|
||||
partial4a = hsum4(partial4a, partial5a, partial6, partial7a);
|
||||
v128_store_unaligned(tmp_cost1, partial4a);
|
||||
return partial4a;
|
||||
}
|
||||
|
||||
/* transpose and reverse the order of the lines -- equivalent to a 90-degree
|
||||
counter-clockwise rotation of the pixels. */
|
||||
static INLINE void array_reverse_transpose_8x8(v128 *in, v128 *res) {
|
||||
const v128 tr0_0 = v128_ziplo_16(in[1], in[0]);
|
||||
const v128 tr0_1 = v128_ziplo_16(in[3], in[2]);
|
||||
const v128 tr0_2 = v128_ziphi_16(in[1], in[0]);
|
||||
const v128 tr0_3 = v128_ziphi_16(in[3], in[2]);
|
||||
const v128 tr0_4 = v128_ziplo_16(in[5], in[4]);
|
||||
const v128 tr0_5 = v128_ziplo_16(in[7], in[6]);
|
||||
const v128 tr0_6 = v128_ziphi_16(in[5], in[4]);
|
||||
const v128 tr0_7 = v128_ziphi_16(in[7], in[6]);
|
||||
|
||||
const v128 tr1_0 = v128_ziplo_32(tr0_1, tr0_0);
|
||||
const v128 tr1_1 = v128_ziplo_32(tr0_5, tr0_4);
|
||||
const v128 tr1_2 = v128_ziphi_32(tr0_1, tr0_0);
|
||||
const v128 tr1_3 = v128_ziphi_32(tr0_5, tr0_4);
|
||||
const v128 tr1_4 = v128_ziplo_32(tr0_3, tr0_2);
|
||||
const v128 tr1_5 = v128_ziplo_32(tr0_7, tr0_6);
|
||||
const v128 tr1_6 = v128_ziphi_32(tr0_3, tr0_2);
|
||||
const v128 tr1_7 = v128_ziphi_32(tr0_7, tr0_6);
|
||||
|
||||
res[7] = v128_ziplo_64(tr1_1, tr1_0);
|
||||
res[6] = v128_ziphi_64(tr1_1, tr1_0);
|
||||
res[5] = v128_ziplo_64(tr1_3, tr1_2);
|
||||
res[4] = v128_ziphi_64(tr1_3, tr1_2);
|
||||
res[3] = v128_ziplo_64(tr1_5, tr1_4);
|
||||
res[2] = v128_ziphi_64(tr1_5, tr1_4);
|
||||
res[1] = v128_ziplo_64(tr1_7, tr1_6);
|
||||
res[0] = v128_ziphi_64(tr1_7, tr1_6);
|
||||
}
|
||||
|
||||
int SIMD_FUNC(od_dir_find8)(const od_dering_in *img, int stride, int32_t *var,
|
||||
int coeff_shift) {
|
||||
int i;
|
||||
int32_t cost[8];
|
||||
int32_t best_cost = 0;
|
||||
int best_dir = 0;
|
||||
v128 lines[8];
|
||||
for (i = 0; i < 8; i++) {
|
||||
lines[i] = v128_load_unaligned(&img[i * stride]);
|
||||
lines[i] =
|
||||
v128_sub_16(v128_shr_s16(lines[i], coeff_shift), v128_dup_16(128));
|
||||
}
|
||||
|
||||
#if defined(__SSE4_1__)
|
||||
/* Compute "mostly vertical" directions. */
|
||||
__m128i dir47 = compute_directions(lines, cost + 4);
|
||||
|
||||
array_reverse_transpose_8x8(lines, lines);
|
||||
|
||||
/* Compute "mostly horizontal" directions. */
|
||||
__m128i dir03 = compute_directions(lines, cost);
|
||||
|
||||
__m128i max = _mm_max_epi32(dir03, dir47);
|
||||
max = _mm_max_epi32(max, _mm_shuffle_epi32(max, _MM_SHUFFLE(1, 0, 3, 2)));
|
||||
max = _mm_max_epi32(max, _mm_shuffle_epi32(max, _MM_SHUFFLE(2, 3, 0, 1)));
|
||||
best_cost = _mm_cvtsi128_si32(max);
|
||||
__m128i t =
|
||||
_mm_packs_epi32(_mm_cmpeq_epi32(max, dir03), _mm_cmpeq_epi32(max, dir47));
|
||||
best_dir = _mm_movemask_epi8(_mm_packs_epi16(t, t));
|
||||
best_dir = get_msb(best_dir ^ (best_dir - 1)); // Count trailing zeros
|
||||
#else
|
||||
/* Compute "mostly vertical" directions. */
|
||||
compute_directions(lines, cost + 4);
|
||||
|
||||
array_reverse_transpose_8x8(lines, lines);
|
||||
|
||||
/* Compute "mostly horizontal" directions. */
|
||||
compute_directions(lines, cost);
|
||||
|
||||
for (i = 0; i < 8; i++) {
|
||||
if (cost[i] > best_cost) {
|
||||
best_cost = cost[i];
|
||||
best_dir = i;
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
/* Difference between the optimal variance and the variance along the
|
||||
orthogonal direction. Again, the sum(x^2) terms cancel out. */
|
||||
*var = best_cost - cost[(best_dir + 4) & 7];
|
||||
/* We'd normally divide by 840, but dividing by 1024 is close enough
|
||||
for what we're going to do with this. */
|
||||
*var >>= 10;
|
||||
return best_dir;
|
||||
}
|
||||
|
||||
void SIMD_FUNC(od_filter_dering_direction_4x4)(uint16_t *y, int ystride,
|
||||
const uint16_t *in,
|
||||
int threshold, int dir,
|
||||
int damping) {
|
||||
int i;
|
||||
v128 p0, p1, sum, row, res;
|
||||
int o1 = OD_DIRECTION_OFFSETS_TABLE[dir][0];
|
||||
int o2 = OD_DIRECTION_OFFSETS_TABLE[dir][1];
|
||||
|
||||
if (threshold) damping -= get_msb(threshold);
|
||||
for (i = 0; i < 4; i += 2) {
|
||||
sum = v128_zero();
|
||||
row = v128_from_v64(v64_load_aligned(&in[i * OD_FILT_BSTRIDE]),
|
||||
v64_load_aligned(&in[(i + 1) * OD_FILT_BSTRIDE]));
|
||||
|
||||
// p0 = constrain16(in[i*OD_FILT_BSTRIDE + offset], row, threshold, damping)
|
||||
p0 = v128_from_v64(v64_load_unaligned(&in[i * OD_FILT_BSTRIDE + o1]),
|
||||
v64_load_unaligned(&in[(i + 1) * OD_FILT_BSTRIDE + o1]));
|
||||
p0 = constrain16(p0, row, threshold, damping);
|
||||
|
||||
// p1 = constrain16(in[i*OD_FILT_BSTRIDE - offset], row, threshold, damping)
|
||||
p1 = v128_from_v64(v64_load_unaligned(&in[i * OD_FILT_BSTRIDE - o1]),
|
||||
v64_load_unaligned(&in[(i + 1) * OD_FILT_BSTRIDE - o1]));
|
||||
p1 = constrain16(p1, row, threshold, damping);
|
||||
|
||||
// sum += 4 * (p0 + p1)
|
||||
sum = v128_add_16(sum, v128_shl_n_16(v128_add_16(p0, p1), 2));
|
||||
|
||||
// p0 = constrain16(in[i*OD_FILT_BSTRIDE + offset], row, threshold, damping)
|
||||
p0 = v128_from_v64(v64_load_unaligned(&in[i * OD_FILT_BSTRIDE + o2]),
|
||||
v64_load_unaligned(&in[(i + 1) * OD_FILT_BSTRIDE + o2]));
|
||||
p0 = constrain16(p0, row, threshold, damping);
|
||||
|
||||
// p1 = constrain16(in[i*OD_FILT_BSTRIDE - offset], row, threshold, damping)
|
||||
p1 = v128_from_v64(v64_load_unaligned(&in[i * OD_FILT_BSTRIDE - o2]),
|
||||
v64_load_unaligned(&in[(i + 1) * OD_FILT_BSTRIDE - o2]));
|
||||
p1 = constrain16(p1, row, threshold, damping);
|
||||
|
||||
// sum += 1 * (p0 + p1)
|
||||
sum = v128_add_16(sum, v128_add_16(p0, p1));
|
||||
|
||||
// res = row + ((sum + 8) >> 4)
|
||||
res = v128_add_16(sum, v128_dup_16(8));
|
||||
res = v128_shr_n_s16(res, 4);
|
||||
res = v128_add_16(row, res);
|
||||
v64_store_aligned(&y[i * ystride], v128_high_v64(res));
|
||||
v64_store_aligned(&y[(i + 1) * ystride], v128_low_v64(res));
|
||||
}
|
||||
}
|
||||
|
||||
void SIMD_FUNC(od_filter_dering_direction_8x8)(uint16_t *y, int ystride,
|
||||
const uint16_t *in,
|
||||
int threshold, int dir,
|
||||
int damping) {
|
||||
int i;
|
||||
v128 sum, p0, p1, row, res;
|
||||
int o1 = OD_DIRECTION_OFFSETS_TABLE[dir][0];
|
||||
int o2 = OD_DIRECTION_OFFSETS_TABLE[dir][1];
|
||||
int o3 = OD_DIRECTION_OFFSETS_TABLE[dir][2];
|
||||
|
||||
if (threshold) damping -= get_msb(threshold);
|
||||
for (i = 0; i < 8; i++) {
|
||||
sum = v128_zero();
|
||||
row = v128_load_aligned(&in[i * OD_FILT_BSTRIDE]);
|
||||
|
||||
// p0 = constrain16(in[i*OD_FILT_BSTRIDE + offset], row, threshold, damping)
|
||||
p0 = v128_load_unaligned(&in[i * OD_FILT_BSTRIDE + o1]);
|
||||
p0 = constrain16(p0, row, threshold, damping);
|
||||
|
||||
// p1 = constrain16(in[i*OD_FILT_BSTRIDE - offset], row, threshold, damping)
|
||||
p1 = v128_load_unaligned(&in[i * OD_FILT_BSTRIDE - o1]);
|
||||
p1 = constrain16(p1, row, threshold, damping);
|
||||
|
||||
// sum += 3 * (p0 + p1)
|
||||
p0 = v128_add_16(p0, p1);
|
||||
p0 = v128_add_16(p0, v128_shl_n_16(p0, 1));
|
||||
sum = v128_add_16(sum, p0);
|
||||
|
||||
// p0 = constrain16(in[i*OD_FILT_BSTRIDE + offset], row, threshold, damping)
|
||||
p0 = v128_load_unaligned(&in[i * OD_FILT_BSTRIDE + o2]);
|
||||
p0 = constrain16(p0, row, threshold, damping);
|
||||
|
||||
// p1 = constrain16(in[i*OD_FILT_BSTRIDE - offset], row, threshold, damping)
|
||||
p1 = v128_load_unaligned(&in[i * OD_FILT_BSTRIDE - o2]);
|
||||
p1 = constrain16(p1, row, threshold, damping);
|
||||
|
||||
// sum += 2 * (p0 + p1)
|
||||
p0 = v128_shl_n_16(v128_add_16(p0, p1), 1);
|
||||
sum = v128_add_16(sum, p0);
|
||||
|
||||
// p0 = constrain16(in[i*OD_FILT_BSTRIDE + offset], row, threshold, damping)
|
||||
p0 = v128_load_unaligned(&in[i * OD_FILT_BSTRIDE + o3]);
|
||||
p0 = constrain16(p0, row, threshold, damping);
|
||||
|
||||
// p1 = constrain16(in[i*OD_FILT_BSTRIDE - offset], row, threshold, damping)
|
||||
p1 = v128_load_unaligned(&in[i * OD_FILT_BSTRIDE - o3]);
|
||||
p1 = constrain16(p1, row, threshold, damping);
|
||||
|
||||
// sum += (p0 + p1)
|
||||
p0 = v128_add_16(p0, p1);
|
||||
sum = v128_add_16(sum, p0);
|
||||
|
||||
// res = row + ((sum + 8) >> 4)
|
||||
res = v128_add_16(sum, v128_dup_16(8));
|
||||
res = v128_shr_n_s16(res, 4);
|
||||
res = v128_add_16(row, res);
|
||||
v128_store_unaligned(&y[i * ystride], res);
|
||||
}
|
||||
}
|
||||
|
||||
void SIMD_FUNC(copy_8x8_16bit_to_8bit)(uint8_t *dst, int dstride,
|
||||
const uint16_t *src, int sstride) {
|
||||
int i;
|
||||
for (i = 0; i < 8; i++) {
|
||||
v128 row = v128_load_unaligned(&src[i * sstride]);
|
||||
row = v128_pack_s16_u8(row, row);
|
||||
v64_store_unaligned(&dst[i * dstride], v128_low_v64(row));
|
||||
}
|
||||
}
|
||||
|
||||
void SIMD_FUNC(copy_4x4_16bit_to_8bit)(uint8_t *dst, int dstride,
|
||||
const uint16_t *src, int sstride) {
|
||||
int i;
|
||||
for (i = 0; i < 4; i++) {
|
||||
v128 row = v128_load_unaligned(&src[i * sstride]);
|
||||
row = v128_pack_s16_u8(row, row);
|
||||
u32_store_unaligned(&dst[i * dstride], v128_low_u32(row));
|
||||
}
|
||||
}
|
||||
|
||||
void SIMD_FUNC(copy_8x8_16bit_to_16bit)(uint16_t *dst, int dstride,
|
||||
const uint16_t *src, int sstride) {
|
||||
int i;
|
||||
for (i = 0; i < 8; i++) {
|
||||
v128 row = v128_load_unaligned(&src[i * sstride]);
|
||||
v128_store_unaligned(&dst[i * dstride], row);
|
||||
}
|
||||
}
|
||||
|
||||
void SIMD_FUNC(copy_4x4_16bit_to_16bit)(uint16_t *dst, int dstride,
|
||||
const uint16_t *src, int sstride) {
|
||||
int i;
|
||||
for (i = 0; i < 4; i++) {
|
||||
v64 row = v64_load_unaligned(&src[i * sstride]);
|
||||
v64_store_unaligned(&dst[i * dstride], row);
|
||||
}
|
||||
}
|
||||
|
||||
void SIMD_FUNC(copy_rect8_8bit_to_16bit)(uint16_t *dst, int dstride,
|
||||
const uint8_t *src, int sstride, int v,
|
||||
int h) {
|
||||
int i, j;
|
||||
for (i = 0; i < v; i++) {
|
||||
for (j = 0; j < (h & ~0x7); j += 8) {
|
||||
v64 row = v64_load_unaligned(&src[i * sstride + j]);
|
||||
v128_store_unaligned(&dst[i * dstride + j], v128_unpack_u8_s16(row));
|
||||
}
|
||||
for (; j < h; j++) {
|
||||
dst[i * dstride + j] = src[i * sstride + j];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void SIMD_FUNC(copy_rect8_16bit_to_16bit)(uint16_t *dst, int dstride,
|
||||
const uint16_t *src, int sstride,
|
||||
int v, int h) {
|
||||
int i, j;
|
||||
for (i = 0; i < v; i++) {
|
||||
for (j = 0; j < (h & ~0x7); j += 8) {
|
||||
v128 row = v128_load_unaligned(&src[i * sstride + j]);
|
||||
v128_store_unaligned(&dst[i * dstride + j], row);
|
||||
}
|
||||
for (; j < h; j++) {
|
||||
dst[i * dstride + j] = src[i * sstride + j];
|
||||
}
|
||||
}
|
||||
}
|
||||
467
third_party/aom/av1/common/onyxc_int.h
vendored
467
third_party/aom/av1/common/onyxc_int.h
vendored
|
|
@ -38,6 +38,10 @@
|
|||
#if CONFIG_CFL
|
||||
#include "av1/common/cfl.h"
|
||||
#endif
|
||||
#if CONFIG_HASH_ME
|
||||
// TODO(youzhou@microsoft.com): Encoder only. Move it out of common
|
||||
#include "av1/encoder/hash_motion.h"
|
||||
#endif
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
|
|
@ -60,7 +64,13 @@ extern "C" {
|
|||
#define FRAME_ID_NUMBERS_PRESENT_FLAG 1
|
||||
#define FRAME_ID_LENGTH_MINUS7 8 // Allows frame id up to 2^15-1
|
||||
#define DELTA_FRAME_ID_LENGTH_MINUS2 12 // Allows frame id deltas up to 2^14-1
|
||||
#endif
|
||||
#endif // CONFIG_REFERENCE_BUFFER
|
||||
|
||||
#if CONFIG_NO_FRAME_CONTEXT_SIGNALING
|
||||
#define FRAME_CONTEXTS (FRAME_BUFFERS + 1)
|
||||
// Extra frame context which is always kept at default values
|
||||
#define FRAME_CONTEXT_DEFAULTS (FRAME_CONTEXTS - 1)
|
||||
#else
|
||||
|
||||
#if CONFIG_EXT_REFS
|
||||
#define FRAME_CONTEXTS_LOG2 3
|
||||
|
|
@ -69,6 +79,7 @@ extern "C" {
|
|||
#endif
|
||||
|
||||
#define FRAME_CONTEXTS (1 << FRAME_CONTEXTS_LOG2)
|
||||
#endif // CONFIG_NO_FRAME_CONTEXT_SIGNALING
|
||||
|
||||
#define NUM_PING_PONG_BUFFERS 2
|
||||
|
||||
|
|
@ -79,11 +90,13 @@ typedef enum {
|
|||
REFERENCE_MODES = 3,
|
||||
} REFERENCE_MODE;
|
||||
|
||||
#if !CONFIG_NO_FRAME_CONTEXT_SIGNALING
|
||||
typedef enum {
|
||||
RESET_FRAME_CONTEXT_NONE = 0,
|
||||
RESET_FRAME_CONTEXT_CURRENT = 1,
|
||||
RESET_FRAME_CONTEXT_ALL = 2,
|
||||
} RESET_FRAME_CONTEXT_MODE;
|
||||
#endif
|
||||
|
||||
typedef enum {
|
||||
/**
|
||||
|
|
@ -98,6 +111,14 @@ typedef enum {
|
|||
REFRESH_FRAME_CONTEXT_BACKWARD,
|
||||
} REFRESH_FRAME_CONTEXT_MODE;
|
||||
|
||||
#if CONFIG_MFMV
|
||||
#define MFMV_STACK_SIZE INTER_REFS_PER_FRAME
|
||||
|
||||
typedef struct {
|
||||
int_mv mfmv[INTER_REFS_PER_FRAME][MFMV_STACK_SIZE];
|
||||
} TPL_MV_REF;
|
||||
#endif
|
||||
|
||||
typedef struct {
|
||||
int_mv mv[2];
|
||||
int_mv pred_mv[2];
|
||||
|
|
@ -106,14 +127,38 @@ typedef struct {
|
|||
|
||||
typedef struct {
|
||||
int ref_count;
|
||||
|
||||
#if CONFIG_FRAME_MARKER
|
||||
int cur_frame_offset;
|
||||
int lst_frame_offset;
|
||||
int alt_frame_offset;
|
||||
int gld_frame_offset;
|
||||
#if CONFIG_EXT_REFS
|
||||
int lst2_frame_offset;
|
||||
int lst3_frame_offset;
|
||||
int bwd_frame_offset;
|
||||
int alt2_frame_offset;
|
||||
#endif
|
||||
#endif // CONFIG_FRAME_MARKER
|
||||
|
||||
#if CONFIG_MFMV
|
||||
TPL_MV_REF *tpl_mvs;
|
||||
#endif
|
||||
MV_REF *mvs;
|
||||
int mi_rows;
|
||||
int mi_cols;
|
||||
// Width and height give the size of the buffer (before any upscaling, unlike
|
||||
// the sizes that can be derived from the buf structure)
|
||||
int width;
|
||||
int height;
|
||||
#if CONFIG_GLOBAL_MOTION
|
||||
WarpedMotionParams global_motion[TOTAL_REFS_PER_FRAME];
|
||||
#endif // CONFIG_GLOBAL_MOTION
|
||||
aom_codec_frame_buffer_t raw_frame_buffer;
|
||||
YV12_BUFFER_CONFIG buf;
|
||||
#if CONFIG_HASH_ME
|
||||
hash_table hash_table;
|
||||
#endif
|
||||
#if CONFIG_TEMPMV_SIGNALING
|
||||
uint8_t intra_only;
|
||||
#endif
|
||||
|
|
@ -150,13 +195,29 @@ typedef struct BufferPool {
|
|||
InternalFrameBufferList int_frame_buffers;
|
||||
} BufferPool;
|
||||
|
||||
#if CONFIG_LV_MAP
|
||||
typedef struct {
|
||||
int base_ctx_table[2 /*row*/][2 /*col*/][2 /*sig_map*/]
|
||||
[BASE_CONTEXT_POSITION_NUM + 1];
|
||||
} LV_MAP_CTX_TABLE;
|
||||
typedef int BASE_CTX_TABLE[2 /*col*/][2 /*sig_map*/]
|
||||
[BASE_CONTEXT_POSITION_NUM + 1];
|
||||
#endif
|
||||
|
||||
#if CONFIG_REFERENCE_BUFFER
|
||||
/* Initial version of sequence header structure */
|
||||
typedef struct SequenceHeader {
|
||||
int frame_id_numbers_present_flag;
|
||||
int frame_id_length_minus7;
|
||||
int delta_frame_id_length_minus2;
|
||||
} SequenceHeader;
|
||||
#endif // CONFIG_REFERENCE_BUFFER
|
||||
|
||||
typedef struct AV1Common {
|
||||
struct aom_internal_error_info error;
|
||||
aom_color_space_t color_space;
|
||||
#if CONFIG_COLORSPACE_HEADERS
|
||||
aom_transfer_function_t transfer_function;
|
||||
aom_chroma_sample_position_t chroma_sample_position;
|
||||
#endif
|
||||
int color_range;
|
||||
int width;
|
||||
int height;
|
||||
|
|
@ -211,21 +272,24 @@ typedef struct AV1Common {
|
|||
uint8_t last_intra_only;
|
||||
|
||||
int allow_high_precision_mv;
|
||||
#if CONFIG_AMVR
|
||||
int seq_mv_precision_level; // 0 the default in AOM, 1 only integer, 2
|
||||
// adaptive
|
||||
int cur_frame_mv_precision_level; // 0 the default in AOM, 1 only integer
|
||||
#endif
|
||||
|
||||
#if CONFIG_PALETTE || CONFIG_INTRABC
|
||||
int allow_screen_content_tools;
|
||||
#endif // CONFIG_PALETTE || CONFIG_INTRABC
|
||||
#if CONFIG_EXT_INTER
|
||||
#if CONFIG_INTERINTRA
|
||||
int allow_interintra_compound;
|
||||
#endif // CONFIG_INTERINTRA
|
||||
#if CONFIG_WEDGE || CONFIG_COMPOUND_SEGMENT
|
||||
int allow_masked_compound;
|
||||
#endif // CONFIG_WEDGE || CONFIG_COMPOUND_SEGMENT
|
||||
#endif // CONFIG_EXT_INTER
|
||||
|
||||
#if !CONFIG_NO_FRAME_CONTEXT_SIGNALING
|
||||
// Flag signaling which frame contexts should be reset to default values.
|
||||
RESET_FRAME_CONTEXT_MODE reset_frame_context;
|
||||
#endif
|
||||
|
||||
// MBs, mb_rows/cols is in 16-pixel units; mi_rows/cols is in
|
||||
// MODE_INFO (8-pixel) units.
|
||||
|
|
@ -304,9 +368,8 @@ typedef struct AV1Common {
|
|||
|
||||
loop_filter_info_n lf_info;
|
||||
#if CONFIG_FRAME_SUPERRES
|
||||
// The numerator of the superres scale; the denominator is fixed.
|
||||
uint8_t superres_scale_numerator;
|
||||
uint8_t superres_kf_scale_numerator;
|
||||
// The denominator of the superres scale; the numerator is fixed.
|
||||
uint8_t superres_scale_denominator;
|
||||
int superres_upscaled_width;
|
||||
int superres_upscaled_height;
|
||||
#endif // CONFIG_FRAME_SUPERRES
|
||||
|
|
@ -343,9 +406,15 @@ typedef struct AV1Common {
|
|||
FRAME_CONTEXT *fc; /* this frame entropy */
|
||||
FRAME_CONTEXT *frame_contexts; // FRAME_CONTEXTS
|
||||
FRAME_CONTEXT *pre_fc; // Context referenced in this frame
|
||||
#if !CONFIG_NO_FRAME_CONTEXT_SIGNALING
|
||||
unsigned int frame_context_idx; /* Context to use/update */
|
||||
#endif
|
||||
FRAME_COUNTS counts;
|
||||
|
||||
#if CONFIG_FRAME_MARKER
|
||||
unsigned int frame_offset;
|
||||
#endif
|
||||
|
||||
unsigned int current_video_frame;
|
||||
BITSTREAM_PROFILE profile;
|
||||
|
||||
|
|
@ -355,9 +424,30 @@ typedef struct AV1Common {
|
|||
|
||||
int error_resilient_mode;
|
||||
|
||||
int log2_tile_cols, log2_tile_rows; // Used in non-large_scale_tile_coding.
|
||||
int tile_cols, tile_rows;
|
||||
int tile_width, tile_height; // In MI units
|
||||
int last_tile_cols, last_tile_rows;
|
||||
|
||||
#if CONFIG_MAX_TILE
|
||||
int min_log2_tile_cols;
|
||||
int max_log2_tile_cols;
|
||||
int max_log2_tile_rows;
|
||||
int min_log2_tile_rows;
|
||||
int min_log2_tiles;
|
||||
int max_tile_width_sb;
|
||||
int max_tile_height_sb;
|
||||
int uniform_tile_spacing_flag;
|
||||
int log2_tile_cols; // only valid for uniform tiles
|
||||
int log2_tile_rows; // only valid for uniform tiles
|
||||
int tile_col_start_sb[MAX_TILE_COLS + 1]; // valid for 0 <= i <= tile_cols
|
||||
int tile_row_start_sb[MAX_TILE_ROWS + 1]; // valid for 0 <= i <= tile_rows
|
||||
#if CONFIG_DEPENDENT_HORZTILES
|
||||
int tile_row_independent[MAX_TILE_ROWS]; // valid for 0 <= i < tile_rows
|
||||
#endif
|
||||
#else
|
||||
int log2_tile_cols, log2_tile_rows; // Used in non-large_scale_tile_coding.
|
||||
int tile_width, tile_height; // In MI units
|
||||
#endif // CONFIG_MAX_TILE
|
||||
|
||||
#if CONFIG_EXT_TILE
|
||||
unsigned int large_scale_tile;
|
||||
unsigned int single_tile_decoding;
|
||||
|
|
@ -407,15 +497,14 @@ typedef struct AV1Common {
|
|||
int mib_size; // Size of the superblock in units of MI blocks
|
||||
int mib_size_log2; // Log 2 of above.
|
||||
#if CONFIG_CDEF
|
||||
int cdef_dering_damping;
|
||||
int cdef_clpf_damping;
|
||||
int cdef_pri_damping;
|
||||
int cdef_sec_damping;
|
||||
int nb_cdef_strengths;
|
||||
int cdef_strengths[CDEF_MAX_STRENGTHS];
|
||||
int cdef_uv_strengths[CDEF_MAX_STRENGTHS];
|
||||
int cdef_bits;
|
||||
#endif
|
||||
|
||||
#if CONFIG_DELTA_Q
|
||||
int delta_q_present_flag;
|
||||
// Resolution of delta quant
|
||||
int delta_q_res;
|
||||
|
|
@ -423,29 +512,39 @@ typedef struct AV1Common {
|
|||
int delta_lf_present_flag;
|
||||
// Resolution of delta lf level
|
||||
int delta_lf_res;
|
||||
#endif
|
||||
#if CONFIG_LOOPFILTER_LEVEL
|
||||
// This is a flag for number of deltas of loop filter level
|
||||
// 0: use 1 delta, for y_vertical, y_horizontal, u, and v
|
||||
// 1: use separate deltas for each filter level
|
||||
int delta_lf_multi;
|
||||
#endif // CONFIG_LOOPFILTER_LEVEL
|
||||
#endif
|
||||
int num_tg;
|
||||
#if CONFIG_REFERENCE_BUFFER
|
||||
SequenceHeader seq_params;
|
||||
int current_frame_id;
|
||||
int ref_frame_id[REF_FRAMES];
|
||||
int valid_for_referencing[REF_FRAMES];
|
||||
int refresh_mask;
|
||||
int invalid_delta_frame_id_minus1;
|
||||
#endif
|
||||
#endif // CONFIG_REFERENCE_BUFFER
|
||||
#if CONFIG_ANS && ANS_MAX_SYMBOLS
|
||||
int ans_window_size_log2;
|
||||
#endif
|
||||
} AV1_COMMON;
|
||||
|
||||
#if CONFIG_REFERENCE_BUFFER
|
||||
/* Initial version of sequence header structure */
|
||||
typedef struct SequenceHeader {
|
||||
int frame_id_numbers_present_flag;
|
||||
int frame_id_length_minus7;
|
||||
int delta_frame_id_length_minus2;
|
||||
} SequenceHeader;
|
||||
#if CONFIG_NCOBMC_ADAPT_WEIGHT
|
||||
NCOBMC_KERNELS ncobmc_kernels[ADAPT_OVERLAP_BLOCKS][ALL_NCOBMC_MODES];
|
||||
uint8_t *ncobmcaw_buf[4];
|
||||
#endif
|
||||
#if CONFIG_LV_MAP
|
||||
LV_MAP_CTX_TABLE coeff_ctx_table;
|
||||
#endif
|
||||
#if CONFIG_LPF_SB
|
||||
int final_lpf_encode;
|
||||
#endif
|
||||
#if CONFIG_ADAPT_SCAN
|
||||
int use_adapt_scan;
|
||||
#endif
|
||||
} AV1_COMMON;
|
||||
|
||||
// TODO(hkuang): Don't need to lock the whole pool after implementing atomic
|
||||
// frame reference count.
|
||||
|
|
@ -507,15 +606,57 @@ static INLINE void ref_cnt_fb(RefCntBuffer *bufs, int *idx, int new_idx) {
|
|||
bufs[new_idx].ref_count++;
|
||||
}
|
||||
|
||||
#if CONFIG_TEMPMV_SIGNALING
|
||||
// Returns 1 if this frame might use mvs from some previous frame. This
|
||||
// function doesn't consider whether prev_frame is actually suitable (see
|
||||
// frame_can_use_prev_frame_mvs for that)
|
||||
static INLINE int frame_might_use_prev_frame_mvs(const AV1_COMMON *cm) {
|
||||
return !cm->error_resilient_mode && !cm->intra_only;
|
||||
}
|
||||
|
||||
// Returns 1 if this frame really can use MVs from some previous frame.
|
||||
static INLINE int frame_can_use_prev_frame_mvs(const AV1_COMMON *cm) {
|
||||
return (frame_might_use_prev_frame_mvs(cm) && cm->last_show_frame &&
|
||||
cm->prev_frame && !cm->prev_frame->intra_only &&
|
||||
cm->width == cm->prev_frame->width &&
|
||||
cm->height == cm->prev_frame->height);
|
||||
}
|
||||
#endif
|
||||
|
||||
static INLINE void ensure_mv_buffer(RefCntBuffer *buf, AV1_COMMON *cm) {
|
||||
if (buf->mvs == NULL || buf->mi_rows < cm->mi_rows ||
|
||||
buf->mi_cols < cm->mi_cols) {
|
||||
aom_free(buf->mvs);
|
||||
buf->mi_rows = cm->mi_rows;
|
||||
buf->mi_cols = cm->mi_cols;
|
||||
#if CONFIG_TMV
|
||||
CHECK_MEM_ERROR(cm, buf->mvs,
|
||||
(MV_REF *)aom_calloc(
|
||||
((cm->mi_rows + 1) >> 1) * ((cm->mi_cols + 1) >> 1),
|
||||
sizeof(*buf->mvs)));
|
||||
#else
|
||||
CHECK_MEM_ERROR(
|
||||
cm, buf->mvs,
|
||||
(MV_REF *)aom_calloc(cm->mi_rows * cm->mi_cols, sizeof(*buf->mvs)));
|
||||
#endif // CONFIG_TMV
|
||||
|
||||
#if CONFIG_MFMV
|
||||
aom_free(buf->tpl_mvs);
|
||||
CHECK_MEM_ERROR(
|
||||
cm, buf->tpl_mvs,
|
||||
(TPL_MV_REF *)aom_calloc((cm->mi_rows + MAX_MIB_SIZE) * cm->mi_stride,
|
||||
sizeof(*buf->tpl_mvs)));
|
||||
#endif
|
||||
}
|
||||
}
|
||||
|
||||
#if CONFIG_VAR_REFS
|
||||
#define LAST_IS_VALID(cm) ((cm)->frame_refs[LAST_FRAME - 1].is_valid)
|
||||
#define LAST2_IS_VALID(cm) ((cm)->frame_refs[LAST2_FRAME - 1].is_valid)
|
||||
#define LAST3_IS_VALID(cm) ((cm)->frame_refs[LAST3_FRAME - 1].is_valid)
|
||||
#define GOLDEN_IS_VALID(cm) ((cm)->frame_refs[GOLDEN_FRAME - 1].is_valid)
|
||||
#define BWDREF_IS_VALID(cm) ((cm)->frame_refs[BWDREF_FRAME - 1].is_valid)
|
||||
#if CONFIG_ALTREF2
|
||||
#define ALTREF2_IS_VALID(cm) ((cm)->frame_refs[ALTREF2_FRAME - 1].is_valid)
|
||||
#endif // CONFIG_ALTREF2
|
||||
#define ALTREF_IS_VALID(cm) ((cm)->frame_refs[ALTREF_FRAME - 1].is_valid)
|
||||
|
||||
#define L_OR_L2(cm) (LAST_IS_VALID(cm) || LAST2_IS_VALID(cm))
|
||||
|
|
@ -526,10 +667,8 @@ static INLINE void ref_cnt_fb(RefCntBuffer *bufs, int *idx, int new_idx) {
|
|||
#define L3_OR_G(cm) (LAST3_IS_VALID(cm) || GOLDEN_IS_VALID(cm))
|
||||
#define L3_AND_G(cm) (LAST3_IS_VALID(cm) && GOLDEN_IS_VALID(cm))
|
||||
|
||||
#if CONFIG_ALTREF2
|
||||
#define BWD_OR_ALT2(cm) (BWDREF_IS_VALID(cm) || ALTREF2_IS_VALID(cm))
|
||||
#define BWD_AND_ALT2(cm) (BWDREF_IS_VALID(cm) && ALTREF2_IS_VALID(cm))
|
||||
#endif // CONFIG_ALTREF2
|
||||
#define BWD_OR_ALT(cm) (BWDREF_IS_VALID(cm) || ALTREF_IS_VALID(cm))
|
||||
#define BWD_AND_ALT(cm) (BWDREF_IS_VALID(cm) && ALTREF_IS_VALID(cm))
|
||||
#endif // CONFIG_VAR_REFS
|
||||
|
|
@ -546,6 +685,15 @@ static INLINE int frame_is_intra_only(const AV1_COMMON *const cm) {
|
|||
return cm->frame_type == KEY_FRAME || cm->intra_only;
|
||||
}
|
||||
|
||||
#if CONFIG_CFL
|
||||
#if CONFIG_CHROMA_SUB8X8 && CONFIG_DEBUG
|
||||
static INLINE void cfl_clear_sub8x8_val(CFL_CTX *cfl) {
|
||||
memset(cfl->sub8x8_val, 0, sizeof(cfl->sub8x8_val));
|
||||
}
|
||||
#endif // CONFIG_CHROMA_SUB8X8 && CONFIG_DEBUG
|
||||
void cfl_init(CFL_CTX *cfl, AV1_COMMON *cm);
|
||||
#endif // CONFIG_CFL
|
||||
|
||||
static INLINE void av1_init_macroblockd(AV1_COMMON *cm, MACROBLOCKD *xd,
|
||||
#if CONFIG_PVQ
|
||||
tran_low_t *pvq_ref_coeff,
|
||||
|
|
@ -602,11 +750,12 @@ static INLINE void set_skip_context(MACROBLOCKD *xd, int mi_row, int mi_col) {
|
|||
for (i = 0; i < MAX_MB_PLANE; ++i) {
|
||||
struct macroblockd_plane *const pd = &xd->plane[i];
|
||||
#if CONFIG_CHROMA_SUB8X8
|
||||
if (xd->mi[0]->mbmi.sb_type < BLOCK_8X8) {
|
||||
// Offset the buffer pointer
|
||||
if (pd->subsampling_y && (mi_row & 0x01)) row_offset = mi_row - 1;
|
||||
if (pd->subsampling_x && (mi_col & 0x01)) col_offset = mi_col - 1;
|
||||
}
|
||||
// Offset the buffer pointer
|
||||
const BLOCK_SIZE bsize = xd->mi[0]->mbmi.sb_type;
|
||||
if (pd->subsampling_y && (mi_row & 0x01) && (mi_size_high[bsize] == 1))
|
||||
row_offset = mi_row - 1;
|
||||
if (pd->subsampling_x && (mi_col & 0x01) && (mi_size_wide[bsize] == 1))
|
||||
col_offset = mi_col - 1;
|
||||
#endif
|
||||
int above_idx = col_offset << (MI_SIZE_LOG2 - tx_size_wide_log2[0]);
|
||||
int left_idx = (row_offset & MAX_MIB_MASK)
|
||||
|
|
@ -713,7 +862,14 @@ static INLINE aom_cdf_prob *get_y_mode_cdf(FRAME_CONTEXT *tile_ctx,
|
|||
int block) {
|
||||
const PREDICTION_MODE above = av1_above_block_mode(mi, above_mi, block);
|
||||
const PREDICTION_MODE left = av1_left_block_mode(mi, left_mi, block);
|
||||
|
||||
#if CONFIG_KF_CTX
|
||||
int above_ctx = intra_mode_context[above];
|
||||
int left_ctx = intra_mode_context[left];
|
||||
return tile_ctx->kf_y_cdf[above_ctx][left_ctx];
|
||||
#else
|
||||
return tile_ctx->kf_y_cdf[above][left];
|
||||
#endif
|
||||
}
|
||||
|
||||
static INLINE void update_partition_context(MACROBLOCKD *xd, int mi_row,
|
||||
|
|
@ -796,14 +952,54 @@ static INLINE BLOCK_SIZE scale_chroma_bsize(BLOCK_SIZE bsize, int subsampling_x,
|
|||
}
|
||||
#endif
|
||||
|
||||
static INLINE aom_cdf_prob cdf_element_prob(const aom_cdf_prob *cdf,
|
||||
size_t element) {
|
||||
assert(cdf != NULL);
|
||||
#if !CONFIG_ANS
|
||||
return (element > 0 ? cdf[element - 1] : CDF_PROB_TOP) - cdf[element];
|
||||
#else
|
||||
return cdf[element] - (element > 0 ? cdf[element - 1] : 0);
|
||||
#endif
|
||||
}
|
||||
|
||||
static INLINE void partition_gather_horz_alike(aom_cdf_prob *out,
|
||||
const aom_cdf_prob *const in) {
|
||||
out[0] = CDF_PROB_TOP;
|
||||
out[0] -= cdf_element_prob(in, PARTITION_HORZ);
|
||||
out[0] -= cdf_element_prob(in, PARTITION_SPLIT);
|
||||
#if CONFIG_EXT_PARTITION_TYPES
|
||||
out[0] -= cdf_element_prob(in, PARTITION_HORZ_A);
|
||||
out[0] -= cdf_element_prob(in, PARTITION_HORZ_B);
|
||||
out[0] -= cdf_element_prob(in, PARTITION_VERT_A);
|
||||
#endif
|
||||
out[0] = AOM_ICDF(out[0]);
|
||||
out[1] = AOM_ICDF(CDF_PROB_TOP);
|
||||
}
|
||||
|
||||
static INLINE void partition_gather_vert_alike(aom_cdf_prob *out,
|
||||
const aom_cdf_prob *const in) {
|
||||
out[0] = CDF_PROB_TOP;
|
||||
out[0] -= cdf_element_prob(in, PARTITION_VERT);
|
||||
out[0] -= cdf_element_prob(in, PARTITION_SPLIT);
|
||||
#if CONFIG_EXT_PARTITION_TYPES
|
||||
out[0] -= cdf_element_prob(in, PARTITION_HORZ_A);
|
||||
out[0] -= cdf_element_prob(in, PARTITION_VERT_A);
|
||||
out[0] -= cdf_element_prob(in, PARTITION_VERT_B);
|
||||
#endif
|
||||
out[0] = AOM_ICDF(out[0]);
|
||||
out[1] = AOM_ICDF(CDF_PROB_TOP);
|
||||
}
|
||||
|
||||
#if CONFIG_EXT_PARTITION_TYPES
|
||||
static INLINE void update_ext_partition_context(MACROBLOCKD *xd, int mi_row,
|
||||
int mi_col, BLOCK_SIZE subsize,
|
||||
BLOCK_SIZE bsize,
|
||||
PARTITION_TYPE partition) {
|
||||
if (bsize >= BLOCK_8X8) {
|
||||
#if !CONFIG_EXT_PARTITION_TYPES_AB
|
||||
const int hbs = mi_size_wide[bsize] / 2;
|
||||
BLOCK_SIZE bsize2 = get_subsize(bsize, PARTITION_SPLIT);
|
||||
#endif
|
||||
switch (partition) {
|
||||
case PARTITION_SPLIT:
|
||||
if (bsize != BLOCK_8X8) break;
|
||||
|
|
@ -814,6 +1010,30 @@ static INLINE void update_ext_partition_context(MACROBLOCKD *xd, int mi_row,
|
|||
case PARTITION_VERT_4:
|
||||
update_partition_context(xd, mi_row, mi_col, subsize, bsize);
|
||||
break;
|
||||
#if CONFIG_EXT_PARTITION_TYPES_AB
|
||||
case PARTITION_HORZ_A:
|
||||
update_partition_context(xd, mi_row, mi_col,
|
||||
get_subsize(bsize, PARTITION_HORZ_4), subsize);
|
||||
update_partition_context(xd, mi_row + mi_size_high[bsize] / 2, mi_col,
|
||||
subsize, subsize);
|
||||
break;
|
||||
case PARTITION_HORZ_B:
|
||||
update_partition_context(xd, mi_row, mi_col, subsize, subsize);
|
||||
update_partition_context(xd, mi_row + mi_size_high[bsize] / 2, mi_col,
|
||||
get_subsize(bsize, PARTITION_HORZ_4), subsize);
|
||||
break;
|
||||
case PARTITION_VERT_A:
|
||||
update_partition_context(xd, mi_row, mi_col,
|
||||
get_subsize(bsize, PARTITION_VERT_4), subsize);
|
||||
update_partition_context(xd, mi_row, mi_col + mi_size_wide[bsize] / 2,
|
||||
subsize, subsize);
|
||||
break;
|
||||
case PARTITION_VERT_B:
|
||||
update_partition_context(xd, mi_row, mi_col, subsize, subsize);
|
||||
update_partition_context(xd, mi_row, mi_col + mi_size_wide[bsize] / 2,
|
||||
get_subsize(bsize, PARTITION_VERT_4), subsize);
|
||||
break;
|
||||
#else
|
||||
case PARTITION_HORZ_A:
|
||||
update_partition_context(xd, mi_row, mi_col, bsize2, subsize);
|
||||
update_partition_context(xd, mi_row + hbs, mi_col, subsize, subsize);
|
||||
|
|
@ -830,6 +1050,7 @@ static INLINE void update_ext_partition_context(MACROBLOCKD *xd, int mi_row,
|
|||
update_partition_context(xd, mi_row, mi_col, subsize, subsize);
|
||||
update_partition_context(xd, mi_row, mi_col + hbs, bsize2, subsize);
|
||||
break;
|
||||
#endif
|
||||
default: assert(0 && "Invalid partition type");
|
||||
}
|
||||
}
|
||||
|
|
@ -842,7 +1063,6 @@ static INLINE int partition_plane_context(const MACROBLOCKD *xd, int mi_row,
|
|||
int has_rows, int has_cols,
|
||||
#endif
|
||||
BLOCK_SIZE bsize) {
|
||||
#if CONFIG_UNPOISON_PARTITION_CTX
|
||||
const PARTITION_CONTEXT *above_ctx = xd->above_seg_context + mi_col;
|
||||
const PARTITION_CONTEXT *left_ctx =
|
||||
xd->left_seg_context + (mi_row & MAX_MIB_MASK);
|
||||
|
|
@ -853,6 +1073,7 @@ static INLINE int partition_plane_context(const MACROBLOCKD *xd, int mi_row,
|
|||
assert(b_width_log2_lookup[bsize] == b_height_log2_lookup[bsize]);
|
||||
assert(bsl >= 0);
|
||||
|
||||
#if CONFIG_UNPOISON_PARTITION_CTX
|
||||
if (has_rows && has_cols)
|
||||
return (left * 2 + above) + bsl * PARTITION_PLOFFSET;
|
||||
else if (has_rows && !has_cols)
|
||||
|
|
@ -860,18 +1081,8 @@ static INLINE int partition_plane_context(const MACROBLOCKD *xd, int mi_row,
|
|||
else if (!has_rows && has_cols)
|
||||
return PARTITION_CONTEXTS_PRIMARY + PARTITION_BLOCK_SIZES + bsl;
|
||||
else
|
||||
return PARTITION_CONTEXTS; // Bogus context, forced SPLIT
|
||||
return INVALID_PARTITION_CTX; // Bogus context, forced SPLIT
|
||||
#else
|
||||
const PARTITION_CONTEXT *above_ctx = xd->above_seg_context + mi_col;
|
||||
const PARTITION_CONTEXT *left_ctx =
|
||||
xd->left_seg_context + (mi_row & MAX_MIB_MASK);
|
||||
// Minimum partition point is 8x8. Offset the bsl accordingly.
|
||||
const int bsl = mi_width_log2_lookup[bsize] - mi_width_log2_lookup[BLOCK_8X8];
|
||||
int above = (*above_ctx >> bsl) & 1, left = (*left_ctx >> bsl) & 1;
|
||||
|
||||
assert(b_width_log2_lookup[bsize] == b_height_log2_lookup[bsize]);
|
||||
assert(bsl >= 0);
|
||||
|
||||
return (left * 2 + above) + bsl * PARTITION_PLOFFSET;
|
||||
#endif
|
||||
}
|
||||
|
|
@ -997,18 +1208,22 @@ static INLINE void txfm_partition_update(TXFM_CONTEXT *above_ctx,
|
|||
}
|
||||
|
||||
static INLINE TX_SIZE get_sqr_tx_size(int tx_dim) {
|
||||
TX_SIZE tx_size;
|
||||
switch (tx_dim) {
|
||||
#if CONFIG_EXT_PARTITION
|
||||
case 128:
|
||||
#endif
|
||||
#endif // CONFIG_EXT_PARTITION
|
||||
case 64:
|
||||
case 32: tx_size = TX_32X32; break;
|
||||
case 16: tx_size = TX_16X16; break;
|
||||
case 8: tx_size = TX_8X8; break;
|
||||
default: tx_size = TX_4X4;
|
||||
#if CONFIG_TX64X64
|
||||
return TX_64X64;
|
||||
#else
|
||||
return TX_32X32;
|
||||
#endif // CONFIG_TX64X64
|
||||
break;
|
||||
case 32: return TX_32X32; break;
|
||||
case 16: return TX_16X16; break;
|
||||
case 8: return TX_8X8; break;
|
||||
default: return TX_4X4;
|
||||
}
|
||||
return tx_size;
|
||||
}
|
||||
|
||||
static INLINE int txfm_partition_context(TXFM_CONTEXT *above_ctx,
|
||||
|
|
@ -1035,49 +1250,114 @@ static INLINE int txfm_partition_context(TXFM_CONTEXT *above_ctx,
|
|||
}
|
||||
#endif
|
||||
|
||||
// Compute the next partition in the direction of the sb_type stored in the mi
|
||||
// array, starting with bsize.
|
||||
static INLINE PARTITION_TYPE get_partition(const AV1_COMMON *const cm,
|
||||
int mi_row, int mi_col,
|
||||
BLOCK_SIZE bsize) {
|
||||
if (mi_row >= cm->mi_rows || mi_col >= cm->mi_cols) {
|
||||
return PARTITION_INVALID;
|
||||
} else {
|
||||
const int offset = mi_row * cm->mi_stride + mi_col;
|
||||
MODE_INFO **mi = cm->mi_grid_visible + offset;
|
||||
const MB_MODE_INFO *const mbmi = &mi[0]->mbmi;
|
||||
const int bsl = b_width_log2_lookup[bsize];
|
||||
const PARTITION_TYPE partition = partition_lookup[bsl][mbmi->sb_type];
|
||||
#if !CONFIG_EXT_PARTITION_TYPES
|
||||
return partition;
|
||||
if (mi_row >= cm->mi_rows || mi_col >= cm->mi_cols) return PARTITION_INVALID;
|
||||
|
||||
const int offset = mi_row * cm->mi_stride + mi_col;
|
||||
MODE_INFO **mi = cm->mi_grid_visible + offset;
|
||||
const BLOCK_SIZE subsize = mi[0]->mbmi.sb_type;
|
||||
|
||||
if (subsize == bsize) return PARTITION_NONE;
|
||||
|
||||
const int bhigh = mi_size_high[bsize];
|
||||
const int bwide = mi_size_wide[bsize];
|
||||
const int sshigh = mi_size_high[subsize];
|
||||
const int sswide = mi_size_wide[subsize];
|
||||
|
||||
#if CONFIG_EXT_PARTITION_TYPES
|
||||
if (bsize > BLOCK_8X8 && mi_row + bwide / 2 < cm->mi_rows &&
|
||||
mi_col + bhigh / 2 < cm->mi_cols) {
|
||||
// In this case, the block might be using an extended partition
|
||||
// type.
|
||||
const MB_MODE_INFO *const mbmi_right = &mi[bwide / 2]->mbmi;
|
||||
const MB_MODE_INFO *const mbmi_below = &mi[bhigh / 2 * cm->mi_stride]->mbmi;
|
||||
|
||||
if (sswide == bwide) {
|
||||
#if CONFIG_EXT_PARTITION_TYPES_AB
|
||||
// Smaller height but same width. Is PARTITION_HORZ, PARTITION_HORZ_4,
|
||||
// PARTITION_HORZ_A or PARTITION_HORZ_B.
|
||||
if (sshigh * 2 == bhigh)
|
||||
return (mbmi_below->sb_type == subsize) ? PARTITION_HORZ
|
||||
: PARTITION_HORZ_B;
|
||||
assert(sshigh * 4 == bhigh);
|
||||
return (mbmi_below->sb_type == subsize) ? PARTITION_HORZ_4
|
||||
: PARTITION_HORZ_A;
|
||||
#else
|
||||
const int hbs = mi_size_wide[bsize] / 2;
|
||||
// Smaller height but same width. Is PARTITION_HORZ_4, PARTITION_HORZ or
|
||||
// PARTITION_HORZ_B. To distinguish the latter two, check if the lower
|
||||
// half was split.
|
||||
if (sshigh * 4 == bhigh) return PARTITION_HORZ_4;
|
||||
assert(sshigh * 2 == bhigh);
|
||||
|
||||
assert(cm->mi_grid_visible[offset] == &cm->mi[offset]);
|
||||
if (mbmi_below->sb_type == subsize)
|
||||
return PARTITION_HORZ;
|
||||
else
|
||||
return PARTITION_HORZ_B;
|
||||
#endif
|
||||
} else if (sshigh == bhigh) {
|
||||
#if CONFIG_EXT_PARTITION_TYPES_AB
|
||||
// Smaller width but same height. Is PARTITION_VERT, PARTITION_VERT_4,
|
||||
// PARTITION_VERT_A or PARTITION_VERT_B.
|
||||
if (sswide * 2 == bwide)
|
||||
return (mbmi_right->sb_type == subsize) ? PARTITION_VERT
|
||||
: PARTITION_VERT_B;
|
||||
assert(sswide * 4 == bwide);
|
||||
return (mbmi_right->sb_type == subsize) ? PARTITION_VERT_4
|
||||
: PARTITION_VERT_A;
|
||||
#else
|
||||
// Smaller width but same height. Is PARTITION_VERT_4, PARTITION_VERT or
|
||||
// PARTITION_VERT_B. To distinguish the latter two, check if the right
|
||||
// half was split.
|
||||
if (sswide * 4 == bwide) return PARTITION_VERT_4;
|
||||
assert(sswide * 2 == bhigh);
|
||||
|
||||
if (partition == PARTITION_HORZ_4 || partition == PARTITION_VERT_4)
|
||||
return partition;
|
||||
if (mbmi_right->sb_type == subsize)
|
||||
return PARTITION_VERT;
|
||||
else
|
||||
return PARTITION_VERT_B;
|
||||
#endif
|
||||
} else {
|
||||
#if !CONFIG_EXT_PARTITION_TYPES_AB
|
||||
// Smaller width and smaller height. Might be PARTITION_SPLIT or could be
|
||||
// PARTITION_HORZ_A or PARTITION_VERT_A. If subsize isn't halved in both
|
||||
// dimensions, we immediately know this is a split (which will recurse to
|
||||
// get to subsize). Otherwise look down and to the right. With
|
||||
// PARTITION_VERT_A, the right block will have height bhigh; with
|
||||
// PARTITION_HORZ_A, the lower block with have width bwide. Otherwise
|
||||
// it's PARTITION_SPLIT.
|
||||
if (sswide * 2 != bwide || sshigh * 2 != bhigh) return PARTITION_SPLIT;
|
||||
|
||||
if (partition != PARTITION_NONE && bsize > BLOCK_8X8 &&
|
||||
mi_row + hbs < cm->mi_rows && mi_col + hbs < cm->mi_cols) {
|
||||
const BLOCK_SIZE h = get_subsize(bsize, PARTITION_HORZ_A);
|
||||
const BLOCK_SIZE v = get_subsize(bsize, PARTITION_VERT_A);
|
||||
const MB_MODE_INFO *const mbmi_right = &mi[hbs]->mbmi;
|
||||
const MB_MODE_INFO *const mbmi_below = &mi[hbs * cm->mi_stride]->mbmi;
|
||||
if (mbmi->sb_type == h) {
|
||||
return mbmi_below->sb_type == h ? PARTITION_HORZ : PARTITION_HORZ_B;
|
||||
} else if (mbmi->sb_type == v) {
|
||||
return mbmi_right->sb_type == v ? PARTITION_VERT : PARTITION_VERT_B;
|
||||
} else if (mbmi_below->sb_type == h) {
|
||||
return PARTITION_HORZ_A;
|
||||
} else if (mbmi_right->sb_type == v) {
|
||||
return PARTITION_VERT_A;
|
||||
} else {
|
||||
return PARTITION_SPLIT;
|
||||
}
|
||||
if (mi_size_wide[mbmi_below->sb_type] == bwide) return PARTITION_HORZ_A;
|
||||
if (mi_size_high[mbmi_right->sb_type] == bhigh) return PARTITION_VERT_A;
|
||||
#endif
|
||||
|
||||
return PARTITION_SPLIT;
|
||||
}
|
||||
|
||||
return partition;
|
||||
#endif // !CONFIG_EXT_PARTITION_TYPES
|
||||
}
|
||||
#endif
|
||||
const int vert_split = sswide < bwide;
|
||||
const int horz_split = sshigh < bhigh;
|
||||
const int split_idx = (vert_split << 1) | horz_split;
|
||||
assert(split_idx != 0);
|
||||
|
||||
static const PARTITION_TYPE base_partitions[4] = {
|
||||
PARTITION_INVALID, PARTITION_HORZ, PARTITION_VERT, PARTITION_SPLIT
|
||||
};
|
||||
|
||||
return base_partitions[split_idx];
|
||||
}
|
||||
|
||||
static INLINE void set_use_reference_buffer(AV1_COMMON *const cm, int use) {
|
||||
#if CONFIG_REFERENCE_BUFFER
|
||||
cm->seq_params.frame_id_numbers_present_flag = use;
|
||||
#else
|
||||
(void)cm;
|
||||
(void)use;
|
||||
#endif
|
||||
}
|
||||
|
||||
static INLINE void set_sb_size(AV1_COMMON *const cm, BLOCK_SIZE sb_size) {
|
||||
|
|
@ -1106,6 +1386,17 @@ static INLINE int all_lossless(const AV1_COMMON *cm, const MACROBLOCKD *xd) {
|
|||
return all_lossless;
|
||||
}
|
||||
|
||||
static INLINE int use_compressed_header(const AV1_COMMON *cm) {
|
||||
(void)cm;
|
||||
#if CONFIG_RESTRICT_COMPRESSED_HDR && CONFIG_NEW_MULTISYMBOL
|
||||
return 0;
|
||||
#elif CONFIG_RESTRICT_COMPRESSED_HDR
|
||||
return cm->refresh_frame_context == REFRESH_FRAME_CONTEXT_FORWARD;
|
||||
#else
|
||||
return 1;
|
||||
#endif // CONFIG_RESTRICT_COMPRESSED_HDR && CONFIG_NEW_MULTISYMBOL
|
||||
}
|
||||
|
||||
#ifdef __cplusplus
|
||||
} // extern "C"
|
||||
#endif
|
||||
|
|
|
|||
297
third_party/aom/av1/common/pred_common.c
vendored
297
third_party/aom/av1/common/pred_common.c
vendored
|
|
@ -22,19 +22,16 @@
|
|||
static InterpFilter get_ref_filter_type(const MODE_INFO *mi,
|
||||
const MACROBLOCKD *xd, int dir,
|
||||
MV_REFERENCE_FRAME ref_frame) {
|
||||
InterpFilter ref_type = SWITCHABLE_FILTERS;
|
||||
const MB_MODE_INFO *ref_mbmi = &mi->mbmi;
|
||||
int use_subpel[2] = {
|
||||
has_subpel_mv_component(mi, xd, dir),
|
||||
has_subpel_mv_component(mi, xd, dir + 2),
|
||||
};
|
||||
|
||||
if (ref_mbmi->ref_frame[0] == ref_frame && use_subpel[0])
|
||||
ref_type = ref_mbmi->interp_filter[(dir & 0x01)];
|
||||
else if (ref_mbmi->ref_frame[1] == ref_frame && use_subpel[1])
|
||||
ref_type = ref_mbmi->interp_filter[(dir & 0x01) + 2];
|
||||
|
||||
return ref_type;
|
||||
return (((ref_mbmi->ref_frame[0] == ref_frame && use_subpel[0]) ||
|
||||
(ref_mbmi->ref_frame[1] == ref_frame && use_subpel[1]))
|
||||
? av1_extract_interp_filter(ref_mbmi->interp_filters, dir & 0x01)
|
||||
: SWITCHABLE_FILTERS);
|
||||
}
|
||||
|
||||
int av1_get_pred_context_switchable_interp(const MACROBLOCKD *xd, int dir) {
|
||||
|
|
@ -79,13 +76,15 @@ int av1_get_pred_context_switchable_interp(const MACROBLOCKD *xd) {
|
|||
// left of the entries corresponding to real macroblocks.
|
||||
// The prediction flags in these dummy entries are initialized to 0.
|
||||
const MB_MODE_INFO *const left_mbmi = xd->left_mbmi;
|
||||
const int left_type = xd->left_available && is_inter_block(left_mbmi)
|
||||
? left_mbmi->interp_filter
|
||||
: SWITCHABLE_FILTERS;
|
||||
const int left_type =
|
||||
xd->left_available && is_inter_block(left_mbmi)
|
||||
? av1_extract_interp_filter(left_mbmi->interp_filters, 0)
|
||||
: SWITCHABLE_FILTERS;
|
||||
const MB_MODE_INFO *const above_mbmi = xd->above_mbmi;
|
||||
const int above_type = xd->up_available && is_inter_block(above_mbmi)
|
||||
? above_mbmi->interp_filter
|
||||
: SWITCHABLE_FILTERS;
|
||||
const int above_type =
|
||||
xd->up_available && is_inter_block(above_mbmi)
|
||||
? av1_extract_interp_filter(above_mbmi->interp_filters, 0)
|
||||
: SWITCHABLE_FILTERS;
|
||||
|
||||
if (left_type == above_type) {
|
||||
return left_type;
|
||||
|
|
@ -110,11 +109,7 @@ static INTRA_FILTER get_ref_intra_filter(const MB_MODE_INFO *ref_mbmi) {
|
|||
if (ref_mbmi->sb_type >= BLOCK_8X8) {
|
||||
const PREDICTION_MODE mode = ref_mbmi->mode;
|
||||
if (is_inter_block(ref_mbmi)) {
|
||||
#if CONFIG_DUAL_FILTER
|
||||
switch (ref_mbmi->interp_filter[0]) {
|
||||
#else
|
||||
switch (ref_mbmi->interp_filter) {
|
||||
#endif
|
||||
switch (av1_extract_interp_filter(ref_mbmi->interp_filters, 0)) {
|
||||
case EIGHTTAP_REGULAR: ref_type = INTRA_FILTER_8TAP; break;
|
||||
case EIGHTTAP_SMOOTH: ref_type = INTRA_FILTER_8TAP_SMOOTH; break;
|
||||
case MULTITAP_SHARP: ref_type = INTRA_FILTER_8TAP_SHARP; break;
|
||||
|
|
@ -153,9 +148,14 @@ int av1_get_pred_context_intra_interp(const MACROBLOCKD *xd) {
|
|||
#endif // CONFIG_INTRA_INTERP
|
||||
#endif // CONFIG_EXT_INTRA
|
||||
|
||||
#if CONFIG_PALETTE && CONFIG_PALETTE_DELTA_ENCODING
|
||||
int av1_get_palette_cache(const MODE_INFO *above_mi, const MODE_INFO *left_mi,
|
||||
int plane, uint16_t *cache) {
|
||||
#if CONFIG_PALETTE_DELTA_ENCODING
|
||||
int av1_get_palette_cache(const MACROBLOCKD *const xd, int plane,
|
||||
uint16_t *cache) {
|
||||
const int row = -xd->mb_to_top_edge >> 3;
|
||||
// Do not refer to above SB row when on SB boundary.
|
||||
const MODE_INFO *const above_mi =
|
||||
(row % (1 << MIN_SB_SIZE_LOG2)) ? xd->above_mi : NULL;
|
||||
const MODE_INFO *const left_mi = xd->left_mi;
|
||||
int above_n = 0, left_n = 0;
|
||||
if (above_mi)
|
||||
above_n = above_mi->mbmi.palette_mode_info.palette_size[plane != 0];
|
||||
|
|
@ -166,8 +166,9 @@ int av1_get_palette_cache(const MODE_INFO *above_mi, const MODE_INFO *left_mi,
|
|||
int left_idx = plane * PALETTE_MAX_SIZE;
|
||||
int n = 0;
|
||||
const uint16_t *above_colors =
|
||||
above_mi->mbmi.palette_mode_info.palette_colors;
|
||||
const uint16_t *left_colors = left_mi->mbmi.palette_mode_info.palette_colors;
|
||||
above_mi ? above_mi->mbmi.palette_mode_info.palette_colors : NULL;
|
||||
const uint16_t *left_colors =
|
||||
left_mi ? left_mi->mbmi.palette_mode_info.palette_colors : NULL;
|
||||
// Merge the sorted lists of base colors from above and left to get
|
||||
// combined sorted color cache.
|
||||
while (above_n > 0 && left_n > 0) {
|
||||
|
|
@ -193,7 +194,7 @@ int av1_get_palette_cache(const MODE_INFO *above_mi, const MODE_INFO *left_mi,
|
|||
assert(n <= 2 * PALETTE_MAX_SIZE);
|
||||
return n;
|
||||
}
|
||||
#endif // CONFIG_PALETTE && CONFIG_PALETTE_DELTA_ENCODING
|
||||
#endif // CONFIG_PALETTE_DELTA_ENCODING
|
||||
|
||||
// The mode info data structure has a one element border above and to the
|
||||
// left of the entries corresponding to real macroblocks.
|
||||
|
|
@ -219,7 +220,7 @@ int av1_get_intra_inter_context(const MACROBLOCKD *xd) {
|
|||
}
|
||||
}
|
||||
|
||||
#if CONFIG_EXT_INTER && CONFIG_COMPOUND_SINGLEREF
|
||||
#if CONFIG_COMPOUND_SINGLEREF
|
||||
// The compound/single mode info data structure has one element border above and
|
||||
// to the left of the entries corresponding to real macroblocks.
|
||||
// The prediction flags in these dummy entries are initialized to 0.
|
||||
|
|
@ -253,7 +254,7 @@ int av1_get_inter_mode_context(const MACROBLOCKD *xd) {
|
|||
return 2;
|
||||
}
|
||||
}
|
||||
#endif // CONFIG_EXT_INTER && CONFIG_COMPOUND_SINGLEREF
|
||||
#endif // CONFIG_COMPOUND_SINGLEREF
|
||||
|
||||
#if CONFIG_EXT_REFS
|
||||
#define CHECK_BACKWARD_REFS(ref_frame) \
|
||||
|
|
@ -314,8 +315,6 @@ int av1_get_reference_mode_context(const AV1_COMMON *cm,
|
|||
}
|
||||
|
||||
#if CONFIG_EXT_COMP_REFS
|
||||
#define CHECK_BWDREF_OR_ALTREF(ref_frame) \
|
||||
((ref_frame) == BWDREF_FRAME || (ref_frame) == ALTREF_FRAME)
|
||||
// TODO(zoeliu): To try on the design of 3 contexts, instead of 5:
|
||||
// COMP_REF_TYPE_CONTEXTS = 3
|
||||
int av1_get_comp_reference_type_context(const MACROBLOCKD *xd) {
|
||||
|
|
@ -345,9 +344,9 @@ int av1_get_comp_reference_type_context(const MACROBLOCKD *xd) {
|
|||
const MV_REFERENCE_FRAME frfl = left_mbmi->ref_frame[0];
|
||||
|
||||
if (a_sg && l_sg) { // single/single
|
||||
pred_context = 1 +
|
||||
2 * (!(CHECK_BWDREF_OR_ALTREF(frfa) ^
|
||||
CHECK_BWDREF_OR_ALTREF(frfl)));
|
||||
pred_context =
|
||||
1 +
|
||||
2 * (!(IS_BACKWARD_REF_FRAME(frfa) ^ IS_BACKWARD_REF_FRAME(frfl)));
|
||||
} else if (l_sg || a_sg) { // single/comp
|
||||
const int uni_rfc =
|
||||
a_sg ? has_uni_comp_refs(left_mbmi) : has_uni_comp_refs(above_mbmi);
|
||||
|
|
@ -355,8 +354,8 @@ int av1_get_comp_reference_type_context(const MACROBLOCKD *xd) {
|
|||
if (!uni_rfc) // comp bidir
|
||||
pred_context = 1;
|
||||
else // comp unidir
|
||||
pred_context = 3 + (!(CHECK_BWDREF_OR_ALTREF(frfa) ^
|
||||
CHECK_BWDREF_OR_ALTREF(frfl)));
|
||||
pred_context = 3 + (!(IS_BACKWARD_REF_FRAME(frfa) ^
|
||||
IS_BACKWARD_REF_FRAME(frfl)));
|
||||
} else { // comp/comp
|
||||
const int a_uni_rfc = has_uni_comp_refs(above_mbmi);
|
||||
const int l_uni_rfc = has_uni_comp_refs(left_mbmi);
|
||||
|
|
@ -580,12 +579,12 @@ int av1_get_pred_context_comp_ref_p(const AV1_COMMON *cm,
|
|||
// The mode info data structure has a one element border above and to the
|
||||
// left of the entries correpsonding to real macroblocks.
|
||||
// The prediction flags in these dummy entries are initialised to 0.
|
||||
#if CONFIG_ONE_SIDED_COMPOUND || CONFIG_EXT_COMP_REFS // No change to bitstream
|
||||
#if CONFIG_ONE_SIDED_COMPOUND || CONFIG_FRAME_SIGN_BIAS
|
||||
// Code seems to assume that signbias of cm->comp_bwd_ref[0] is always 1
|
||||
const int bwd_ref_sign_idx = 1;
|
||||
#else
|
||||
const int bwd_ref_sign_idx = cm->ref_frame_sign_bias[cm->comp_bwd_ref[0]];
|
||||
#endif // CONFIG_ONE_SIDED_COMPOUND || CONFIG_EXT_COMP_REFS
|
||||
#endif // CONFIG_ONE_SIDED_COMPOUND || CONFIG_FRAME_SIGN_BIAS
|
||||
const int fwd_ref_sign_idx = !bwd_ref_sign_idx;
|
||||
|
||||
(void)cm;
|
||||
|
|
@ -690,12 +689,12 @@ int av1_get_pred_context_comp_ref_p1(const AV1_COMMON *cm,
|
|||
// The mode info data structure has a one element border above and to the
|
||||
// left of the entries correpsonding to real macroblocks.
|
||||
// The prediction flags in these dummy entries are initialised to 0.
|
||||
#if CONFIG_ONE_SIDED_COMPOUND || CONFIG_EXT_COMP_REFS // No change to bitstream
|
||||
#if CONFIG_ONE_SIDED_COMPOUND || CONFIG_FRAME_SIGN_BIAS
|
||||
// Code seems to assume that signbias of cm->comp_bwd_ref[0] is always 1
|
||||
const int bwd_ref_sign_idx = 1;
|
||||
#else
|
||||
const int bwd_ref_sign_idx = cm->ref_frame_sign_bias[cm->comp_bwd_ref[0]];
|
||||
#endif // CONFIG_ONE_SIDED_COMPOUND || CONFIG_EXT_COMP_REFS
|
||||
#endif // CONFIG_ONE_SIDED_COMPOUND || CONFIG_FRAME_SIGN_BIAS
|
||||
const int fwd_ref_sign_idx = !bwd_ref_sign_idx;
|
||||
|
||||
(void)cm;
|
||||
|
|
@ -798,12 +797,11 @@ int av1_get_pred_context_comp_ref_p2(const AV1_COMMON *cm,
|
|||
// The mode info data structure has a one element border above and to the
|
||||
// left of the entries correpsonding to real macroblocks.
|
||||
// The prediction flags in these dummy entries are initialised to 0.
|
||||
#if CONFIG_ONE_SIDED_COMPOUND || CONFIG_EXT_COMP_REFS // No change to bitstream
|
||||
// Code seems to assume that signbias of cm->comp_bwd_ref[0] is always 1
|
||||
#if CONFIG_ONE_SIDED_COMPOUND || CONFIG_FRAME_SIGN_BIAS
|
||||
const int bwd_ref_sign_idx = 1;
|
||||
#else
|
||||
const int bwd_ref_sign_idx = cm->ref_frame_sign_bias[cm->comp_bwd_ref[0]];
|
||||
#endif // CONFIG_ONE_SIDED_COMPOUND || CONFIG_EXT_COMP_REFS
|
||||
#endif // CONFIG_ONE_SIDED_COMPOUND || CONFIG_FRAME_SIGN_BIAS
|
||||
const int fwd_ref_sign_idx = !bwd_ref_sign_idx;
|
||||
|
||||
(void)cm;
|
||||
|
|
@ -887,8 +885,6 @@ int av1_get_pred_context_comp_ref_p2(const AV1_COMMON *cm,
|
|||
return pred_context;
|
||||
}
|
||||
|
||||
#if CONFIG_ALTREF2
|
||||
|
||||
// Obtain contexts to signal a reference frame be either BWDREF/ALTREF2, or
|
||||
// ALTREF.
|
||||
int av1_get_pred_context_brfarf2_or_arf(const MACROBLOCKD *xd) {
|
||||
|
|
@ -989,132 +985,6 @@ int av1_get_pred_context_comp_bwdref_p1(const AV1_COMMON *cm,
|
|||
return av1_get_pred_context_brf_or_arf2(xd);
|
||||
}
|
||||
|
||||
#else // !CONFIG_ALTREF2
|
||||
|
||||
// Returns a context number for the given MB prediction signal
|
||||
int av1_get_pred_context_comp_bwdref_p(const AV1_COMMON *cm,
|
||||
const MACROBLOCKD *xd) {
|
||||
int pred_context;
|
||||
const MB_MODE_INFO *const above_mbmi = xd->above_mbmi;
|
||||
const MB_MODE_INFO *const left_mbmi = xd->left_mbmi;
|
||||
const int above_in_image = xd->up_available;
|
||||
const int left_in_image = xd->left_available;
|
||||
|
||||
// Note:
|
||||
// The mode info data structure has a one element border above and to the
|
||||
// left of the entries corresponding to real macroblocks.
|
||||
// The prediction flags in these dummy entries are initialized to 0.
|
||||
#if CONFIG_ONE_SIDED_COMPOUND || CONFIG_EXT_COMP_REFS // No change to bitstream
|
||||
// Code seems to assume that signbias of cm->comp_bwd_ref[0] is always 1
|
||||
const int bwd_ref_sign_idx = 1;
|
||||
#else
|
||||
const int bwd_ref_sign_idx = cm->ref_frame_sign_bias[cm->comp_bwd_ref[0]];
|
||||
#endif // CONFIG_ONE_SIDED_COMPOUND || CONFIG_EXT_COMP_REFS
|
||||
const int fwd_ref_sign_idx = !bwd_ref_sign_idx;
|
||||
|
||||
(void)cm;
|
||||
|
||||
if (above_in_image && left_in_image) { // both edges available
|
||||
const int above_intra = !is_inter_block(above_mbmi);
|
||||
const int left_intra = !is_inter_block(left_mbmi);
|
||||
|
||||
if (above_intra && left_intra) { // intra/intra (2)
|
||||
pred_context = 2;
|
||||
} else if (above_intra || left_intra) { // intra/inter
|
||||
const MB_MODE_INFO *edge_mbmi = above_intra ? left_mbmi : above_mbmi;
|
||||
|
||||
if (!has_second_ref(edge_mbmi)) // single pred (1/3)
|
||||
pred_context = 1 + 2 * (edge_mbmi->ref_frame[0] != cm->comp_bwd_ref[1]);
|
||||
else // comp pred (1/3)
|
||||
pred_context =
|
||||
1 +
|
||||
2 * (edge_mbmi->ref_frame[bwd_ref_sign_idx] != cm->comp_bwd_ref[1]);
|
||||
} else { // inter/inter
|
||||
const int l_comp = has_second_ref(left_mbmi);
|
||||
const int a_comp = has_second_ref(above_mbmi);
|
||||
|
||||
const MV_REFERENCE_FRAME l_brf =
|
||||
l_comp ? left_mbmi->ref_frame[bwd_ref_sign_idx] : NONE_FRAME;
|
||||
const MV_REFERENCE_FRAME a_brf =
|
||||
a_comp ? above_mbmi->ref_frame[bwd_ref_sign_idx] : NONE_FRAME;
|
||||
|
||||
const MV_REFERENCE_FRAME l_frf =
|
||||
!l_comp ? left_mbmi->ref_frame[0]
|
||||
: left_mbmi->ref_frame[fwd_ref_sign_idx];
|
||||
const MV_REFERENCE_FRAME a_frf =
|
||||
!a_comp ? above_mbmi->ref_frame[0]
|
||||
: above_mbmi->ref_frame[fwd_ref_sign_idx];
|
||||
|
||||
if (l_comp && a_comp) { // comp/comp
|
||||
if (l_brf == a_brf && l_brf == cm->comp_bwd_ref[1]) {
|
||||
pred_context = 0;
|
||||
} else if (l_brf == cm->comp_bwd_ref[1] ||
|
||||
a_brf == cm->comp_bwd_ref[1]) {
|
||||
pred_context = 1;
|
||||
} else {
|
||||
// NOTE: Backward ref should be either BWDREF or ALTREF.
|
||||
#if !USE_UNI_COMP_REFS
|
||||
// TODO(zoeliu): To further study the UNIDIR scenario
|
||||
assert(l_brf == a_brf && l_brf != cm->comp_bwd_ref[1]);
|
||||
#endif // !USE_UNI_COMP_REFS
|
||||
pred_context = 3;
|
||||
}
|
||||
} else if (!l_comp && !a_comp) { // single/single
|
||||
if (l_frf == a_frf && l_frf == cm->comp_bwd_ref[1]) {
|
||||
pred_context = 0;
|
||||
} else if (l_frf == cm->comp_bwd_ref[1] ||
|
||||
a_frf == cm->comp_bwd_ref[1]) {
|
||||
pred_context = 1;
|
||||
} else if (l_frf == a_frf) {
|
||||
pred_context = 3;
|
||||
} else {
|
||||
#if !USE_UNI_COMP_REFS
|
||||
// TODO(zoeliu): To further study the UNIDIR scenario
|
||||
assert(l_frf != a_frf && l_frf != cm->comp_bwd_ref[1] &&
|
||||
a_frf != cm->comp_bwd_ref[1]);
|
||||
#endif // !USE_UNI_COMP_REFS
|
||||
pred_context = 4;
|
||||
}
|
||||
} else { // comp/single
|
||||
assert((l_comp && !a_comp) || (!l_comp && a_comp));
|
||||
|
||||
if ((l_comp && l_brf == cm->comp_bwd_ref[1] &&
|
||||
a_frf == cm->comp_bwd_ref[1]) ||
|
||||
(a_comp && a_brf == cm->comp_bwd_ref[1] &&
|
||||
l_frf == cm->comp_bwd_ref[1])) {
|
||||
pred_context = 1;
|
||||
} else if ((l_comp && l_brf == cm->comp_bwd_ref[1]) ||
|
||||
(a_comp && a_brf == cm->comp_bwd_ref[1]) ||
|
||||
(!l_comp && l_frf == cm->comp_bwd_ref[1]) ||
|
||||
(!a_comp && a_frf == cm->comp_bwd_ref[1])) {
|
||||
pred_context = 2;
|
||||
} else {
|
||||
pred_context = 4;
|
||||
}
|
||||
}
|
||||
}
|
||||
} else if (above_in_image || left_in_image) { // one edge available
|
||||
const MB_MODE_INFO *edge_mbmi = above_in_image ? above_mbmi : left_mbmi;
|
||||
|
||||
if (!is_inter_block(edge_mbmi)) {
|
||||
pred_context = 2;
|
||||
} else {
|
||||
if (has_second_ref(edge_mbmi)) {
|
||||
pred_context =
|
||||
4 * (edge_mbmi->ref_frame[bwd_ref_sign_idx] != cm->comp_bwd_ref[1]);
|
||||
} else {
|
||||
pred_context = 3 * (edge_mbmi->ref_frame[0] != cm->comp_bwd_ref[1]);
|
||||
}
|
||||
}
|
||||
} else { // no edges available (2)
|
||||
pred_context = 2;
|
||||
}
|
||||
assert(pred_context >= 0 && pred_context < REF_CONTEXTS);
|
||||
|
||||
return pred_context;
|
||||
}
|
||||
#endif // CONFIG_ALTREF2
|
||||
|
||||
#else // !CONFIG_EXT_REFS
|
||||
|
||||
// Returns a context number for the given MB prediction signal
|
||||
|
|
@ -1270,96 +1140,7 @@ int av1_get_pred_context_single_ref_p1(const MACROBLOCKD *xd) {
|
|||
// non-ALTREF backward reference frame, knowing that it shall be either of
|
||||
// these 2 choices.
|
||||
int av1_get_pred_context_single_ref_p2(const MACROBLOCKD *xd) {
|
||||
#if CONFIG_ALTREF2
|
||||
return av1_get_pred_context_brfarf2_or_arf(xd);
|
||||
#else // !CONFIG_ALTREF2
|
||||
int pred_context;
|
||||
const MB_MODE_INFO *const above_mbmi = xd->above_mbmi;
|
||||
const MB_MODE_INFO *const left_mbmi = xd->left_mbmi;
|
||||
const int has_above = xd->up_available;
|
||||
const int has_left = xd->left_available;
|
||||
|
||||
// Note:
|
||||
// The mode info data structure has a one element border above and to the
|
||||
// left of the entries correpsonding to real macroblocks.
|
||||
// The prediction flags in these dummy entries are initialised to 0.
|
||||
if (has_above && has_left) { // both edges available
|
||||
const int above_intra = !is_inter_block(above_mbmi);
|
||||
const int left_intra = !is_inter_block(left_mbmi);
|
||||
|
||||
if (above_intra && left_intra) { // intra/intra
|
||||
pred_context = 2;
|
||||
} else if (above_intra || left_intra) { // intra/inter or inter/intra
|
||||
const MB_MODE_INFO *edge_mbmi = above_intra ? left_mbmi : above_mbmi;
|
||||
if (!has_second_ref(edge_mbmi)) { // single
|
||||
if (!CHECK_BACKWARD_REFS(edge_mbmi->ref_frame[0]))
|
||||
pred_context = 3;
|
||||
else
|
||||
pred_context = 4 * (edge_mbmi->ref_frame[0] == BWDREF_FRAME);
|
||||
} else { // comp
|
||||
pred_context = 1 +
|
||||
2 * (edge_mbmi->ref_frame[0] == BWDREF_FRAME ||
|
||||
edge_mbmi->ref_frame[1] == BWDREF_FRAME);
|
||||
}
|
||||
} else { // inter/inter
|
||||
const int above_has_second = has_second_ref(above_mbmi);
|
||||
const int left_has_second = has_second_ref(left_mbmi);
|
||||
const MV_REFERENCE_FRAME above0 = above_mbmi->ref_frame[0];
|
||||
const MV_REFERENCE_FRAME above1 = above_mbmi->ref_frame[1];
|
||||
const MV_REFERENCE_FRAME left0 = left_mbmi->ref_frame[0];
|
||||
const MV_REFERENCE_FRAME left1 = left_mbmi->ref_frame[1];
|
||||
|
||||
if (above_has_second && left_has_second) { // comp/comp
|
||||
if (above0 == left0 && above1 == left1)
|
||||
pred_context =
|
||||
3 * (above0 == BWDREF_FRAME || above1 == BWDREF_FRAME ||
|
||||
left0 == BWDREF_FRAME || left1 == BWDREF_FRAME);
|
||||
else
|
||||
pred_context = 2;
|
||||
} else if (above_has_second || left_has_second) { // single/comp
|
||||
const MV_REFERENCE_FRAME rfs = !above_has_second ? above0 : left0;
|
||||
const MV_REFERENCE_FRAME crf1 = above_has_second ? above0 : left0;
|
||||
const MV_REFERENCE_FRAME crf2 = above_has_second ? above1 : left1;
|
||||
|
||||
if (rfs == BWDREF_FRAME)
|
||||
pred_context = 3 + (crf1 == BWDREF_FRAME || crf2 == BWDREF_FRAME);
|
||||
else if (rfs == ALTREF_FRAME)
|
||||
pred_context = (crf1 == BWDREF_FRAME || crf2 == BWDREF_FRAME);
|
||||
else
|
||||
pred_context = 1 + 2 * (crf1 == BWDREF_FRAME || crf2 == BWDREF_FRAME);
|
||||
} else { // single/single
|
||||
if (!CHECK_BACKWARD_REFS(above0) && !CHECK_BACKWARD_REFS(left0)) {
|
||||
pred_context = 2 + (above0 == left0);
|
||||
} else if (!CHECK_BACKWARD_REFS(above0) ||
|
||||
!CHECK_BACKWARD_REFS(left0)) {
|
||||
const MV_REFERENCE_FRAME edge0 =
|
||||
!CHECK_BACKWARD_REFS(above0) ? left0 : above0;
|
||||
pred_context = 4 * (edge0 == BWDREF_FRAME);
|
||||
} else {
|
||||
pred_context =
|
||||
2 * (above0 == BWDREF_FRAME) + 2 * (left0 == BWDREF_FRAME);
|
||||
}
|
||||
}
|
||||
}
|
||||
} else if (has_above || has_left) { // one edge available
|
||||
const MB_MODE_INFO *edge_mbmi = has_above ? above_mbmi : left_mbmi;
|
||||
|
||||
if (!is_inter_block(edge_mbmi) ||
|
||||
(!CHECK_BACKWARD_REFS(edge_mbmi->ref_frame[0]) &&
|
||||
!has_second_ref(edge_mbmi)))
|
||||
pred_context = 2;
|
||||
else if (!has_second_ref(edge_mbmi)) // single
|
||||
pred_context = 4 * (edge_mbmi->ref_frame[0] == BWDREF_FRAME);
|
||||
else // comp
|
||||
pred_context = 3 * (edge_mbmi->ref_frame[0] == BWDREF_FRAME ||
|
||||
edge_mbmi->ref_frame[1] == BWDREF_FRAME);
|
||||
} else { // no edges available (2)
|
||||
pred_context = 2;
|
||||
}
|
||||
|
||||
assert(pred_context >= 0 && pred_context < REF_CONTEXTS);
|
||||
return pred_context;
|
||||
#endif // CONFIG_ALTREF2
|
||||
}
|
||||
|
||||
// For the bit to signal whether the single reference is LAST3/GOLDEN or
|
||||
|
|
@ -1640,13 +1421,11 @@ int av1_get_pred_context_single_ref_p5(const MACROBLOCKD *xd) {
|
|||
return pred_context;
|
||||
}
|
||||
|
||||
#if CONFIG_ALTREF2
|
||||
// For the bit to signal whether the single reference is ALTREF2_FRAME or
|
||||
// BWDREF_FRAME, knowing that it shall be either of these 2 choices.
|
||||
int av1_get_pred_context_single_ref_p6(const MACROBLOCKD *xd) {
|
||||
return av1_get_pred_context_brf_or_arf2(xd);
|
||||
}
|
||||
#endif // CONFIG_ALTREF2
|
||||
|
||||
#else // !CONFIG_EXT_REFS
|
||||
|
||||
|
|
|
|||
31
third_party/aom/av1/common/pred_common.h
vendored
31
third_party/aom/av1/common/pred_common.h
vendored
|
|
@ -86,14 +86,14 @@ int av1_get_pred_context_intra_interp(const MACROBLOCKD *xd);
|
|||
#endif // CONFIG_INTRA_INTERP
|
||||
#endif // CONFIG_EXT_INTRA
|
||||
|
||||
#if CONFIG_PALETTE && CONFIG_PALETTE_DELTA_ENCODING
|
||||
#if CONFIG_PALETTE_DELTA_ENCODING
|
||||
// Get a list of palette base colors that are used in the above and left blocks,
|
||||
// referred to as "color cache". The return value is the number of colors in the
|
||||
// cache (<= 2 * PALETTE_MAX_SIZE). The color values are stored in "cache"
|
||||
// in ascending order.
|
||||
int av1_get_palette_cache(const MODE_INFO *above_mi, const MODE_INFO *left_mi,
|
||||
int plane, uint16_t *cache);
|
||||
#endif // CONFIG_PALETTE && CONFIG_PALETTE_DELTA_ENCODING
|
||||
int av1_get_palette_cache(const MACROBLOCKD *const xd, int plane,
|
||||
uint16_t *cache);
|
||||
#endif // CONFIG_PALETTE_DELTA_ENCODING
|
||||
|
||||
int av1_get_intra_inter_context(const MACROBLOCKD *xd);
|
||||
|
||||
|
|
@ -243,17 +243,22 @@ static INLINE aom_prob av1_get_pred_prob_comp_bwdref_p(const AV1_COMMON *cm,
|
|||
return cm->fc->comp_bwdref_prob[pred_context][0];
|
||||
}
|
||||
|
||||
#if CONFIG_ALTREF2
|
||||
// TODO(zoeliu): ALTREF2 to work with NEW_MULTISYMBOL
|
||||
int av1_get_pred_context_comp_bwdref_p1(const AV1_COMMON *cm,
|
||||
const MACROBLOCKD *xd);
|
||||
|
||||
#if CONFIG_NEW_MULTISYMBOL
|
||||
static INLINE aom_cdf_prob *av1_get_pred_cdf_comp_bwdref_p1(
|
||||
const AV1_COMMON *cm, const MACROBLOCKD *xd) {
|
||||
const int pred_context = av1_get_pred_context_comp_bwdref_p1(cm, xd);
|
||||
return xd->tile_ctx->comp_bwdref_cdf[pred_context][1];
|
||||
}
|
||||
#endif // CONFIG_NEW_MULTISYMBOL
|
||||
|
||||
static INLINE aom_prob av1_get_pred_prob_comp_bwdref_p1(const AV1_COMMON *cm,
|
||||
const MACROBLOCKD *xd) {
|
||||
const int pred_context = av1_get_pred_context_comp_bwdref_p1(cm, xd);
|
||||
return cm->fc->comp_bwdref_prob[pred_context][1];
|
||||
}
|
||||
#endif // CONFIG_ALTREF2
|
||||
#endif // CONFIG_EXT_REFS
|
||||
|
||||
int av1_get_pred_context_single_ref_p1(const MACROBLOCKD *xd);
|
||||
|
|
@ -292,14 +297,12 @@ static INLINE aom_prob av1_get_pred_prob_single_ref_p5(const AV1_COMMON *cm,
|
|||
return cm->fc->single_ref_prob[av1_get_pred_context_single_ref_p5(xd)][4];
|
||||
}
|
||||
|
||||
#if CONFIG_ALTREF2
|
||||
int av1_get_pred_context_single_ref_p6(const MACROBLOCKD *xd);
|
||||
|
||||
static INLINE aom_prob av1_get_pred_prob_single_ref_p6(const AV1_COMMON *cm,
|
||||
const MACROBLOCKD *xd) {
|
||||
return cm->fc->single_ref_prob[av1_get_pred_context_single_ref_p6(xd)][5];
|
||||
}
|
||||
#endif // CONFIG_ALTREF2
|
||||
#endif // CONFIG_EXT_REFS
|
||||
|
||||
#if CONFIG_NEW_MULTISYMBOL
|
||||
|
|
@ -334,17 +337,23 @@ static INLINE aom_cdf_prob *av1_get_pred_cdf_single_ref_p5(
|
|||
return xd->tile_ctx
|
||||
->single_ref_cdf[av1_get_pred_context_single_ref_p5(xd)][4];
|
||||
}
|
||||
static INLINE aom_cdf_prob *av1_get_pred_cdf_single_ref_p6(
|
||||
const AV1_COMMON *cm, const MACROBLOCKD *xd) {
|
||||
(void)cm;
|
||||
return xd->tile_ctx
|
||||
->single_ref_cdf[av1_get_pred_context_single_ref_p6(xd)][5];
|
||||
}
|
||||
#endif // CONFIG_EXT_REFS
|
||||
#endif // CONFIG_NEW_MULTISYMBOL
|
||||
|
||||
#if CONFIG_EXT_INTER && CONFIG_COMPOUND_SINGLEREF
|
||||
#if CONFIG_COMPOUND_SINGLEREF
|
||||
int av1_get_inter_mode_context(const MACROBLOCKD *xd);
|
||||
|
||||
static INLINE aom_prob av1_get_inter_mode_prob(const AV1_COMMON *cm,
|
||||
const MACROBLOCKD *xd) {
|
||||
return cm->fc->comp_inter_mode_prob[av1_get_inter_mode_context(xd)];
|
||||
}
|
||||
#endif // CONFIG_EXT_INTER && CONFIG_COMPOUND_SINGLEREF
|
||||
#endif // CONFIG_COMPOUND_SINGLEREF
|
||||
|
||||
// Returns a context number for the given MB prediction signal
|
||||
// The mode info data structure has a one element border above and to the
|
||||
|
|
|
|||
4
third_party/aom/av1/common/pvq.c
vendored
4
third_party/aom/av1/common/pvq.c
vendored
|
|
@ -591,7 +591,7 @@ static int32_t od_pow(int32_t x, od_val16 beta)
|
|||
/*log2(g/OD_COMPAND_SCALE) = log2(x) - OD_COMPAND_SHIFT in
|
||||
Q(OD_LOG2_OUTSHIFT).*/
|
||||
logr = od_log2(t) + (log2_x - OD_COMPAND_SHIFT)*OD_LOG2_OUTSCALE;
|
||||
logr = OD_MULT16_32_QBETA(beta, logr);
|
||||
logr = (od_val32)OD_MULT16_32_QBETA(beta, logr);
|
||||
return od_exp2(logr);
|
||||
}
|
||||
#endif
|
||||
|
|
@ -974,7 +974,7 @@ void od_pvq_synthesis_partial(od_coeff *xcoeff, const od_coeff *ypulse,
|
|||
od_val32 x;
|
||||
/* This multiply doesn't round, so it introduces some bias.
|
||||
It would be nice (but not critical) to fix this. */
|
||||
x = OD_MULT16_32_Q16(ypulse[i], scale);
|
||||
x = (od_val32)OD_MULT16_32_Q16(ypulse[i], scale);
|
||||
#if defined(OD_FLOAT_PVQ)
|
||||
xcoeff[i] = (od_coeff)floor(.5
|
||||
+ x*(qm_inv[i]*OD_QM_INV_SCALE_1));
|
||||
|
|
|
|||
4
third_party/aom/av1/common/pvq.h
vendored
4
third_party/aom/av1/common/pvq.h
vendored
|
|
@ -19,11 +19,7 @@
|
|||
extern const uint16_t EXP_CDF_TABLE[][16];
|
||||
extern const uint16_t LAPLACE_OFFSET[];
|
||||
|
||||
#if CONFIG_DAALA_DIST
|
||||
#define AV1_PVQ_ENABLE_ACTIVITY_MASKING (1)
|
||||
#else
|
||||
#define AV1_PVQ_ENABLE_ACTIVITY_MASKING (0)
|
||||
#endif
|
||||
|
||||
# define PVQ_MAX_PARTITIONS (1 + 3*(OD_TXSIZES-1))
|
||||
|
||||
|
|
|
|||
25
third_party/aom/av1/common/quant_common.c
vendored
25
third_party/aom/av1/common/quant_common.c
vendored
|
|
@ -360,21 +360,28 @@ static uint16_t wt_matrix_ref[NUM_QM_LEVELS][2][QM_TOTAL_SIZE];
|
|||
static uint16_t iwt_matrix_ref[NUM_QM_LEVELS][2][QM_TOTAL_SIZE];
|
||||
|
||||
void aom_qm_init(AV1_COMMON *cm) {
|
||||
int q, c, f, t, size;
|
||||
int q, c, f, t;
|
||||
int current;
|
||||
for (q = 0; q < NUM_QM_LEVELS; ++q) {
|
||||
for (c = 0; c < 2; ++c) {
|
||||
for (f = 0; f < 2; ++f) {
|
||||
current = 0;
|
||||
for (t = 0; t < TX_SIZES_ALL; ++t) {
|
||||
size = tx_size_2d[t];
|
||||
cm->gqmatrix[q][c][f][t] = &wt_matrix_ref[AOMMIN(
|
||||
NUM_QM_LEVELS - 1, f == 0 ? q + DEFAULT_QM_INTER_OFFSET : q)][c]
|
||||
[current];
|
||||
cm->giqmatrix[q][c][f][t] = &iwt_matrix_ref[AOMMIN(
|
||||
NUM_QM_LEVELS - 1, f == 0 ? q + DEFAULT_QM_INTER_OFFSET : q)][c]
|
||||
const int size = tx_size_2d[t];
|
||||
// Don't use QM for sizes > 32x32
|
||||
if (q == NUM_QM_LEVELS - 1 || size > 1024) {
|
||||
cm->gqmatrix[q][c][f][t] = NULL;
|
||||
cm->giqmatrix[q][c][f][t] = NULL;
|
||||
} else {
|
||||
assert(current + size <= QM_TOTAL_SIZE);
|
||||
cm->gqmatrix[q][c][f][t] = &wt_matrix_ref[AOMMIN(
|
||||
NUM_QM_LEVELS - 1, f == 0 ? q + DEFAULT_QM_INTER_OFFSET : q)][c]
|
||||
[current];
|
||||
current += size;
|
||||
cm->giqmatrix[q][c][f][t] = &iwt_matrix_ref[AOMMIN(
|
||||
NUM_QM_LEVELS - 1, f == 0 ? q + DEFAULT_QM_INTER_OFFSET : q)][c]
|
||||
[current];
|
||||
current += size;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
|
@ -14039,7 +14046,7 @@ static uint16_t wt_matrix_ref[NUM_QM_LEVELS][2][QM_TOTAL_SIZE] = {
|
|||
};
|
||||
#endif
|
||||
|
||||
#if CONFIG_PVQ || CONFIG_DAALA_DIST
|
||||
#if CONFIG_PVQ
|
||||
/* Quantization matrices for 8x8. For other block sizes, we currently just do
|
||||
resampling. */
|
||||
/* Flat quantization, i.e. optimize for PSNR. */
|
||||
|
|
|
|||
6
third_party/aom/av1/common/quant_common.h
vendored
6
third_party/aom/av1/common/quant_common.h
vendored
|
|
@ -48,9 +48,7 @@ int av1_get_qindex(const struct segmentation *seg, int segment_id,
|
|||
// Reduce the large number of quantizers to a smaller number of levels for which
|
||||
// different matrices may be defined
|
||||
static INLINE int aom_get_qmlevel(int qindex, int first, int last) {
|
||||
int qmlevel = (qindex * (last + 1 - first) + QINDEX_RANGE / 2) / QINDEX_RANGE;
|
||||
qmlevel = AOMMIN(qmlevel + first, NUM_QM_LEVELS - 1);
|
||||
return qmlevel;
|
||||
return first + (qindex * (last + 1 - first)) / QINDEX_RANGE;
|
||||
}
|
||||
void aom_qm_init(struct AV1Common *cm);
|
||||
qm_val_t *aom_iqmatrix(struct AV1Common *cm, int qindex, int comp,
|
||||
|
|
@ -99,7 +97,7 @@ static INLINE int get_dq_profile_from_ctx(int qindex, int q_ctx, int is_inter,
|
|||
}
|
||||
#endif // CONFIG_NEW_QUANT
|
||||
|
||||
#if CONFIG_PVQ || CONFIG_DAALA_DIST
|
||||
#if CONFIG_PVQ
|
||||
extern const int OD_QM8_Q4_FLAT[];
|
||||
extern const int OD_QM8_Q4_HVS[];
|
||||
#endif
|
||||
|
|
|
|||
2199
third_party/aom/av1/common/reconinter.c
vendored
2199
third_party/aom/av1/common/reconinter.c
vendored
File diff suppressed because it is too large
Load diff
482
third_party/aom/av1/common/reconinter.h
vendored
482
third_party/aom/av1/common/reconinter.h
vendored
|
|
@ -40,34 +40,27 @@ static INLINE void inter_predictor(const uint8_t *src, int src_stride,
|
|||
uint8_t *dst, int dst_stride, int subpel_x,
|
||||
int subpel_y, const struct scale_factors *sf,
|
||||
int w, int h, ConvolveParams *conv_params,
|
||||
#if CONFIG_DUAL_FILTER
|
||||
const InterpFilter *interp_filter,
|
||||
#else
|
||||
const InterpFilter interp_filter,
|
||||
#endif
|
||||
int xs, int ys) {
|
||||
#if CONFIG_DUAL_FILTER
|
||||
const InterpFilter filter_x = av1_get_plane_interp_filter(
|
||||
interp_filter[1 + 2 * conv_params->ref], conv_params->plane);
|
||||
const InterpFilter filter_y = av1_get_plane_interp_filter(
|
||||
interp_filter[0 + 2 * conv_params->ref], conv_params->plane);
|
||||
const InterpFilterParams interp_filter_params_x =
|
||||
av1_get_interp_filter_params(filter_x);
|
||||
const InterpFilterParams interp_filter_params_y =
|
||||
av1_get_interp_filter_params(filter_y);
|
||||
#else
|
||||
const InterpFilterParams interp_filter_params_x =
|
||||
av1_get_interp_filter_params(interp_filter);
|
||||
const InterpFilterParams interp_filter_params_y = interp_filter_params_x;
|
||||
#endif
|
||||
|
||||
InterpFilters interp_filters, int xs,
|
||||
int ys) {
|
||||
assert(conv_params->do_average == 0 || conv_params->do_average == 1);
|
||||
assert(sf);
|
||||
if (has_scale(xs, ys)) {
|
||||
// TODO(afergs, debargha): Use a different scale convolve function
|
||||
// that uses higher precision for subpel_x, subpel_y, xs, ys
|
||||
av1_convolve_scale(src, src_stride, dst, dst_stride, w, h, interp_filter,
|
||||
subpel_x, xs, subpel_y, ys, conv_params);
|
||||
if (conv_params->round == CONVOLVE_OPT_NO_ROUND) {
|
||||
#if CONFIG_CONVOLVE_ROUND
|
||||
av1_convolve_2d_facade(src, src_stride, dst, dst_stride, w, h,
|
||||
interp_filters, subpel_x, xs, subpel_y, ys, 1,
|
||||
conv_params);
|
||||
conv_params->do_post_rounding = 1;
|
||||
#else
|
||||
assert(0);
|
||||
#endif // CONFIG_CONVOLVE_ROUND
|
||||
} else {
|
||||
assert(conv_params->round == CONVOLVE_OPT_ROUND);
|
||||
av1_convolve_scale(src, src_stride, dst, dst_stride, w, h, interp_filters,
|
||||
subpel_x, xs, subpel_y, ys, conv_params);
|
||||
}
|
||||
} else {
|
||||
subpel_x >>= SCALE_EXTRA_BITS;
|
||||
subpel_y >>= SCALE_EXTRA_BITS;
|
||||
|
|
@ -80,31 +73,32 @@ static INLINE void inter_predictor(const uint8_t *src, int src_stride,
|
|||
if (conv_params->round == CONVOLVE_OPT_NO_ROUND) {
|
||||
#if CONFIG_CONVOLVE_ROUND
|
||||
av1_convolve_2d_facade(src, src_stride, dst, dst_stride, w, h,
|
||||
#if CONFIG_DUAL_FILTER
|
||||
interp_filter,
|
||||
#else // CONFIG_DUAL_FILTER
|
||||
&interp_filter,
|
||||
#endif // CONFIG_DUAL_FILTER
|
||||
subpel_x, xs, subpel_y, ys, conv_params);
|
||||
interp_filters, subpel_x, xs, subpel_y, ys, 0,
|
||||
conv_params);
|
||||
conv_params->do_post_rounding = 1;
|
||||
#else
|
||||
assert(0);
|
||||
#endif // CONFIG_CONVOLVE_ROUND
|
||||
} else {
|
||||
assert(conv_params->round == CONVOLVE_OPT_ROUND);
|
||||
|
||||
InterpFilterParams filter_params_x, filter_params_y;
|
||||
av1_get_convolve_filter_params(interp_filters, 0, &filter_params_x,
|
||||
&filter_params_y);
|
||||
|
||||
if (w <= 2 || h <= 2) {
|
||||
av1_convolve_c(src, src_stride, dst, dst_stride, w, h, interp_filter,
|
||||
av1_convolve_c(src, src_stride, dst, dst_stride, w, h, interp_filters,
|
||||
subpel_x, xs, subpel_y, ys, conv_params);
|
||||
} else if (interp_filter_params_x.taps == SUBPEL_TAPS &&
|
||||
interp_filter_params_y.taps == SUBPEL_TAPS) {
|
||||
const int16_t *kernel_x = av1_get_interp_filter_subpel_kernel(
|
||||
interp_filter_params_x, subpel_x);
|
||||
const int16_t *kernel_y = av1_get_interp_filter_subpel_kernel(
|
||||
interp_filter_params_y, subpel_y);
|
||||
} else if (filter_params_x.taps == SUBPEL_TAPS &&
|
||||
filter_params_y.taps == SUBPEL_TAPS) {
|
||||
const int16_t *kernel_x =
|
||||
av1_get_interp_filter_subpel_kernel(filter_params_x, subpel_x);
|
||||
const int16_t *kernel_y =
|
||||
av1_get_interp_filter_subpel_kernel(filter_params_y, subpel_y);
|
||||
sf->predict[subpel_x != 0][subpel_y != 0][conv_params->do_average](
|
||||
src, src_stride, dst, dst_stride, kernel_x, xs, kernel_y, ys, w, h);
|
||||
} else {
|
||||
av1_convolve(src, src_stride, dst, dst_stride, w, h, interp_filter,
|
||||
av1_convolve(src, src_stride, dst, dst_stride, w, h, interp_filters,
|
||||
subpel_x, xs, subpel_y, ys, conv_params);
|
||||
}
|
||||
}
|
||||
|
|
@ -117,31 +111,26 @@ static INLINE void highbd_inter_predictor(const uint8_t *src, int src_stride,
|
|||
int subpel_x, int subpel_y,
|
||||
const struct scale_factors *sf, int w,
|
||||
int h, ConvolveParams *conv_params,
|
||||
#if CONFIG_DUAL_FILTER
|
||||
const InterpFilter *interp_filter,
|
||||
#else
|
||||
const InterpFilter interp_filter,
|
||||
#endif
|
||||
int xs, int ys, int bd) {
|
||||
InterpFilters interp_filters, int xs,
|
||||
int ys, int bd) {
|
||||
const int avg = conv_params->do_average;
|
||||
assert(avg == 0 || avg == 1);
|
||||
#if CONFIG_DUAL_FILTER
|
||||
const int ref = conv_params->ref;
|
||||
const InterpFilterParams interp_filter_params_x =
|
||||
av1_get_interp_filter_params(interp_filter[1 + 2 * ref]);
|
||||
const InterpFilterParams interp_filter_params_y =
|
||||
av1_get_interp_filter_params(interp_filter[0 + 2 * ref]);
|
||||
#else
|
||||
const InterpFilterParams interp_filter_params_x =
|
||||
av1_get_interp_filter_params(interp_filter);
|
||||
const InterpFilterParams interp_filter_params_y = interp_filter_params_x;
|
||||
#endif
|
||||
|
||||
if (has_scale(xs, ys)) {
|
||||
av1_highbd_convolve_scale(
|
||||
src, src_stride, dst, dst_stride, w, h, interp_filter,
|
||||
subpel_x >> SCALE_EXTRA_BITS, xs >> SCALE_EXTRA_BITS,
|
||||
subpel_y >> SCALE_EXTRA_BITS, ys >> SCALE_EXTRA_BITS, avg, bd);
|
||||
if (conv_params->round == CONVOLVE_OPT_NO_ROUND) {
|
||||
#if CONFIG_CONVOLVE_ROUND
|
||||
av1_highbd_convolve_2d_facade(src, src_stride, dst, dst_stride, w, h,
|
||||
interp_filters, subpel_x, xs, subpel_y, ys,
|
||||
1, conv_params, bd);
|
||||
conv_params->do_post_rounding = 1;
|
||||
#else
|
||||
assert(0);
|
||||
#endif // CONFIG_CONVOLVE_ROUND
|
||||
} else {
|
||||
av1_highbd_convolve_scale(src, src_stride, dst, dst_stride, w, h,
|
||||
interp_filters, subpel_x, xs, subpel_y, ys, avg,
|
||||
bd);
|
||||
}
|
||||
} else {
|
||||
subpel_x >>= SCALE_EXTRA_BITS;
|
||||
subpel_y >>= SCALE_EXTRA_BITS;
|
||||
|
|
@ -154,37 +143,36 @@ static INLINE void highbd_inter_predictor(const uint8_t *src, int src_stride,
|
|||
if (conv_params->round == CONVOLVE_OPT_NO_ROUND) {
|
||||
#if CONFIG_CONVOLVE_ROUND
|
||||
av1_highbd_convolve_2d_facade(src, src_stride, dst, dst_stride, w, h,
|
||||
#if CONFIG_DUAL_FILTER
|
||||
interp_filter,
|
||||
#else // CONFIG_DUAL_FILTER
|
||||
&interp_filter,
|
||||
#endif // CONFIG_DUAL_FILTER
|
||||
subpel_x, xs, subpel_y, ys, conv_params,
|
||||
bd);
|
||||
interp_filters, subpel_x, xs, subpel_y, ys,
|
||||
0, conv_params, bd);
|
||||
conv_params->do_post_rounding = 1;
|
||||
#else
|
||||
assert(0);
|
||||
#endif // CONFIG_CONVOLVE_ROUND
|
||||
} else {
|
||||
if (interp_filter_params_x.taps == SUBPEL_TAPS &&
|
||||
interp_filter_params_y.taps == SUBPEL_TAPS && w > 2 && h > 2) {
|
||||
const int16_t *kernel_x = av1_get_interp_filter_subpel_kernel(
|
||||
interp_filter_params_x, subpel_x);
|
||||
const int16_t *kernel_y = av1_get_interp_filter_subpel_kernel(
|
||||
interp_filter_params_y, subpel_y);
|
||||
InterpFilterParams filter_params_x, filter_params_y;
|
||||
av1_get_convolve_filter_params(interp_filters, 0, &filter_params_x,
|
||||
&filter_params_y);
|
||||
|
||||
if (filter_params_x.taps == SUBPEL_TAPS &&
|
||||
filter_params_y.taps == SUBPEL_TAPS && w > 2 && h > 2) {
|
||||
const int16_t *kernel_x =
|
||||
av1_get_interp_filter_subpel_kernel(filter_params_x, subpel_x);
|
||||
const int16_t *kernel_y =
|
||||
av1_get_interp_filter_subpel_kernel(filter_params_y, subpel_y);
|
||||
sf->highbd_predict[subpel_x != 0][subpel_y != 0][avg](
|
||||
src, src_stride, dst, dst_stride, kernel_x, xs, kernel_y, ys, w, h,
|
||||
bd);
|
||||
} else {
|
||||
av1_highbd_convolve(src, src_stride, dst, dst_stride, w, h,
|
||||
interp_filter, subpel_x, xs, subpel_y, ys, avg, bd);
|
||||
interp_filters, subpel_x, xs, subpel_y, ys, avg,
|
||||
bd);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
|
||||
#if CONFIG_EXT_INTER
|
||||
// Set to (1 << 5) if the 32-ary codebooks are used for any bock size
|
||||
#define MAX_WEDGE_TYPES (1 << 4)
|
||||
|
||||
|
|
@ -239,7 +227,8 @@ static INLINE int is_interinter_compound_used(COMPOUND_TYPE type,
|
|||
case COMPOUND_WEDGE: return wedge_params_lookup[sb_type].bits > 0;
|
||||
#endif // CONFIG_WEDGE
|
||||
#if CONFIG_COMPOUND_SEGMENT
|
||||
case COMPOUND_SEG: return sb_type >= BLOCK_8X8;
|
||||
case COMPOUND_SEG:
|
||||
return AOMMIN(block_size_wide[sb_type], block_size_high[sb_type]) >= 8;
|
||||
#endif // CONFIG_COMPOUND_SEGMENT
|
||||
default: assert(0); return 0;
|
||||
}
|
||||
|
|
@ -288,225 +277,20 @@ void build_compound_seg_mask_highbd(uint8_t *mask, SEG_MASK_TYPE mask_type,
|
|||
BLOCK_SIZE sb_type, int h, int w, int bd);
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
#endif // CONFIG_COMPOUND_SEGMENT
|
||||
#endif // CONFIG_EXT_INTER
|
||||
|
||||
void build_inter_predictors(const AV1_COMMON *cm, MACROBLOCKD *xd, int plane,
|
||||
#if CONFIG_MOTION_VAR
|
||||
int mi_col_offset, int mi_row_offset,
|
||||
#endif // CONFIG_MOTION_VAR
|
||||
int block, int bw, int bh, int x, int y, int w,
|
||||
int h,
|
||||
#if CONFIG_SUPERTX && CONFIG_EXT_INTER
|
||||
int wedge_offset_x, int wedge_offset_y,
|
||||
#endif // CONFIG_SUPERTX && CONFIG_EXT_INTER
|
||||
int mi_x, int mi_y);
|
||||
|
||||
#if CONFIG_GLOBAL_MOTION || CONFIG_WARPED_MOTION
|
||||
// This function will determine whether or not to create a warped
|
||||
// prediction and return the appropriate motion model depending
|
||||
// on the configuration. Behavior will change with different
|
||||
// combinations of GLOBAL_MOTION, WARPED_MOTION and MOTION_VAR.
|
||||
static INLINE int allow_warp(const MODE_INFO *const mi,
|
||||
const WarpTypesAllowed *const warp_types,
|
||||
#if CONFIG_GLOBAL_MOTION
|
||||
const WarpedMotionParams *const gm_params,
|
||||
#endif // CONFIG_GLOBAL_MOTION
|
||||
#if CONFIG_MOTION_VAR
|
||||
int mi_col_offset, int mi_row_offset,
|
||||
#endif // CONFIG_MOTION_VAR
|
||||
WarpedMotionParams *final_warp_params) {
|
||||
const MB_MODE_INFO *const mbmi = &mi->mbmi;
|
||||
set_default_warp_params(final_warp_params);
|
||||
|
||||
// Only global motion configured
|
||||
#if CONFIG_GLOBAL_MOTION && !CONFIG_WARPED_MOTION && !CONFIG_MOTION_VAR
|
||||
(void)mbmi;
|
||||
if (warp_types->global_warp_allowed) {
|
||||
memcpy(final_warp_params, gm_params, sizeof(*final_warp_params));
|
||||
return 1;
|
||||
}
|
||||
#endif // CONFIG_GLOBAL_MOTION && !CONFIG_WARPED_MOTION && !CONFIG_MOTION_VAR
|
||||
|
||||
// Only warped motion configured
|
||||
#if CONFIG_WARPED_MOTION && !CONFIG_GLOBAL_MOTION && !CONFIG_MOTION_VAR
|
||||
if (warp_types->local_warp_allowed) {
|
||||
memcpy(final_warp_params, &mbmi->wm_params[0], sizeof(*final_warp_params));
|
||||
return 1;
|
||||
}
|
||||
#endif // CONFIG_WARPED_MOTION && !CONFIG_GLOBAL_MOTION && !CONFIG_MOTION_VAR
|
||||
|
||||
// Warped and global motion configured
|
||||
#if CONFIG_GLOBAL_MOTION && CONFIG_WARPED_MOTION && !CONFIG_MOTION_VAR
|
||||
// When both are enabled, warped will take priority. The global parameters
|
||||
// will only be used to compute projection samples to find the warped model.
|
||||
// Note that when a block chooses global, it will not be possible to
|
||||
// select WARPED_CAUSAL.
|
||||
if (warp_types->local_warp_allowed) {
|
||||
memcpy(final_warp_params, &mbmi->wm_params[0], sizeof(*final_warp_params));
|
||||
return 1;
|
||||
} else if (warp_types->global_warp_allowed) {
|
||||
memcpy(final_warp_params, gm_params, sizeof(*final_warp_params));
|
||||
return 1;
|
||||
}
|
||||
#endif // CONFIG_GLOBAL_MOTION && CONFIG_WARPED_MOTION && !CONFIG_MOTION_VAR
|
||||
|
||||
// Motion var and global motion configured
|
||||
#if CONFIG_GLOBAL_MOTION && CONFIG_MOTION_VAR && !CONFIG_WARPED_MOTION
|
||||
// We warp if either case is true:
|
||||
// 1.) We are predicting a block which uses global motion
|
||||
// 2.) We are predicting a neighboring block of a block using OBMC,
|
||||
// the neighboring block uses global motion, and we have enabled
|
||||
// WARP_GM_NEIGHBORS_WITH_OBMC
|
||||
const int build_for_obmc = !(mi_col_offset == 0 && mi_row_offset == 0);
|
||||
(void)mbmi;
|
||||
if (warp_types->global_warp_allowed &&
|
||||
(WARP_GM_NEIGHBORS_WITH_OBMC || !build_for_obmc)) {
|
||||
memcpy(final_warp_params, gm_params, sizeof(*final_warp_params));
|
||||
return 1;
|
||||
}
|
||||
#endif // CONFIG_GLOBAL_MOTION && CONFIG_MOTION_VAR && !CONFIG_WARPED_MOTION
|
||||
|
||||
// Motion var and warped motion configured
|
||||
#if CONFIG_WARPED_MOTION && CONFIG_MOTION_VAR && !CONFIG_GLOBAL_MOTION
|
||||
// We warp if either case is true:
|
||||
// 1.) We are predicting a block with motion mode WARPED_CAUSAL
|
||||
// 2.) We are predicting a neighboring block of a block using OBMC,
|
||||
// the neighboring block has mode WARPED_CAUSAL, and we have enabled
|
||||
// WARP_WM_NEIGHBORS_WITH_OBMC
|
||||
const int build_for_obmc = !(mi_col_offset == 0 && mi_row_offset == 0);
|
||||
if (warp_types->local_warp_allowed) {
|
||||
if ((build_for_obmc && WARP_WM_NEIGHBORS_WITH_OBMC) || (!build_for_obmc)) {
|
||||
memcpy(final_warp_params, &mbmi->wm_params[0],
|
||||
sizeof(*final_warp_params));
|
||||
return 1;
|
||||
}
|
||||
}
|
||||
#endif // CONFIG_WARPED_MOTION && CONFIG_MOTION_VAR && !CONFIG_GLOBAL_MOTION
|
||||
|
||||
// Motion var, warped motion and global motion all configured
|
||||
#if CONFIG_WARPED_MOTION && CONFIG_MOTION_VAR && CONFIG_GLOBAL_MOTION
|
||||
const int build_for_obmc = !(mi_col_offset == 0 && mi_row_offset == 0);
|
||||
if (warp_types->local_warp_allowed) {
|
||||
if ((build_for_obmc && WARP_WM_NEIGHBORS_WITH_OBMC) || (!build_for_obmc)) {
|
||||
memcpy(final_warp_params, &mbmi->wm_params[0],
|
||||
sizeof(*final_warp_params));
|
||||
return 1;
|
||||
}
|
||||
} else if (warp_types->global_warp_allowed &&
|
||||
(WARP_GM_NEIGHBORS_WITH_OBMC || !build_for_obmc)) {
|
||||
memcpy(final_warp_params, gm_params, sizeof(*final_warp_params));
|
||||
return 1;
|
||||
}
|
||||
#endif // CONFIG_WARPED_MOTION && CONFIG_MOTION_VAR && CONFIG_GLOBAL_MOTION
|
||||
|
||||
return 0;
|
||||
}
|
||||
#endif // CONFIG_GLOBAL_MOTION ||CONFIG_WARPED_MOTION
|
||||
|
||||
static INLINE void av1_make_inter_predictor(
|
||||
const uint8_t *src, int src_stride, uint8_t *dst, int dst_stride,
|
||||
void av1_make_masked_inter_predictor(
|
||||
const uint8_t *pre, int pre_stride, uint8_t *dst, int dst_stride,
|
||||
const int subpel_x, const int subpel_y, const struct scale_factors *sf,
|
||||
int w, int h, ConvolveParams *conv_params,
|
||||
#if CONFIG_DUAL_FILTER
|
||||
const InterpFilter *interp_filter,
|
||||
#else
|
||||
const InterpFilter interp_filter,
|
||||
#endif
|
||||
#if CONFIG_GLOBAL_MOTION || CONFIG_WARPED_MOTION
|
||||
const WarpTypesAllowed *warp_types, int p_col, int p_row, int plane,
|
||||
int ref,
|
||||
#endif // CONFIG_GLOBAL_MOTION || CONFIG_WARPED_MOTION
|
||||
#if CONFIG_MOTION_VAR
|
||||
int mi_col_offset, int mi_row_offset,
|
||||
#endif
|
||||
int xs, int ys, const MACROBLOCKD *xd) {
|
||||
(void)xd;
|
||||
|
||||
#if CONFIG_MOTION_VAR
|
||||
const MODE_INFO *mi = xd->mi[mi_col_offset + xd->mi_stride * mi_row_offset];
|
||||
#else
|
||||
const MODE_INFO *mi = xd->mi[0];
|
||||
(void)mi;
|
||||
#endif // CONFIG_MOTION_VAR
|
||||
|
||||
// Make sure the selected motion mode is valid for this configuration
|
||||
#if CONFIG_MOTION_VAR || CONFIG_WARPED_MOTION
|
||||
assert_motion_mode_valid(mi->mbmi.motion_mode,
|
||||
#if CONFIG_GLOBAL_MOTION
|
||||
0, xd->global_motion,
|
||||
#endif // CONFIG_GLOBAL_MOTION
|
||||
#if CONFIG_WARPED_MOTION
|
||||
xd,
|
||||
#endif
|
||||
mi);
|
||||
#endif // CONFIG MOTION_VAR || CONFIG_WARPED_MOTION
|
||||
|
||||
#if CONFIG_WARPED_MOTION || CONFIG_GLOBAL_MOTION
|
||||
WarpedMotionParams final_warp_params;
|
||||
const int do_warp = allow_warp(
|
||||
mi, warp_types,
|
||||
#if CONFIG_GLOBAL_MOTION
|
||||
#if CONFIG_EXT_INTER && CONFIG_COMPOUND_SINGLEREF
|
||||
// TODO(zoeliu): To further check the single
|
||||
// ref comp mode to work together with
|
||||
// global motion.
|
||||
has_second_ref(&mi->mbmi) ? &xd->global_motion[mi->mbmi.ref_frame[ref]]
|
||||
: &xd->global_motion[mi->mbmi.ref_frame[0]],
|
||||
#else // !(CONFIG_EXT_INTER && CONFIG_COMPOUND_SINGLEREF)
|
||||
&xd->global_motion[mi->mbmi.ref_frame[ref]],
|
||||
#endif // CONFIG_EXT_INTER && CONFIG_COMPOUND_SINGLEREF
|
||||
#endif // CONFIG_GLOBAL_MOTION
|
||||
#if CONFIG_MOTION_VAR
|
||||
mi_col_offset, mi_row_offset,
|
||||
#endif // CONFIG_MOTION_VAR
|
||||
&final_warp_params);
|
||||
if (do_warp) {
|
||||
const struct macroblockd_plane *const pd = &xd->plane[plane];
|
||||
const struct buf_2d *const pre_buf = &pd->pre[ref];
|
||||
av1_warp_plane(&final_warp_params,
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
xd->cur_buf->flags & YV12_FLAG_HIGHBITDEPTH, xd->bd,
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
pre_buf->buf0, pre_buf->width, pre_buf->height,
|
||||
pre_buf->stride, dst, p_col, p_row, w, h, dst_stride,
|
||||
pd->subsampling_x, pd->subsampling_y, xs, ys, conv_params);
|
||||
return;
|
||||
}
|
||||
#endif // CONFIG_GLOBAL_MOTION || CONFIG_WARPED_MOTION
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
if (xd->cur_buf->flags & YV12_FLAG_HIGHBITDEPTH) {
|
||||
highbd_inter_predictor(src, src_stride, dst, dst_stride, subpel_x, subpel_y,
|
||||
sf, w, h, conv_params, interp_filter, xs, ys,
|
||||
xd->bd);
|
||||
return;
|
||||
}
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
inter_predictor(src, src_stride, dst, dst_stride, subpel_x, subpel_y, sf, w,
|
||||
h, conv_params, interp_filter, xs, ys);
|
||||
}
|
||||
|
||||
#if CONFIG_EXT_INTER
|
||||
void av1_make_masked_inter_predictor(const uint8_t *pre, int pre_stride,
|
||||
uint8_t *dst, int dst_stride,
|
||||
const int subpel_x, const int subpel_y,
|
||||
const struct scale_factors *sf, int w,
|
||||
int h, ConvolveParams *conv_params,
|
||||
#if CONFIG_DUAL_FILTER
|
||||
const InterpFilter *interp_filter,
|
||||
#else
|
||||
const InterpFilter interp_filter,
|
||||
#endif
|
||||
int xs, int ys,
|
||||
int w, int h, ConvolveParams *conv_params, InterpFilters interp_filters,
|
||||
int xs, int ys,
|
||||
#if CONFIG_SUPERTX
|
||||
int wedge_offset_x, int wedge_offset_y,
|
||||
int wedge_offset_x, int wedge_offset_y,
|
||||
#endif // CONFIG_SUPERTX
|
||||
int plane,
|
||||
int plane,
|
||||
#if CONFIG_GLOBAL_MOTION || CONFIG_WARPED_MOTION
|
||||
const WarpTypesAllowed *warp_types,
|
||||
int p_col, int p_row, int ref,
|
||||
const WarpTypesAllowed *warp_types, int p_col, int p_row, int ref,
|
||||
#endif // CONFIG_GLOBAL_MOTION || CONFIG_WARPED_MOTION
|
||||
MACROBLOCKD *xd);
|
||||
#endif // CONFIG_EXT_INTER
|
||||
MACROBLOCKD *xd);
|
||||
|
||||
static INLINE int round_mv_comp_q4(int value) {
|
||||
return (value < 0 ? value - 2 : value + 2) / 4;
|
||||
|
|
@ -588,18 +372,13 @@ void av1_build_inter_predictors_sb(const AV1_COMMON *cm, MACROBLOCKD *xd,
|
|||
|
||||
#if CONFIG_SUPERTX
|
||||
void av1_build_inter_predictor_sb_sub8x8_extend(const AV1_COMMON *cm,
|
||||
MACROBLOCKD *xd,
|
||||
#if CONFIG_EXT_INTER
|
||||
int mi_row_ori, int mi_col_ori,
|
||||
#endif // CONFIG_EXT_INTER
|
||||
int mi_row, int mi_col,
|
||||
int plane, BLOCK_SIZE bsize,
|
||||
int block);
|
||||
MACROBLOCKD *xd, int mi_row_ori,
|
||||
int mi_col_ori, int mi_row,
|
||||
int mi_col, int plane,
|
||||
BLOCK_SIZE bsize, int block);
|
||||
|
||||
void av1_build_inter_predictor_sb_extend(const AV1_COMMON *cm, MACROBLOCKD *xd,
|
||||
#if CONFIG_EXT_INTER
|
||||
int mi_row_ori, int mi_col_ori,
|
||||
#endif // CONFIG_EXT_INTER
|
||||
int mi_row, int mi_col, int plane,
|
||||
BLOCK_SIZE bsize);
|
||||
struct macroblockd_plane;
|
||||
|
|
@ -614,11 +393,7 @@ void av1_build_inter_predictor(const uint8_t *src, int src_stride, uint8_t *dst,
|
|||
int dst_stride, const MV *src_mv,
|
||||
const struct scale_factors *sf, int w, int h,
|
||||
ConvolveParams *conv_params,
|
||||
#if CONFIG_DUAL_FILTER
|
||||
const InterpFilter *interp_filter,
|
||||
#else
|
||||
const InterpFilter interp_filter,
|
||||
#endif
|
||||
InterpFilters interp_filters,
|
||||
#if CONFIG_GLOBAL_MOTION || CONFIG_WARPED_MOTION
|
||||
const WarpTypesAllowed *warp_types, int p_col,
|
||||
int p_row, int plane, int ref,
|
||||
|
|
@ -630,11 +405,7 @@ void av1_build_inter_predictor(const uint8_t *src, int src_stride, uint8_t *dst,
|
|||
void av1_highbd_build_inter_predictor(
|
||||
const uint8_t *src, int src_stride, uint8_t *dst, int dst_stride,
|
||||
const MV *mv_q3, const struct scale_factors *sf, int w, int h, int do_avg,
|
||||
#if CONFIG_DUAL_FILTER
|
||||
const InterpFilter *interp_filter,
|
||||
#else
|
||||
const InterpFilter interp_filter,
|
||||
#endif
|
||||
InterpFilters interp_filters,
|
||||
#if CONFIG_GLOBAL_MOTION || CONFIG_WARPED_MOTION
|
||||
const WarpTypesAllowed *warp_types, int p_col, int p_row,
|
||||
#endif // CONFIG_GLOBAL_MOTION || CONFIG_WARPED_MOTION
|
||||
|
|
@ -657,11 +428,11 @@ static INLINE void setup_pred_plane(struct buf_2d *dst, BLOCK_SIZE bsize,
|
|||
const struct scale_factors *scale,
|
||||
int subsampling_x, int subsampling_y) {
|
||||
#if CONFIG_CHROMA_SUB8X8
|
||||
if (bsize < BLOCK_8X8) {
|
||||
// Offset the buffer pointer
|
||||
if (subsampling_y && (mi_row & 0x01)) mi_row -= 1;
|
||||
if (subsampling_x && (mi_col & 0x01)) mi_col -= 1;
|
||||
}
|
||||
// Offset the buffer pointer
|
||||
if (subsampling_y && (mi_row & 0x01) && (mi_size_high[bsize] == 1))
|
||||
mi_row -= 1;
|
||||
if (subsampling_x && (mi_col & 0x01) && (mi_size_wide[bsize] == 1))
|
||||
mi_col -= 1;
|
||||
#else
|
||||
(void)bsize;
|
||||
#endif
|
||||
|
|
@ -740,16 +511,8 @@ static INLINE int has_subpel_mv_component(const MODE_INFO *const mi,
|
|||
|
||||
static INLINE void set_default_interp_filters(
|
||||
MB_MODE_INFO *const mbmi, InterpFilter frame_interp_filter) {
|
||||
#if CONFIG_DUAL_FILTER
|
||||
int dir;
|
||||
for (dir = 0; dir < 4; ++dir)
|
||||
mbmi->interp_filter[dir] = frame_interp_filter == SWITCHABLE
|
||||
? EIGHTTAP_REGULAR
|
||||
: frame_interp_filter;
|
||||
#else
|
||||
mbmi->interp_filter = frame_interp_filter == SWITCHABLE ? EIGHTTAP_REGULAR
|
||||
: frame_interp_filter;
|
||||
#endif // CONFIG_DUAL_FILTER
|
||||
mbmi->interp_filters =
|
||||
av1_broadcast_interp_filter(av1_unswitchable_filter(frame_interp_filter));
|
||||
}
|
||||
|
||||
static INLINE int av1_is_interp_needed(const MACROBLOCKD *const xd) {
|
||||
|
|
@ -810,7 +573,6 @@ void av1_build_ncobmc_inter_predictors_sb(const AV1_COMMON *cm, MACROBLOCKD *xd,
|
|||
#endif
|
||||
#endif // CONFIG_MOTION_VAR
|
||||
|
||||
#if CONFIG_EXT_INTER
|
||||
#define MASK_MASTER_SIZE ((MAX_WEDGE_SIZE) << 1)
|
||||
#define MASK_MASTER_STRIDE (MASK_MASTER_SIZE)
|
||||
|
||||
|
|
@ -836,26 +598,26 @@ const uint8_t *av1_get_compound_type_mask_inverse(
|
|||
const uint8_t *av1_get_compound_type_mask(
|
||||
const INTERINTER_COMPOUND_DATA *const comp_data, BLOCK_SIZE sb_type);
|
||||
#if CONFIG_INTERINTRA
|
||||
void av1_build_interintra_predictors(MACROBLOCKD *xd, uint8_t *ypred,
|
||||
uint8_t *upred, uint8_t *vpred,
|
||||
int ystride, int ustride, int vstride,
|
||||
BUFFER_SET *ctx, BLOCK_SIZE bsize);
|
||||
void av1_build_interintra_predictors_sby(MACROBLOCKD *xd, uint8_t *ypred,
|
||||
int ystride, BUFFER_SET *ctx,
|
||||
void av1_build_interintra_predictors(const AV1_COMMON *cm, MACROBLOCKD *xd,
|
||||
uint8_t *ypred, uint8_t *upred,
|
||||
uint8_t *vpred, int ystride, int ustride,
|
||||
int vstride, BUFFER_SET *ctx,
|
||||
BLOCK_SIZE bsize);
|
||||
void av1_build_interintra_predictors_sby(const AV1_COMMON *cm, MACROBLOCKD *xd,
|
||||
uint8_t *ypred, int ystride,
|
||||
BUFFER_SET *ctx, BLOCK_SIZE bsize);
|
||||
void av1_build_interintra_predictors_sbc(const AV1_COMMON *cm, MACROBLOCKD *xd,
|
||||
uint8_t *upred, int ustride,
|
||||
BUFFER_SET *ctx, int plane,
|
||||
BLOCK_SIZE bsize);
|
||||
void av1_build_interintra_predictors_sbc(MACROBLOCKD *xd, uint8_t *upred,
|
||||
int ustride, BUFFER_SET *ctx,
|
||||
int plane, BLOCK_SIZE bsize);
|
||||
void av1_build_interintra_predictors_sbuv(MACROBLOCKD *xd, uint8_t *upred,
|
||||
uint8_t *vpred, int ustride,
|
||||
int vstride, BUFFER_SET *ctx,
|
||||
BLOCK_SIZE bsize);
|
||||
void av1_build_interintra_predictors_sbuv(const AV1_COMMON *cm, MACROBLOCKD *xd,
|
||||
uint8_t *upred, uint8_t *vpred,
|
||||
int ustride, int vstride,
|
||||
BUFFER_SET *ctx, BLOCK_SIZE bsize);
|
||||
|
||||
void av1_build_intra_predictors_for_interintra(MACROBLOCKD *xd,
|
||||
BLOCK_SIZE bsize, int plane,
|
||||
BUFFER_SET *ctx,
|
||||
uint8_t *intra_pred,
|
||||
int intra_stride);
|
||||
void av1_build_intra_predictors_for_interintra(
|
||||
const AV1_COMMON *cm, MACROBLOCKD *xd, BLOCK_SIZE bsize, int plane,
|
||||
BUFFER_SET *ctx, uint8_t *intra_pred, int intra_stride);
|
||||
void av1_combine_interintra(MACROBLOCKD *xd, BLOCK_SIZE bsize, int plane,
|
||||
const uint8_t *inter_pred, int inter_stride,
|
||||
const uint8_t *intra_pred, int intra_stride);
|
||||
|
|
@ -871,7 +633,45 @@ void av1_build_wedge_inter_predictor_from_buf(
|
|||
#endif // CONFIG_SUPERTX
|
||||
uint8_t *ext_dst0[3], int ext_dst_stride0[3], uint8_t *ext_dst1[3],
|
||||
int ext_dst_stride1[3]);
|
||||
#endif // CONFIG_EXT_INTER
|
||||
|
||||
#if CONFIG_NCOBMC_ADAPT_WEIGHT
|
||||
#define ASSIGN_ALIGNED_PTRS(p, a, s) \
|
||||
p[0] = a; \
|
||||
p[1] = a + s; \
|
||||
p[2] = a + 2 * s;
|
||||
|
||||
#define ASSIGN_ALIGNED_PTRS_HBD(p, a, s, l) \
|
||||
p[0] = CONVERT_TO_BYTEPTR(a); \
|
||||
p[1] = CONVERT_TO_BYTEPTR(a + s * l); \
|
||||
p[2] = CONVERT_TO_BYTEPTR(a + 2 * s * l);
|
||||
|
||||
void alloc_ncobmc_pred_buffer(MACROBLOCKD *const xd);
|
||||
void free_ncobmc_pred_buffer(MACROBLOCKD *const xd);
|
||||
void set_sb_mi_boundaries(const AV1_COMMON *const cm, MACROBLOCKD *const xd,
|
||||
const int mi_row, const int mi_col);
|
||||
|
||||
void reset_xd_boundary(MACROBLOCKD *xd, int mi_row, int bh, int mi_col, int bw,
|
||||
int mi_rows, int mi_cols);
|
||||
|
||||
void get_pred_from_intrpl_buf(MACROBLOCKD *xd, int mi_row, int mi_col,
|
||||
BLOCK_SIZE bsize, int plane);
|
||||
|
||||
void build_ncobmc_intrpl_pred(const AV1_COMMON *const cm, MACROBLOCKD *xd,
|
||||
int plane, int pxl_row, int pxl_col,
|
||||
BLOCK_SIZE bsize, uint8_t *preds[][MAX_MB_PLANE],
|
||||
int ps[MAX_MB_PLANE], // pred buffer strides
|
||||
int mode);
|
||||
|
||||
void av1_get_ext_blk_preds(const AV1_COMMON *cm, MACROBLOCKD *xd, int bsize,
|
||||
int mi_row, int mi_col,
|
||||
uint8_t *dst_buf[][MAX_MB_PLANE],
|
||||
int dst_stride[MAX_MB_PLANE]);
|
||||
|
||||
void av1_get_ori_blk_pred(const AV1_COMMON *cm, MACROBLOCKD *xd, int bsize,
|
||||
int mi_row, int mi_col,
|
||||
uint8_t *dst_buf[MAX_MB_PLANE],
|
||||
int dst_stride[MAX_MB_PLANE]);
|
||||
#endif // CONFIG_NCOBMC_ADAPT_WEIGHT
|
||||
|
||||
#ifdef __cplusplus
|
||||
} // extern "C"
|
||||
|
|
|
|||
722
third_party/aom/av1/common/reconintra.c
vendored
722
third_party/aom/av1/common/reconintra.c
vendored
File diff suppressed because it is too large
Load diff
62
third_party/aom/av1/common/reconintra.h
vendored
62
third_party/aom/av1/common/reconintra.h
vendored
|
|
@ -14,60 +14,34 @@
|
|||
|
||||
#include "aom/aom_integer.h"
|
||||
#include "av1/common/blockd.h"
|
||||
#include "av1/common/onyxc_int.h"
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
|
||||
#if CONFIG_DPCM_INTRA
|
||||
static INLINE int av1_use_dpcm_intra(int plane, PREDICTION_MODE mode,
|
||||
TX_TYPE tx_type,
|
||||
const MB_MODE_INFO *const mbmi) {
|
||||
(void)mbmi;
|
||||
(void)plane;
|
||||
#if CONFIG_EXT_INTRA
|
||||
if (mbmi->sb_type >= BLOCK_8X8 && mbmi->angle_delta[plane != 0]) return 0;
|
||||
#endif // CONFIG_EXT_INTRA
|
||||
return (mode == V_PRED && (tx_type == IDTX || tx_type == H_DCT)) ||
|
||||
(mode == H_PRED && (tx_type == IDTX || tx_type == V_DCT));
|
||||
}
|
||||
#endif // CONFIG_DPCM_INTRA
|
||||
|
||||
void av1_init_intra_predictors(void);
|
||||
void av1_predict_intra_block_facade(MACROBLOCKD *xd, int plane, int block_idx,
|
||||
int blk_col, int blk_row, TX_SIZE tx_size);
|
||||
void av1_predict_intra_block(const MACROBLOCKD *xd, int bw, int bh,
|
||||
BLOCK_SIZE bsize, PREDICTION_MODE mode,
|
||||
const uint8_t *ref, int ref_stride, uint8_t *dst,
|
||||
int dst_stride, int aoff, int loff, int plane);
|
||||
void av1_predict_intra_block_facade(const AV1_COMMON *cm, MACROBLOCKD *xd,
|
||||
int plane, int block_idx, int blk_col,
|
||||
int blk_row, TX_SIZE tx_size);
|
||||
void av1_predict_intra_block(const AV1_COMMON *cm, const MACROBLOCKD *xd,
|
||||
int bw, int bh, BLOCK_SIZE bsize,
|
||||
PREDICTION_MODE mode, const uint8_t *ref,
|
||||
int ref_stride, uint8_t *dst, int dst_stride,
|
||||
int aoff, int loff, int plane);
|
||||
|
||||
#if CONFIG_EXT_INTER && CONFIG_INTERINTRA
|
||||
#if CONFIG_INTERINTRA
|
||||
// Mapping of interintra to intra mode for use in the intra component
|
||||
static const PREDICTION_MODE interintra_to_intra_mode[INTERINTRA_MODES] = {
|
||||
DC_PRED, V_PRED, H_PRED,
|
||||
#if CONFIG_ALT_INTRA
|
||||
SMOOTH_PRED
|
||||
#else
|
||||
TM_PRED
|
||||
#endif
|
||||
DC_PRED, V_PRED, H_PRED, SMOOTH_PRED
|
||||
};
|
||||
|
||||
// Mapping of intra mode to the interintra mode
|
||||
static const INTERINTRA_MODE intra_to_interintra_mode[INTRA_MODES] = {
|
||||
II_DC_PRED, II_V_PRED, II_H_PRED, II_V_PRED,
|
||||
#if CONFIG_ALT_INTRA
|
||||
II_SMOOTH_PRED,
|
||||
#else
|
||||
II_TM_PRED,
|
||||
#endif
|
||||
II_V_PRED, II_H_PRED, II_H_PRED, II_V_PRED,
|
||||
#if CONFIG_ALT_INTRA
|
||||
II_SMOOTH_PRED, II_SMOOTH_PRED
|
||||
#else
|
||||
II_TM_PRED
|
||||
#endif
|
||||
II_DC_PRED, II_V_PRED, II_H_PRED, II_V_PRED, II_SMOOTH_PRED, II_V_PRED,
|
||||
II_H_PRED, II_H_PRED, II_V_PRED, II_SMOOTH_PRED, II_SMOOTH_PRED
|
||||
};
|
||||
#endif // CONFIG_EXT_INTER && CONFIG_INTERINTRA
|
||||
#endif // CONFIG_INTERINTRA
|
||||
|
||||
#if CONFIG_FILTER_INTRA
|
||||
#define FILTER_INTRA_PREC_BITS 10
|
||||
|
|
@ -97,6 +71,14 @@ static INLINE int av1_use_angle_delta(BLOCK_SIZE bsize) {
|
|||
}
|
||||
#endif // CONFIG_EXT_INTRA
|
||||
|
||||
#if CONFIG_INTRABC
|
||||
static INLINE int av1_allow_intrabc(BLOCK_SIZE bsize,
|
||||
const AV1_COMMON *const cm) {
|
||||
return (bsize >= BLOCK_8X8 || bsize == BLOCK_4X4) &&
|
||||
cm->allow_screen_content_tools;
|
||||
}
|
||||
#endif // CONFIG_INTRABC
|
||||
|
||||
#ifdef __cplusplus
|
||||
} // extern "C"
|
||||
#endif
|
||||
|
|
|
|||
614
third_party/aom/av1/common/resize.c
vendored
614
third_party/aom/av1/common/resize.c
vendored
|
|
@ -32,7 +32,7 @@
|
|||
#define INTERP_TAPS 8
|
||||
#define SUBPEL_BITS_RS 6
|
||||
#define SUBPEL_MASK_RS ((1 << SUBPEL_BITS_RS) - 1)
|
||||
#define INTERP_PRECISION_BITS 32
|
||||
#define INTERP_PRECISION_BITS 16
|
||||
#define SUBPEL_INTERP_EXTRA_BITS (INTERP_PRECISION_BITS - SUBPEL_BITS_RS)
|
||||
#define SUBPEL_INTERP_EXTRA_OFF (1 << (SUBPEL_INTERP_EXTRA_BITS - 1))
|
||||
|
||||
|
|
@ -40,24 +40,6 @@ typedef int16_t interp_kernel[INTERP_TAPS];
|
|||
|
||||
// Filters for interpolation (0.5-band) - note this also filters integer pels.
|
||||
static const interp_kernel filteredinterp_filters500[(1 << SUBPEL_BITS_RS)] = {
|
||||
#if SUBPEL_BITS_RS == 5
|
||||
{ -3, 0, 35, 64, 35, 0, -3, 0 }, { -3, -1, 34, 64, 36, 1, -3, 0 },
|
||||
{ -3, -1, 32, 64, 38, 1, -3, 0 }, { -2, -2, 31, 63, 39, 2, -3, 0 },
|
||||
{ -2, -2, 29, 63, 41, 2, -3, 0 }, { -2, -2, 28, 63, 42, 3, -4, 0 },
|
||||
{ -2, -3, 27, 63, 43, 4, -4, 0 }, { -2, -3, 25, 62, 45, 5, -4, 0 },
|
||||
{ -2, -3, 24, 62, 46, 5, -4, 0 }, { -2, -3, 23, 61, 47, 6, -4, 0 },
|
||||
{ -2, -3, 21, 60, 49, 7, -4, 0 }, { -1, -4, 20, 60, 50, 8, -4, -1 },
|
||||
{ -1, -4, 19, 59, 51, 9, -4, -1 }, { -1, -4, 17, 58, 52, 10, -4, 0 },
|
||||
{ -1, -4, 16, 57, 53, 12, -4, -1 }, { -1, -4, 15, 56, 54, 13, -4, -1 },
|
||||
{ -1, -4, 14, 55, 55, 14, -4, -1 }, { -1, -4, 13, 54, 56, 15, -4, -1 },
|
||||
{ -1, -4, 12, 53, 57, 16, -4, -1 }, { 0, -4, 10, 52, 58, 17, -4, -1 },
|
||||
{ -1, -4, 9, 51, 59, 19, -4, -1 }, { -1, -4, 8, 50, 60, 20, -4, -1 },
|
||||
{ 0, -4, 7, 49, 60, 21, -3, -2 }, { 0, -4, 6, 47, 61, 23, -3, -2 },
|
||||
{ 0, -4, 5, 46, 62, 24, -3, -2 }, { 0, -4, 5, 45, 62, 25, -3, -2 },
|
||||
{ 0, -4, 4, 43, 63, 27, -3, -2 }, { 0, -4, 3, 42, 63, 28, -2, -2 },
|
||||
{ 0, -3, 2, 41, 63, 29, -2, -2 }, { 0, -3, 2, 39, 63, 31, -2, -2 },
|
||||
{ 0, -3, 1, 38, 64, 32, -1, -3 }, { 0, -3, 1, 36, 64, 34, -1, -3 },
|
||||
#elif SUBPEL_BITS_RS == 6
|
||||
{ -3, 0, 35, 64, 35, 0, -3, 0 }, { -3, 0, 34, 64, 36, 0, -3, 0 },
|
||||
{ -3, -1, 34, 64, 36, 1, -3, 0 }, { -3, -1, 33, 64, 37, 1, -3, 0 },
|
||||
{ -3, -1, 32, 64, 38, 1, -3, 0 }, { -3, -1, 31, 64, 39, 1, -3, 0 },
|
||||
|
|
@ -90,29 +72,10 @@ static const interp_kernel filteredinterp_filters500[(1 << SUBPEL_BITS_RS)] = {
|
|||
{ 0, -3, 2, 39, 63, 31, -1, -3 }, { 0, -3, 1, 39, 64, 31, -1, -3 },
|
||||
{ 0, -3, 1, 38, 64, 32, -1, -3 }, { 0, -3, 1, 37, 64, 33, -1, -3 },
|
||||
{ 0, -3, 1, 36, 64, 34, -1, -3 }, { 0, -3, 0, 36, 64, 34, 0, -3 },
|
||||
#endif // SUBPEL_BITS_RS == 5
|
||||
};
|
||||
|
||||
// Filters for interpolation (0.625-band) - note this also filters integer pels.
|
||||
static const interp_kernel filteredinterp_filters625[(1 << SUBPEL_BITS_RS)] = {
|
||||
#if SUBPEL_BITS_RS == 5
|
||||
{ -1, -8, 33, 80, 33, -8, -1, 0 }, { -1, -8, 30, 80, 35, -8, -1, 1 },
|
||||
{ -1, -8, 28, 80, 37, -7, -2, 1 }, { 0, -8, 26, 79, 39, -7, -2, 1 },
|
||||
{ 0, -8, 24, 79, 41, -7, -2, 1 }, { 0, -8, 22, 78, 43, -6, -2, 1 },
|
||||
{ 0, -8, 20, 78, 45, -5, -3, 1 }, { 0, -8, 18, 77, 48, -5, -3, 1 },
|
||||
{ 0, -8, 16, 76, 50, -4, -3, 1 }, { 0, -8, 15, 75, 52, -3, -4, 1 },
|
||||
{ 0, -7, 13, 74, 54, -3, -4, 1 }, { 0, -7, 11, 73, 56, -2, -4, 1 },
|
||||
{ 0, -7, 10, 71, 58, -1, -4, 1 }, { 1, -7, 8, 70, 60, 0, -5, 1 },
|
||||
{ 1, -6, 6, 68, 62, 1, -5, 1 }, { 1, -6, 5, 67, 63, 2, -5, 1 },
|
||||
{ 1, -6, 4, 65, 65, 4, -6, 1 }, { 1, -5, 2, 63, 67, 5, -6, 1 },
|
||||
{ 1, -5, 1, 62, 68, 6, -6, 1 }, { 1, -5, 0, 60, 70, 8, -7, 1 },
|
||||
{ 1, -4, -1, 58, 71, 10, -7, 0 }, { 1, -4, -2, 56, 73, 11, -7, 0 },
|
||||
{ 1, -4, -3, 54, 74, 13, -7, 0 }, { 1, -4, -3, 52, 75, 15, -8, 0 },
|
||||
{ 1, -3, -4, 50, 76, 16, -8, 0 }, { 1, -3, -5, 48, 77, 18, -8, 0 },
|
||||
{ 1, -3, -5, 45, 78, 20, -8, 0 }, { 1, -2, -6, 43, 78, 22, -8, 0 },
|
||||
{ 1, -2, -7, 41, 79, 24, -8, 0 }, { 1, -2, -7, 39, 79, 26, -8, 0 },
|
||||
{ 1, -2, -7, 37, 80, 28, -8, -1 }, { 1, -1, -8, 35, 80, 30, -8, -1 },
|
||||
#elif SUBPEL_BITS_RS == 6
|
||||
{ -1, -8, 33, 80, 33, -8, -1, 0 }, { -1, -8, 31, 80, 34, -8, -1, 1 },
|
||||
{ -1, -8, 30, 80, 35, -8, -1, 1 }, { -1, -8, 29, 80, 36, -7, -2, 1 },
|
||||
{ -1, -8, 28, 80, 37, -7, -2, 1 }, { -1, -8, 27, 80, 38, -7, -2, 1 },
|
||||
|
|
@ -145,29 +108,10 @@ static const interp_kernel filteredinterp_filters625[(1 << SUBPEL_BITS_RS)] = {
|
|||
{ 1, -2, -7, 39, 79, 26, -8, 0 }, { 1, -2, -7, 38, 80, 27, -8, -1 },
|
||||
{ 1, -2, -7, 37, 80, 28, -8, -1 }, { 1, -2, -7, 36, 80, 29, -8, -1 },
|
||||
{ 1, -1, -8, 35, 80, 30, -8, -1 }, { 1, -1, -8, 34, 80, 31, -8, -1 },
|
||||
#endif // SUBPEL_BITS_RS == 5
|
||||
};
|
||||
|
||||
// Filters for interpolation (0.75-band) - note this also filters integer pels.
|
||||
static const interp_kernel filteredinterp_filters750[(1 << SUBPEL_BITS_RS)] = {
|
||||
#if SUBPEL_BITS_RS == 5
|
||||
{ 2, -11, 25, 96, 25, -11, 2, 0 }, { 2, -11, 22, 96, 28, -11, 2, 0 },
|
||||
{ 2, -10, 19, 95, 31, -11, 2, 0 }, { 2, -10, 17, 95, 34, -12, 2, 0 },
|
||||
{ 2, -9, 14, 94, 37, -12, 2, 0 }, { 2, -8, 12, 93, 40, -12, 1, 0 },
|
||||
{ 2, -8, 9, 92, 43, -12, 1, 1 }, { 2, -7, 7, 91, 46, -12, 1, 0 },
|
||||
{ 2, -7, 5, 90, 49, -12, 1, 0 }, { 2, -6, 3, 88, 52, -12, 0, 1 },
|
||||
{ 2, -5, 1, 86, 55, -12, 0, 1 }, { 2, -5, -1, 84, 58, -11, 0, 1 },
|
||||
{ 2, -4, -2, 82, 61, -11, -1, 1 }, { 2, -4, -4, 80, 64, -10, -1, 1 },
|
||||
{ 1, -3, -5, 77, 67, -9, -1, 1 }, { 1, -3, -6, 75, 70, -8, -2, 1 },
|
||||
{ 1, -2, -7, 72, 72, -7, -2, 1 }, { 1, -2, -8, 70, 75, -6, -3, 1 },
|
||||
{ 1, -1, -9, 67, 77, -5, -3, 1 }, { 1, -1, -10, 64, 80, -4, -4, 2 },
|
||||
{ 1, -1, -11, 61, 82, -2, -4, 2 }, { 1, 0, -11, 58, 84, -1, -5, 2 },
|
||||
{ 1, 0, -12, 55, 86, 1, -5, 2 }, { 1, 0, -12, 52, 88, 3, -6, 2 },
|
||||
{ 0, 1, -12, 49, 90, 5, -7, 2 }, { 0, 1, -12, 46, 91, 7, -7, 2 },
|
||||
{ 1, 1, -12, 43, 92, 9, -8, 2 }, { 0, 1, -12, 40, 93, 12, -8, 2 },
|
||||
{ 0, 2, -12, 37, 94, 14, -9, 2 }, { 0, 2, -12, 34, 95, 17, -10, 2 },
|
||||
{ 0, 2, -11, 31, 95, 19, -10, 2 }, { 0, 2, -11, 28, 96, 22, -11, 2 },
|
||||
#elif SUBPEL_BITS_RS == 6
|
||||
{ 2, -11, 25, 96, 25, -11, 2, 0 }, { 2, -11, 24, 96, 26, -11, 2, 0 },
|
||||
{ 2, -11, 22, 96, 28, -11, 2, 0 }, { 2, -10, 21, 96, 29, -12, 2, 0 },
|
||||
{ 2, -10, 19, 96, 31, -12, 2, 0 }, { 2, -10, 18, 95, 32, -11, 2, 0 },
|
||||
|
|
@ -200,29 +144,10 @@ static const interp_kernel filteredinterp_filters750[(1 << SUBPEL_BITS_RS)] = {
|
|||
{ 0, 2, -12, 34, 95, 17, -10, 2 }, { 0, 2, -11, 32, 95, 18, -10, 2 },
|
||||
{ 0, 2, -12, 31, 96, 19, -10, 2 }, { 0, 2, -12, 29, 96, 21, -10, 2 },
|
||||
{ 0, 2, -11, 28, 96, 22, -11, 2 }, { 0, 2, -11, 26, 96, 24, -11, 2 },
|
||||
#endif // SUBPEL_BITS_RS == 5
|
||||
};
|
||||
|
||||
// Filters for interpolation (0.875-band) - note this also filters integer pels.
|
||||
static const interp_kernel filteredinterp_filters875[(1 << SUBPEL_BITS_RS)] = {
|
||||
#if SUBPEL_BITS_RS == 5
|
||||
{ 3, -8, 13, 112, 13, -8, 3, 0 }, { 3, -7, 10, 112, 17, -9, 3, -1 },
|
||||
{ 2, -6, 7, 111, 21, -9, 3, -1 }, { 2, -5, 4, 111, 24, -10, 3, -1 },
|
||||
{ 2, -4, 1, 110, 28, -11, 3, -1 }, { 1, -3, -1, 108, 32, -12, 4, -1 },
|
||||
{ 1, -2, -3, 106, 36, -13, 4, -1 }, { 1, -1, -6, 105, 40, -14, 4, -1 },
|
||||
{ 1, -1, -7, 102, 44, -14, 4, -1 }, { 1, 0, -9, 100, 48, -15, 4, -1 },
|
||||
{ 1, 1, -11, 97, 53, -16, 4, -1 }, { 0, 1, -12, 95, 57, -16, 4, -1 },
|
||||
{ 0, 2, -13, 91, 61, -16, 4, -1 }, { 0, 2, -14, 88, 65, -16, 4, -1 },
|
||||
{ 0, 3, -15, 84, 69, -17, 4, 0 }, { 0, 3, -16, 81, 73, -16, 3, 0 },
|
||||
{ 0, 3, -16, 77, 77, -16, 3, 0 }, { 0, 3, -16, 73, 81, -16, 3, 0 },
|
||||
{ 0, 4, -17, 69, 84, -15, 3, 0 }, { -1, 4, -16, 65, 88, -14, 2, 0 },
|
||||
{ -1, 4, -16, 61, 91, -13, 2, 0 }, { -1, 4, -16, 57, 95, -12, 1, 0 },
|
||||
{ -1, 4, -16, 53, 97, -11, 1, 1 }, { -1, 4, -15, 48, 100, -9, 0, 1 },
|
||||
{ -1, 4, -14, 44, 102, -7, -1, 1 }, { -1, 4, -14, 40, 105, -6, -1, 1 },
|
||||
{ -1, 4, -13, 36, 106, -3, -2, 1 }, { -1, 4, -12, 32, 108, -1, -3, 1 },
|
||||
{ -1, 3, -11, 28, 110, 1, -4, 2 }, { -1, 3, -10, 24, 111, 4, -5, 2 },
|
||||
{ -1, 3, -9, 21, 111, 7, -6, 2 }, { -1, 3, -9, 17, 112, 10, -7, 3 },
|
||||
#elif SUBPEL_BITS_RS == 6
|
||||
{ 3, -8, 13, 112, 13, -8, 3, 0 }, { 2, -7, 12, 112, 15, -8, 3, -1 },
|
||||
{ 3, -7, 10, 112, 17, -9, 3, -1 }, { 2, -6, 8, 112, 19, -9, 3, -1 },
|
||||
{ 2, -6, 7, 112, 21, -10, 3, -1 }, { 2, -5, 6, 111, 22, -10, 3, -1 },
|
||||
|
|
@ -255,29 +180,10 @@ static const interp_kernel filteredinterp_filters875[(1 << SUBPEL_BITS_RS)] = {
|
|||
{ -1, 3, -10, 24, 111, 4, -5, 2 }, { -1, 3, -10, 22, 111, 6, -5, 2 },
|
||||
{ -1, 3, -10, 21, 112, 7, -6, 2 }, { -1, 3, -9, 19, 112, 8, -6, 2 },
|
||||
{ -1, 3, -9, 17, 112, 10, -7, 3 }, { -1, 3, -8, 15, 112, 12, -7, 2 },
|
||||
#endif // SUBPEL_BITS_RS == 5
|
||||
};
|
||||
|
||||
// Filters for interpolation (full-band) - no filtering for integer pixels
|
||||
static const interp_kernel filteredinterp_filters1000[(1 << SUBPEL_BITS_RS)] = {
|
||||
#if SUBPEL_BITS_RS == 5
|
||||
{ 0, 0, 0, 128, 0, 0, 0, 0 }, { 0, 1, -3, 128, 3, -1, 0, 0 },
|
||||
{ -1, 2, -6, 127, 7, -2, 1, 0 }, { -1, 3, -9, 126, 12, -4, 1, 0 },
|
||||
{ -1, 4, -12, 125, 16, -5, 1, 0 }, { -1, 4, -14, 123, 20, -6, 2, 0 },
|
||||
{ -1, 5, -15, 120, 25, -8, 2, 0 }, { -1, 5, -17, 118, 30, -9, 3, -1 },
|
||||
{ -1, 6, -18, 114, 35, -10, 3, -1 }, { -1, 6, -19, 111, 41, -12, 3, -1 },
|
||||
{ -1, 6, -20, 107, 46, -13, 4, -1 }, { -1, 6, -21, 103, 52, -14, 4, -1 },
|
||||
{ -1, 6, -21, 99, 57, -16, 5, -1 }, { -1, 6, -21, 94, 63, -17, 5, -1 },
|
||||
{ -1, 6, -20, 89, 68, -18, 5, -1 }, { -1, 6, -20, 84, 73, -19, 6, -1 },
|
||||
{ -1, 6, -20, 79, 79, -20, 6, -1 }, { -1, 6, -19, 73, 84, -20, 6, -1 },
|
||||
{ -1, 5, -18, 68, 89, -20, 6, -1 }, { -1, 5, -17, 63, 94, -21, 6, -1 },
|
||||
{ -1, 5, -16, 57, 99, -21, 6, -1 }, { -1, 4, -14, 52, 103, -21, 6, -1 },
|
||||
{ -1, 4, -13, 46, 107, -20, 6, -1 }, { -1, 3, -12, 41, 111, -19, 6, -1 },
|
||||
{ -1, 3, -10, 35, 114, -18, 6, -1 }, { -1, 3, -9, 30, 118, -17, 5, -1 },
|
||||
{ 0, 2, -8, 25, 120, -15, 5, -1 }, { 0, 2, -6, 20, 123, -14, 4, -1 },
|
||||
{ 0, 1, -5, 16, 125, -12, 4, -1 }, { 0, 1, -4, 12, 126, -9, 3, -1 },
|
||||
{ 0, 1, -2, 7, 127, -6, 2, -1 }, { 0, 0, -1, 3, 128, -3, 1, 0 },
|
||||
#elif SUBPEL_BITS_RS == 6
|
||||
{ 0, 0, 0, 128, 0, 0, 0, 0 }, { 0, 0, -1, 128, 2, -1, 0, 0 },
|
||||
{ 0, 1, -3, 127, 4, -2, 1, 0 }, { 0, 1, -4, 127, 6, -3, 1, 0 },
|
||||
{ 0, 2, -6, 126, 8, -3, 1, 0 }, { 0, 2, -7, 125, 11, -4, 1, 0 },
|
||||
|
|
@ -310,9 +216,86 @@ static const interp_kernel filteredinterp_filters1000[(1 << SUBPEL_BITS_RS)] = {
|
|||
{ 0, 2, -5, 13, 125, -8, 2, -1 }, { 0, 1, -4, 11, 125, -7, 2, 0 },
|
||||
{ 0, 1, -3, 8, 126, -6, 2, 0 }, { 0, 1, -3, 6, 127, -4, 1, 0 },
|
||||
{ 0, 1, -2, 4, 127, -3, 1, 0 }, { 0, 0, -1, 2, 128, -1, 0, 0 },
|
||||
#endif // SUBPEL_BITS_RS == 5
|
||||
};
|
||||
|
||||
#if CONFIG_FRAME_SUPERRES && CONFIG_LOOP_RESTORATION
|
||||
#define INTERP_SIMPLE_TAPS 4
|
||||
static const int16_t filter_simple[(1
|
||||
<< SUBPEL_BITS_RS)][INTERP_SIMPLE_TAPS] = {
|
||||
#if INTERP_SIMPLE_TAPS == 2
|
||||
{ 128, 0 }, { 126, 2 }, { 124, 4 }, { 122, 6 }, { 120, 8 }, { 118, 10 },
|
||||
{ 116, 12 }, { 114, 14 }, { 112, 16 }, { 110, 18 }, { 108, 20 }, { 106, 22 },
|
||||
{ 104, 24 }, { 102, 26 }, { 100, 28 }, { 98, 30 }, { 96, 32 }, { 94, 34 },
|
||||
{ 92, 36 }, { 90, 38 }, { 88, 40 }, { 86, 42 }, { 84, 44 }, { 82, 46 },
|
||||
{ 80, 48 }, { 78, 50 }, { 76, 52 }, { 74, 54 }, { 72, 56 }, { 70, 58 },
|
||||
{ 68, 60 }, { 66, 62 }, { 64, 64 }, { 62, 66 }, { 60, 68 }, { 58, 70 },
|
||||
{ 56, 72 }, { 54, 74 }, { 52, 76 }, { 50, 78 }, { 48, 80 }, { 46, 82 },
|
||||
{ 44, 84 }, { 42, 86 }, { 40, 88 }, { 38, 90 }, { 36, 92 }, { 34, 94 },
|
||||
{ 32, 96 }, { 30, 98 }, { 28, 100 }, { 26, 102 }, { 24, 104 }, { 22, 106 },
|
||||
{ 20, 108 }, { 18, 110 }, { 16, 112 }, { 14, 114 }, { 12, 116 }, { 10, 118 },
|
||||
{ 8, 120 }, { 6, 122 }, { 4, 124 }, { 2, 126 },
|
||||
#elif INTERP_SIMPLE_TAPS == 4
|
||||
{ 0, 128, 0, 0 }, { -1, 128, 2, -1 }, { -2, 127, 4, -1 },
|
||||
{ -3, 126, 7, -2 }, { -4, 125, 9, -2 }, { -5, 125, 11, -3 },
|
||||
{ -6, 124, 13, -3 }, { -7, 123, 16, -4 }, { -7, 122, 18, -5 },
|
||||
{ -8, 121, 20, -5 }, { -9, 120, 23, -6 }, { -9, 118, 25, -6 },
|
||||
{ -10, 117, 28, -7 }, { -11, 116, 30, -7 }, { -11, 114, 33, -8 },
|
||||
{ -12, 113, 35, -8 }, { -12, 111, 38, -9 }, { -13, 109, 41, -9 },
|
||||
{ -13, 108, 43, -10 }, { -13, 106, 45, -10 }, { -13, 104, 48, -11 },
|
||||
{ -14, 102, 51, -11 }, { -14, 100, 53, -11 }, { -14, 98, 56, -12 },
|
||||
{ -14, 96, 58, -12 }, { -14, 94, 61, -13 }, { -15, 92, 64, -13 },
|
||||
{ -15, 90, 66, -13 }, { -15, 87, 69, -13 }, { -14, 85, 71, -14 },
|
||||
{ -14, 83, 73, -14 }, { -14, 80, 76, -14 }, { -14, 78, 78, -14 },
|
||||
{ -14, 76, 80, -14 }, { -14, 73, 83, -14 }, { -14, 71, 85, -14 },
|
||||
{ -13, 69, 87, -15 }, { -13, 66, 90, -15 }, { -13, 64, 92, -15 },
|
||||
{ -13, 61, 94, -14 }, { -12, 58, 96, -14 }, { -12, 56, 98, -14 },
|
||||
{ -11, 53, 100, -14 }, { -11, 51, 102, -14 }, { -11, 48, 104, -13 },
|
||||
{ -10, 45, 106, -13 }, { -10, 43, 108, -13 }, { -9, 41, 109, -13 },
|
||||
{ -9, 38, 111, -12 }, { -8, 35, 113, -12 }, { -8, 33, 114, -11 },
|
||||
{ -7, 30, 116, -11 }, { -7, 28, 117, -10 }, { -6, 25, 118, -9 },
|
||||
{ -6, 23, 120, -9 }, { -5, 20, 121, -8 }, { -5, 18, 122, -7 },
|
||||
{ -4, 16, 123, -7 }, { -3, 13, 124, -6 }, { -3, 11, 125, -5 },
|
||||
{ -2, 9, 125, -4 }, { -2, 7, 126, -3 }, { -1, 4, 127, -2 },
|
||||
{ -1, 2, 128, -1 },
|
||||
#elif INTERP_SIMPLE_TAPS == 6
|
||||
{ 0, 0, 128, 0, 0, 0 }, { 0, -1, 128, 2, -1, 0 },
|
||||
{ 1, -3, 127, 4, -2, 1 }, { 1, -4, 127, 6, -3, 1 },
|
||||
{ 2, -6, 126, 8, -3, 1 }, { 2, -7, 125, 11, -4, 1 },
|
||||
{ 2, -9, 125, 13, -5, 2 }, { 3, -10, 124, 15, -6, 2 },
|
||||
{ 3, -11, 123, 18, -7, 2 }, { 3, -12, 122, 20, -8, 3 },
|
||||
{ 4, -13, 121, 22, -9, 3 }, { 4, -14, 119, 25, -9, 3 },
|
||||
{ 4, -15, 118, 27, -10, 4 }, { 4, -16, 117, 30, -11, 4 },
|
||||
{ 5, -17, 116, 32, -12, 4 }, { 5, -17, 114, 35, -13, 4 },
|
||||
{ 5, -18, 112, 37, -13, 5 }, { 5, -19, 111, 40, -14, 5 },
|
||||
{ 6, -19, 109, 42, -15, 5 }, { 6, -20, 107, 45, -15, 5 },
|
||||
{ 6, -20, 105, 48, -16, 5 }, { 6, -21, 103, 51, -17, 6 },
|
||||
{ 6, -21, 101, 53, -17, 6 }, { 6, -21, 99, 56, -18, 6 },
|
||||
{ 7, -22, 97, 58, -18, 6 }, { 7, -22, 95, 61, -19, 6 },
|
||||
{ 7, -22, 93, 63, -19, 6 }, { 7, -22, 91, 66, -20, 6 },
|
||||
{ 7, -22, 88, 69, -20, 6 }, { 7, -22, 86, 71, -21, 7 },
|
||||
{ 7, -22, 83, 74, -21, 7 }, { 7, -22, 81, 76, -21, 7 },
|
||||
{ 7, -22, 79, 79, -22, 7 }, { 7, -21, 76, 81, -22, 7 },
|
||||
{ 7, -21, 74, 83, -22, 7 }, { 7, -21, 71, 86, -22, 7 },
|
||||
{ 6, -20, 69, 88, -22, 7 }, { 6, -20, 66, 91, -22, 7 },
|
||||
{ 6, -19, 63, 93, -22, 7 }, { 6, -19, 61, 95, -22, 7 },
|
||||
{ 6, -18, 58, 97, -22, 7 }, { 6, -18, 56, 99, -21, 6 },
|
||||
{ 6, -17, 53, 101, -21, 6 }, { 6, -17, 51, 103, -21, 6 },
|
||||
{ 5, -16, 48, 105, -20, 6 }, { 5, -15, 45, 107, -20, 6 },
|
||||
{ 5, -15, 42, 109, -19, 6 }, { 5, -14, 40, 111, -19, 5 },
|
||||
{ 5, -13, 37, 112, -18, 5 }, { 4, -13, 35, 114, -17, 5 },
|
||||
{ 4, -12, 32, 116, -17, 5 }, { 4, -11, 30, 117, -16, 4 },
|
||||
{ 4, -10, 27, 118, -15, 4 }, { 3, -9, 25, 119, -14, 4 },
|
||||
{ 3, -9, 22, 121, -13, 4 }, { 3, -8, 20, 122, -12, 3 },
|
||||
{ 2, -7, 18, 123, -11, 3 }, { 2, -6, 15, 124, -10, 3 },
|
||||
{ 2, -5, 13, 125, -9, 2 }, { 1, -4, 11, 125, -7, 2 },
|
||||
{ 1, -3, 8, 126, -6, 2 }, { 1, -3, 6, 127, -4, 1 },
|
||||
{ 1, -2, 4, 127, -3, 1 }, { 0, -1, 2, 128, -1, 0 },
|
||||
#else
|
||||
#error "Invalid value of INTERP_SIMPLE_TAPS"
|
||||
#endif // INTERP_SIMPLE_TAPS == 2
|
||||
};
|
||||
#endif // CONFIG_FRAME_SUPERRES && CONFIG_LOOP_RESTORATION
|
||||
|
||||
// Filters for factor of 2 downsampling.
|
||||
static const int16_t av1_down2_symeven_half_filter[] = { 56, 12, -3, -1 };
|
||||
static const int16_t av1_down2_symodd_half_filter[] = { 64, 35, 0, -3 };
|
||||
|
|
@ -331,33 +314,34 @@ static const interp_kernel *choose_interp_filter(int inlength, int outlength) {
|
|||
return filteredinterp_filters500;
|
||||
}
|
||||
|
||||
static void interpolate(const uint8_t *const input, int inlength,
|
||||
uint8_t *output, int outlength) {
|
||||
const int64_t delta =
|
||||
(((uint64_t)inlength << 32) + outlength / 2) / outlength;
|
||||
const int64_t offset =
|
||||
static void interpolate_core(const uint8_t *const input, int inlength,
|
||||
uint8_t *output, int outlength,
|
||||
const int16_t *interp_filters, int interp_taps) {
|
||||
const int32_t delta =
|
||||
(((uint32_t)inlength << INTERP_PRECISION_BITS) + outlength / 2) /
|
||||
outlength;
|
||||
const int32_t offset =
|
||||
inlength > outlength
|
||||
? (((int64_t)(inlength - outlength) << 31) + outlength / 2) /
|
||||
? (((int32_t)(inlength - outlength) << (INTERP_PRECISION_BITS - 1)) +
|
||||
outlength / 2) /
|
||||
outlength
|
||||
: -(((int64_t)(outlength - inlength) << 31) + outlength / 2) /
|
||||
: -(((int32_t)(outlength - inlength) << (INTERP_PRECISION_BITS - 1)) +
|
||||
outlength / 2) /
|
||||
outlength;
|
||||
uint8_t *optr = output;
|
||||
int x, x1, x2, sum, k, int_pel, sub_pel;
|
||||
int64_t y;
|
||||
|
||||
const interp_kernel *interp_filters =
|
||||
choose_interp_filter(inlength, outlength);
|
||||
int32_t y;
|
||||
|
||||
x = 0;
|
||||
y = offset + SUBPEL_INTERP_EXTRA_OFF;
|
||||
while ((y >> INTERP_PRECISION_BITS) < (INTERP_TAPS / 2 - 1)) {
|
||||
while ((y >> INTERP_PRECISION_BITS) < (interp_taps / 2 - 1)) {
|
||||
x++;
|
||||
y += delta;
|
||||
}
|
||||
x1 = x;
|
||||
x = outlength - 1;
|
||||
y = delta * x + offset + SUBPEL_INTERP_EXTRA_OFF;
|
||||
while ((y >> INTERP_PRECISION_BITS) + (int64_t)(INTERP_TAPS / 2) >=
|
||||
while ((y >> INTERP_PRECISION_BITS) + (int32_t)(interp_taps / 2) >=
|
||||
inlength) {
|
||||
x--;
|
||||
y -= delta;
|
||||
|
|
@ -366,13 +350,12 @@ static void interpolate(const uint8_t *const input, int inlength,
|
|||
if (x1 > x2) {
|
||||
for (x = 0, y = offset + SUBPEL_INTERP_EXTRA_OFF; x < outlength;
|
||||
++x, y += delta) {
|
||||
const int16_t *filter;
|
||||
int_pel = y >> INTERP_PRECISION_BITS;
|
||||
sub_pel = (y >> SUBPEL_INTERP_EXTRA_BITS) & SUBPEL_MASK_RS;
|
||||
filter = interp_filters[sub_pel];
|
||||
const int16_t *filter = &interp_filters[sub_pel * interp_taps];
|
||||
sum = 0;
|
||||
for (k = 0; k < INTERP_TAPS; ++k) {
|
||||
const int pk = int_pel - INTERP_TAPS / 2 + 1 + k;
|
||||
for (k = 0; k < interp_taps; ++k) {
|
||||
const int pk = int_pel - interp_taps / 2 + 1 + k;
|
||||
sum += filter[k] * input[AOMMAX(AOMMIN(pk, inlength - 1), 0)];
|
||||
}
|
||||
*optr++ = clip_pixel(ROUND_POWER_OF_TWO(sum, FILTER_BITS));
|
||||
|
|
@ -380,41 +363,55 @@ static void interpolate(const uint8_t *const input, int inlength,
|
|||
} else {
|
||||
// Initial part.
|
||||
for (x = 0, y = offset + SUBPEL_INTERP_EXTRA_OFF; x < x1; ++x, y += delta) {
|
||||
const int16_t *filter;
|
||||
int_pel = y >> INTERP_PRECISION_BITS;
|
||||
sub_pel = (y >> SUBPEL_INTERP_EXTRA_BITS) & SUBPEL_MASK_RS;
|
||||
filter = interp_filters[sub_pel];
|
||||
const int16_t *filter = &interp_filters[sub_pel * interp_taps];
|
||||
sum = 0;
|
||||
for (k = 0; k < INTERP_TAPS; ++k)
|
||||
sum += filter[k] * input[AOMMAX(int_pel - INTERP_TAPS / 2 + 1 + k, 0)];
|
||||
for (k = 0; k < interp_taps; ++k)
|
||||
sum += filter[k] * input[AOMMAX(int_pel - interp_taps / 2 + 1 + k, 0)];
|
||||
*optr++ = clip_pixel(ROUND_POWER_OF_TWO(sum, FILTER_BITS));
|
||||
}
|
||||
// Middle part.
|
||||
for (; x <= x2; ++x, y += delta) {
|
||||
const int16_t *filter;
|
||||
int_pel = y >> INTERP_PRECISION_BITS;
|
||||
sub_pel = (y >> SUBPEL_INTERP_EXTRA_BITS) & SUBPEL_MASK_RS;
|
||||
filter = interp_filters[sub_pel];
|
||||
const int16_t *filter = &interp_filters[sub_pel * interp_taps];
|
||||
sum = 0;
|
||||
for (k = 0; k < INTERP_TAPS; ++k)
|
||||
sum += filter[k] * input[int_pel - INTERP_TAPS / 2 + 1 + k];
|
||||
for (k = 0; k < interp_taps; ++k)
|
||||
sum += filter[k] * input[int_pel - interp_taps / 2 + 1 + k];
|
||||
*optr++ = clip_pixel(ROUND_POWER_OF_TWO(sum, FILTER_BITS));
|
||||
}
|
||||
// End part.
|
||||
for (; x < outlength; ++x, y += delta) {
|
||||
const int16_t *filter;
|
||||
int_pel = y >> INTERP_PRECISION_BITS;
|
||||
sub_pel = (y >> SUBPEL_INTERP_EXTRA_BITS) & SUBPEL_MASK_RS;
|
||||
filter = interp_filters[sub_pel];
|
||||
const int16_t *filter = &interp_filters[sub_pel * interp_taps];
|
||||
sum = 0;
|
||||
for (k = 0; k < INTERP_TAPS; ++k)
|
||||
for (k = 0; k < interp_taps; ++k)
|
||||
sum += filter[k] *
|
||||
input[AOMMIN(int_pel - INTERP_TAPS / 2 + 1 + k, inlength - 1)];
|
||||
input[AOMMIN(int_pel - interp_taps / 2 + 1 + k, inlength - 1)];
|
||||
*optr++ = clip_pixel(ROUND_POWER_OF_TWO(sum, FILTER_BITS));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static void interpolate(const uint8_t *const input, int inlength,
|
||||
uint8_t *output, int outlength) {
|
||||
const interp_kernel *interp_filters =
|
||||
choose_interp_filter(inlength, outlength);
|
||||
|
||||
interpolate_core(input, inlength, output, outlength, &interp_filters[0][0],
|
||||
INTERP_TAPS);
|
||||
}
|
||||
|
||||
#if CONFIG_FRAME_SUPERRES && CONFIG_LOOP_RESTORATION
|
||||
static void interpolate_simple(const uint8_t *const input, int inlength,
|
||||
uint8_t *output, int outlength) {
|
||||
interpolate_core(input, inlength, output, outlength, &filter_simple[0][0],
|
||||
INTERP_SIMPLE_TAPS);
|
||||
}
|
||||
#endif // CONFIG_FRAME_SUPERRES && CONFIG_LOOP_RESTORATION
|
||||
|
||||
#ifndef __clang_analyzer__
|
||||
static void down2_symeven(const uint8_t *const input, int length,
|
||||
uint8_t *output) {
|
||||
|
|
@ -596,14 +593,15 @@ static void fill_arr_to_col(uint8_t *img, int stride, int len, uint8_t *arr) {
|
|||
}
|
||||
}
|
||||
|
||||
void av1_resize_plane(const uint8_t *const input, int height, int width,
|
||||
int in_stride, uint8_t *output, int height2, int width2,
|
||||
int out_stride) {
|
||||
static void resize_plane(const uint8_t *const input, int height, int width,
|
||||
int in_stride, uint8_t *output, int height2,
|
||||
int width2, int out_stride) {
|
||||
int i;
|
||||
uint8_t *intbuf = (uint8_t *)malloc(sizeof(uint8_t) * width2 * height);
|
||||
uint8_t *tmpbuf = (uint8_t *)malloc(sizeof(uint8_t) * AOMMAX(width, height));
|
||||
uint8_t *arrbuf = (uint8_t *)malloc(sizeof(uint8_t) * height);
|
||||
uint8_t *arrbuf2 = (uint8_t *)malloc(sizeof(uint8_t) * height2);
|
||||
uint8_t *intbuf = (uint8_t *)aom_malloc(sizeof(uint8_t) * width2 * height);
|
||||
uint8_t *tmpbuf =
|
||||
(uint8_t *)aom_malloc(sizeof(uint8_t) * AOMMAX(width, height));
|
||||
uint8_t *arrbuf = (uint8_t *)aom_malloc(sizeof(uint8_t) * height);
|
||||
uint8_t *arrbuf2 = (uint8_t *)aom_malloc(sizeof(uint8_t) * height2);
|
||||
if (intbuf == NULL || tmpbuf == NULL || arrbuf == NULL || arrbuf2 == NULL)
|
||||
goto Error;
|
||||
assert(width > 0);
|
||||
|
|
@ -620,40 +618,80 @@ void av1_resize_plane(const uint8_t *const input, int height, int width,
|
|||
}
|
||||
|
||||
Error:
|
||||
free(intbuf);
|
||||
free(tmpbuf);
|
||||
free(arrbuf);
|
||||
free(arrbuf2);
|
||||
aom_free(intbuf);
|
||||
aom_free(tmpbuf);
|
||||
aom_free(arrbuf);
|
||||
aom_free(arrbuf2);
|
||||
}
|
||||
|
||||
#if CONFIG_FRAME_SUPERRES
|
||||
static void upscale_normative(const uint8_t *const input, int length,
|
||||
uint8_t *output, int olength) {
|
||||
#if CONFIG_LOOP_RESTORATION
|
||||
interpolate_simple(input, length, output, olength);
|
||||
#else
|
||||
interpolate(input, length, output, olength);
|
||||
#endif // CONFIG_LOOP_RESTORATION
|
||||
}
|
||||
|
||||
static void upscale_normative_plane(const uint8_t *const input, int height,
|
||||
int width, int in_stride, uint8_t *output,
|
||||
int height2, int width2, int out_stride) {
|
||||
int i;
|
||||
uint8_t *intbuf = (uint8_t *)aom_malloc(sizeof(uint8_t) * width2 * height);
|
||||
uint8_t *arrbuf = (uint8_t *)aom_malloc(sizeof(uint8_t) * height);
|
||||
uint8_t *arrbuf2 = (uint8_t *)aom_malloc(sizeof(uint8_t) * height2);
|
||||
if (intbuf == NULL || arrbuf == NULL || arrbuf2 == NULL) goto Error;
|
||||
assert(width > 0);
|
||||
assert(height > 0);
|
||||
assert(width2 > 0);
|
||||
assert(height2 > 0);
|
||||
for (i = 0; i < height; ++i)
|
||||
upscale_normative(input + in_stride * i, width, intbuf + width2 * i,
|
||||
width2);
|
||||
for (i = 0; i < width2; ++i) {
|
||||
fill_col_to_arr(intbuf + i, width2, height, arrbuf);
|
||||
upscale_normative(arrbuf, height, arrbuf2, height2);
|
||||
fill_arr_to_col(output + i, out_stride, height2, arrbuf2);
|
||||
}
|
||||
|
||||
Error:
|
||||
aom_free(intbuf);
|
||||
aom_free(arrbuf);
|
||||
aom_free(arrbuf2);
|
||||
}
|
||||
#endif // CONFIG_FRAME_SUPERRES
|
||||
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
static void highbd_interpolate(const uint16_t *const input, int inlength,
|
||||
uint16_t *output, int outlength, int bd) {
|
||||
const int64_t delta =
|
||||
(((uint64_t)inlength << 32) + outlength / 2) / outlength;
|
||||
const int64_t offset =
|
||||
static void highbd_interpolate_core(const uint16_t *const input, int inlength,
|
||||
uint16_t *output, int outlength, int bd,
|
||||
const int16_t *interp_filters,
|
||||
int interp_taps) {
|
||||
const int32_t delta =
|
||||
(((uint32_t)inlength << INTERP_PRECISION_BITS) + outlength / 2) /
|
||||
outlength;
|
||||
const int32_t offset =
|
||||
inlength > outlength
|
||||
? (((int64_t)(inlength - outlength) << 31) + outlength / 2) /
|
||||
? (((int32_t)(inlength - outlength) << (INTERP_PRECISION_BITS - 1)) +
|
||||
outlength / 2) /
|
||||
outlength
|
||||
: -(((int64_t)(outlength - inlength) << 31) + outlength / 2) /
|
||||
: -(((int32_t)(outlength - inlength) << (INTERP_PRECISION_BITS - 1)) +
|
||||
outlength / 2) /
|
||||
outlength;
|
||||
uint16_t *optr = output;
|
||||
int x, x1, x2, sum, k, int_pel, sub_pel;
|
||||
int64_t y;
|
||||
|
||||
const interp_kernel *interp_filters =
|
||||
choose_interp_filter(inlength, outlength);
|
||||
int32_t y;
|
||||
|
||||
x = 0;
|
||||
y = offset + SUBPEL_INTERP_EXTRA_OFF;
|
||||
while ((y >> INTERP_PRECISION_BITS) < (INTERP_TAPS / 2 - 1)) {
|
||||
while ((y >> INTERP_PRECISION_BITS) < (interp_taps / 2 - 1)) {
|
||||
x++;
|
||||
y += delta;
|
||||
}
|
||||
x1 = x;
|
||||
x = outlength - 1;
|
||||
y = delta * x + offset + SUBPEL_INTERP_EXTRA_OFF;
|
||||
while ((y >> INTERP_PRECISION_BITS) + (int64_t)(INTERP_TAPS / 2) >=
|
||||
while ((y >> INTERP_PRECISION_BITS) + (int32_t)(interp_taps / 2) >=
|
||||
inlength) {
|
||||
x--;
|
||||
y -= delta;
|
||||
|
|
@ -662,13 +700,12 @@ static void highbd_interpolate(const uint16_t *const input, int inlength,
|
|||
if (x1 > x2) {
|
||||
for (x = 0, y = offset + SUBPEL_INTERP_EXTRA_OFF; x < outlength;
|
||||
++x, y += delta) {
|
||||
const int16_t *filter;
|
||||
int_pel = y >> INTERP_PRECISION_BITS;
|
||||
sub_pel = (y >> SUBPEL_INTERP_EXTRA_BITS) & SUBPEL_MASK_RS;
|
||||
filter = interp_filters[sub_pel];
|
||||
const int16_t *filter = &interp_filters[sub_pel * interp_taps];
|
||||
sum = 0;
|
||||
for (k = 0; k < INTERP_TAPS; ++k) {
|
||||
const int pk = int_pel - INTERP_TAPS / 2 + 1 + k;
|
||||
for (k = 0; k < interp_taps; ++k) {
|
||||
const int pk = int_pel - interp_taps / 2 + 1 + k;
|
||||
sum += filter[k] * input[AOMMAX(AOMMIN(pk, inlength - 1), 0)];
|
||||
}
|
||||
*optr++ = clip_pixel_highbd(ROUND_POWER_OF_TWO(sum, FILTER_BITS), bd);
|
||||
|
|
@ -676,41 +713,55 @@ static void highbd_interpolate(const uint16_t *const input, int inlength,
|
|||
} else {
|
||||
// Initial part.
|
||||
for (x = 0, y = offset + SUBPEL_INTERP_EXTRA_OFF; x < x1; ++x, y += delta) {
|
||||
const int16_t *filter;
|
||||
int_pel = y >> INTERP_PRECISION_BITS;
|
||||
sub_pel = (y >> SUBPEL_INTERP_EXTRA_BITS) & SUBPEL_MASK_RS;
|
||||
filter = interp_filters[sub_pel];
|
||||
const int16_t *filter = &interp_filters[sub_pel * interp_taps];
|
||||
sum = 0;
|
||||
for (k = 0; k < INTERP_TAPS; ++k)
|
||||
sum += filter[k] * input[AOMMAX(int_pel - INTERP_TAPS / 2 + 1 + k, 0)];
|
||||
for (k = 0; k < interp_taps; ++k)
|
||||
sum += filter[k] * input[AOMMAX(int_pel - interp_taps / 2 + 1 + k, 0)];
|
||||
*optr++ = clip_pixel_highbd(ROUND_POWER_OF_TWO(sum, FILTER_BITS), bd);
|
||||
}
|
||||
// Middle part.
|
||||
for (; x <= x2; ++x, y += delta) {
|
||||
const int16_t *filter;
|
||||
int_pel = y >> INTERP_PRECISION_BITS;
|
||||
sub_pel = (y >> SUBPEL_INTERP_EXTRA_BITS) & SUBPEL_MASK_RS;
|
||||
filter = interp_filters[sub_pel];
|
||||
const int16_t *filter = &interp_filters[sub_pel * interp_taps];
|
||||
sum = 0;
|
||||
for (k = 0; k < INTERP_TAPS; ++k)
|
||||
sum += filter[k] * input[int_pel - INTERP_TAPS / 2 + 1 + k];
|
||||
for (k = 0; k < interp_taps; ++k)
|
||||
sum += filter[k] * input[int_pel - interp_taps / 2 + 1 + k];
|
||||
*optr++ = clip_pixel_highbd(ROUND_POWER_OF_TWO(sum, FILTER_BITS), bd);
|
||||
}
|
||||
// End part.
|
||||
for (; x < outlength; ++x, y += delta) {
|
||||
const int16_t *filter;
|
||||
int_pel = y >> INTERP_PRECISION_BITS;
|
||||
sub_pel = (y >> SUBPEL_INTERP_EXTRA_BITS) & SUBPEL_MASK_RS;
|
||||
filter = interp_filters[sub_pel];
|
||||
const int16_t *filter = &interp_filters[sub_pel * interp_taps];
|
||||
sum = 0;
|
||||
for (k = 0; k < INTERP_TAPS; ++k)
|
||||
for (k = 0; k < interp_taps; ++k)
|
||||
sum += filter[k] *
|
||||
input[AOMMIN(int_pel - INTERP_TAPS / 2 + 1 + k, inlength - 1)];
|
||||
input[AOMMIN(int_pel - interp_taps / 2 + 1 + k, inlength - 1)];
|
||||
*optr++ = clip_pixel_highbd(ROUND_POWER_OF_TWO(sum, FILTER_BITS), bd);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static void highbd_interpolate(const uint16_t *const input, int inlength,
|
||||
uint16_t *output, int outlength, int bd) {
|
||||
const interp_kernel *interp_filters =
|
||||
choose_interp_filter(inlength, outlength);
|
||||
|
||||
highbd_interpolate_core(input, inlength, output, outlength, bd,
|
||||
&interp_filters[0][0], INTERP_TAPS);
|
||||
}
|
||||
|
||||
#if CONFIG_FRAME_SUPERRES && CONFIG_LOOP_RESTORATION
|
||||
static void highbd_interpolate_simple(const uint16_t *const input, int inlength,
|
||||
uint16_t *output, int outlength, int bd) {
|
||||
highbd_interpolate_core(input, inlength, output, outlength, bd,
|
||||
&filter_simple[0][0], INTERP_SIMPLE_TAPS);
|
||||
}
|
||||
#endif // CONFIG_FRAME_SUPERRES && CONFIG_LOOP_RESTORATION
|
||||
|
||||
#ifndef __clang_analyzer__
|
||||
static void highbd_down2_symeven(const uint16_t *const input, int length,
|
||||
uint16_t *output, int bd) {
|
||||
|
|
@ -877,15 +928,16 @@ static void highbd_fill_arr_to_col(uint16_t *img, int stride, int len,
|
|||
}
|
||||
}
|
||||
|
||||
void av1_highbd_resize_plane(const uint8_t *const input, int height, int width,
|
||||
int in_stride, uint8_t *output, int height2,
|
||||
int width2, int out_stride, int bd) {
|
||||
static void highbd_resize_plane(const uint8_t *const input, int height,
|
||||
int width, int in_stride, uint8_t *output,
|
||||
int height2, int width2, int out_stride,
|
||||
int bd) {
|
||||
int i;
|
||||
uint16_t *intbuf = (uint16_t *)malloc(sizeof(uint16_t) * width2 * height);
|
||||
uint16_t *intbuf = (uint16_t *)aom_malloc(sizeof(uint16_t) * width2 * height);
|
||||
uint16_t *tmpbuf =
|
||||
(uint16_t *)malloc(sizeof(uint16_t) * AOMMAX(width, height));
|
||||
uint16_t *arrbuf = (uint16_t *)malloc(sizeof(uint16_t) * height);
|
||||
uint16_t *arrbuf2 = (uint16_t *)malloc(sizeof(uint16_t) * height2);
|
||||
(uint16_t *)aom_malloc(sizeof(uint16_t) * AOMMAX(width, height));
|
||||
uint16_t *arrbuf = (uint16_t *)aom_malloc(sizeof(uint16_t) * height);
|
||||
uint16_t *arrbuf2 = (uint16_t *)aom_malloc(sizeof(uint16_t) * height2);
|
||||
if (intbuf == NULL || tmpbuf == NULL || arrbuf == NULL || arrbuf2 == NULL)
|
||||
goto Error;
|
||||
for (i = 0; i < height; ++i) {
|
||||
|
|
@ -900,11 +952,49 @@ void av1_highbd_resize_plane(const uint8_t *const input, int height, int width,
|
|||
}
|
||||
|
||||
Error:
|
||||
free(intbuf);
|
||||
free(tmpbuf);
|
||||
free(arrbuf);
|
||||
free(arrbuf2);
|
||||
aom_free(intbuf);
|
||||
aom_free(tmpbuf);
|
||||
aom_free(arrbuf);
|
||||
aom_free(arrbuf2);
|
||||
}
|
||||
|
||||
#if CONFIG_FRAME_SUPERRES
|
||||
static void highbd_upscale_normative(const uint16_t *const input, int length,
|
||||
uint16_t *output, int olength, int bd) {
|
||||
#if CONFIG_LOOP_RESTORATION
|
||||
highbd_interpolate_simple(input, length, output, olength, bd);
|
||||
#else
|
||||
highbd_interpolate(input, length, output, olength, bd);
|
||||
#endif // CONFIG_LOOP_RESTORATION
|
||||
}
|
||||
|
||||
static void highbd_upscale_normative_plane(const uint8_t *const input,
|
||||
int height, int width, int in_stride,
|
||||
uint8_t *output, int height2,
|
||||
int width2, int out_stride, int bd) {
|
||||
int i;
|
||||
uint16_t *intbuf = (uint16_t *)aom_malloc(sizeof(uint16_t) * width2 * height);
|
||||
uint16_t *arrbuf = (uint16_t *)aom_malloc(sizeof(uint16_t) * height);
|
||||
uint16_t *arrbuf2 = (uint16_t *)aom_malloc(sizeof(uint16_t) * height2);
|
||||
if (intbuf == NULL || arrbuf == NULL || arrbuf2 == NULL) goto Error;
|
||||
for (i = 0; i < height; ++i) {
|
||||
highbd_upscale_normative(CONVERT_TO_SHORTPTR(input + in_stride * i), width,
|
||||
intbuf + width2 * i, width2, bd);
|
||||
}
|
||||
for (i = 0; i < width2; ++i) {
|
||||
highbd_fill_col_to_arr(intbuf + i, width2, height, arrbuf);
|
||||
highbd_upscale_normative(arrbuf, height, arrbuf2, height2, bd);
|
||||
highbd_fill_arr_to_col(CONVERT_TO_SHORTPTR(output + i), out_stride, height2,
|
||||
arrbuf2);
|
||||
}
|
||||
|
||||
Error:
|
||||
aom_free(intbuf);
|
||||
aom_free(arrbuf);
|
||||
aom_free(arrbuf2);
|
||||
}
|
||||
#endif // CONFIG_FRAME_SUPERRES
|
||||
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
|
||||
void av1_resize_frame420(const uint8_t *const y, int y_stride,
|
||||
|
|
@ -912,11 +1002,11 @@ void av1_resize_frame420(const uint8_t *const y, int y_stride,
|
|||
int uv_stride, int height, int width, uint8_t *oy,
|
||||
int oy_stride, uint8_t *ou, uint8_t *ov,
|
||||
int ouv_stride, int oheight, int owidth) {
|
||||
av1_resize_plane(y, height, width, y_stride, oy, oheight, owidth, oy_stride);
|
||||
av1_resize_plane(u, height / 2, width / 2, uv_stride, ou, oheight / 2,
|
||||
owidth / 2, ouv_stride);
|
||||
av1_resize_plane(v, height / 2, width / 2, uv_stride, ov, oheight / 2,
|
||||
owidth / 2, ouv_stride);
|
||||
resize_plane(y, height, width, y_stride, oy, oheight, owidth, oy_stride);
|
||||
resize_plane(u, height / 2, width / 2, uv_stride, ou, oheight / 2, owidth / 2,
|
||||
ouv_stride);
|
||||
resize_plane(v, height / 2, width / 2, uv_stride, ov, oheight / 2, owidth / 2,
|
||||
ouv_stride);
|
||||
}
|
||||
|
||||
void av1_resize_frame422(const uint8_t *const y, int y_stride,
|
||||
|
|
@ -924,11 +1014,11 @@ void av1_resize_frame422(const uint8_t *const y, int y_stride,
|
|||
int uv_stride, int height, int width, uint8_t *oy,
|
||||
int oy_stride, uint8_t *ou, uint8_t *ov,
|
||||
int ouv_stride, int oheight, int owidth) {
|
||||
av1_resize_plane(y, height, width, y_stride, oy, oheight, owidth, oy_stride);
|
||||
av1_resize_plane(u, height, width / 2, uv_stride, ou, oheight, owidth / 2,
|
||||
ouv_stride);
|
||||
av1_resize_plane(v, height, width / 2, uv_stride, ov, oheight, owidth / 2,
|
||||
ouv_stride);
|
||||
resize_plane(y, height, width, y_stride, oy, oheight, owidth, oy_stride);
|
||||
resize_plane(u, height, width / 2, uv_stride, ou, oheight, owidth / 2,
|
||||
ouv_stride);
|
||||
resize_plane(v, height, width / 2, uv_stride, ov, oheight, owidth / 2,
|
||||
ouv_stride);
|
||||
}
|
||||
|
||||
void av1_resize_frame444(const uint8_t *const y, int y_stride,
|
||||
|
|
@ -936,11 +1026,9 @@ void av1_resize_frame444(const uint8_t *const y, int y_stride,
|
|||
int uv_stride, int height, int width, uint8_t *oy,
|
||||
int oy_stride, uint8_t *ou, uint8_t *ov,
|
||||
int ouv_stride, int oheight, int owidth) {
|
||||
av1_resize_plane(y, height, width, y_stride, oy, oheight, owidth, oy_stride);
|
||||
av1_resize_plane(u, height, width, uv_stride, ou, oheight, owidth,
|
||||
ouv_stride);
|
||||
av1_resize_plane(v, height, width, uv_stride, ov, oheight, owidth,
|
||||
ouv_stride);
|
||||
resize_plane(y, height, width, y_stride, oy, oheight, owidth, oy_stride);
|
||||
resize_plane(u, height, width, uv_stride, ou, oheight, owidth, ouv_stride);
|
||||
resize_plane(v, height, width, uv_stride, ov, oheight, owidth, ouv_stride);
|
||||
}
|
||||
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
|
|
@ -950,12 +1038,12 @@ void av1_highbd_resize_frame420(const uint8_t *const y, int y_stride,
|
|||
uint8_t *oy, int oy_stride, uint8_t *ou,
|
||||
uint8_t *ov, int ouv_stride, int oheight,
|
||||
int owidth, int bd) {
|
||||
av1_highbd_resize_plane(y, height, width, y_stride, oy, oheight, owidth,
|
||||
oy_stride, bd);
|
||||
av1_highbd_resize_plane(u, height / 2, width / 2, uv_stride, ou, oheight / 2,
|
||||
owidth / 2, ouv_stride, bd);
|
||||
av1_highbd_resize_plane(v, height / 2, width / 2, uv_stride, ov, oheight / 2,
|
||||
owidth / 2, ouv_stride, bd);
|
||||
highbd_resize_plane(y, height, width, y_stride, oy, oheight, owidth,
|
||||
oy_stride, bd);
|
||||
highbd_resize_plane(u, height / 2, width / 2, uv_stride, ou, oheight / 2,
|
||||
owidth / 2, ouv_stride, bd);
|
||||
highbd_resize_plane(v, height / 2, width / 2, uv_stride, ov, oheight / 2,
|
||||
owidth / 2, ouv_stride, bd);
|
||||
}
|
||||
|
||||
void av1_highbd_resize_frame422(const uint8_t *const y, int y_stride,
|
||||
|
|
@ -964,12 +1052,12 @@ void av1_highbd_resize_frame422(const uint8_t *const y, int y_stride,
|
|||
uint8_t *oy, int oy_stride, uint8_t *ou,
|
||||
uint8_t *ov, int ouv_stride, int oheight,
|
||||
int owidth, int bd) {
|
||||
av1_highbd_resize_plane(y, height, width, y_stride, oy, oheight, owidth,
|
||||
oy_stride, bd);
|
||||
av1_highbd_resize_plane(u, height, width / 2, uv_stride, ou, oheight,
|
||||
owidth / 2, ouv_stride, bd);
|
||||
av1_highbd_resize_plane(v, height, width / 2, uv_stride, ov, oheight,
|
||||
owidth / 2, ouv_stride, bd);
|
||||
highbd_resize_plane(y, height, width, y_stride, oy, oheight, owidth,
|
||||
oy_stride, bd);
|
||||
highbd_resize_plane(u, height, width / 2, uv_stride, ou, oheight, owidth / 2,
|
||||
ouv_stride, bd);
|
||||
highbd_resize_plane(v, height, width / 2, uv_stride, ov, oheight, owidth / 2,
|
||||
ouv_stride, bd);
|
||||
}
|
||||
|
||||
void av1_highbd_resize_frame444(const uint8_t *const y, int y_stride,
|
||||
|
|
@ -978,12 +1066,12 @@ void av1_highbd_resize_frame444(const uint8_t *const y, int y_stride,
|
|||
uint8_t *oy, int oy_stride, uint8_t *ou,
|
||||
uint8_t *ov, int ouv_stride, int oheight,
|
||||
int owidth, int bd) {
|
||||
av1_highbd_resize_plane(y, height, width, y_stride, oy, oheight, owidth,
|
||||
oy_stride, bd);
|
||||
av1_highbd_resize_plane(u, height, width, uv_stride, ou, oheight, owidth,
|
||||
ouv_stride, bd);
|
||||
av1_highbd_resize_plane(v, height, width, uv_stride, ov, oheight, owidth,
|
||||
ouv_stride, bd);
|
||||
highbd_resize_plane(y, height, width, y_stride, oy, oheight, owidth,
|
||||
oy_stride, bd);
|
||||
highbd_resize_plane(u, height, width, uv_stride, ou, oheight, owidth,
|
||||
ouv_stride, bd);
|
||||
highbd_resize_plane(v, height, width, uv_stride, ov, oheight, owidth,
|
||||
ouv_stride, bd);
|
||||
}
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
|
||||
|
|
@ -1013,30 +1101,56 @@ void av1_resize_and_extend_frame(const YV12_BUFFER_CONFIG *src,
|
|||
for (i = 0; i < MAX_MB_PLANE; ++i) {
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
if (src->flags & YV12_FLAG_HIGHBITDEPTH)
|
||||
av1_highbd_resize_plane(srcs[i], src_heights[i], src_widths[i],
|
||||
src_strides[i], dsts[i], dst_heights[i],
|
||||
dst_widths[i], dst_strides[i], bd);
|
||||
highbd_resize_plane(srcs[i], src_heights[i], src_widths[i],
|
||||
src_strides[i], dsts[i], dst_heights[i],
|
||||
dst_widths[i], dst_strides[i], bd);
|
||||
else
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
av1_resize_plane(srcs[i], src_heights[i], src_widths[i], src_strides[i],
|
||||
dsts[i], dst_heights[i], dst_widths[i], dst_strides[i]);
|
||||
resize_plane(srcs[i], src_heights[i], src_widths[i], src_strides[i],
|
||||
dsts[i], dst_heights[i], dst_widths[i], dst_strides[i]);
|
||||
}
|
||||
aom_extend_frame_borders(dst);
|
||||
}
|
||||
|
||||
YV12_BUFFER_CONFIG *av1_scale_if_required_fast(AV1_COMMON *cm,
|
||||
YV12_BUFFER_CONFIG *unscaled,
|
||||
YV12_BUFFER_CONFIG *scaled) {
|
||||
if (cm->width != unscaled->y_crop_width ||
|
||||
cm->height != unscaled->y_crop_height) {
|
||||
// For 2x2 scaling down.
|
||||
aom_scale_frame(unscaled, scaled, unscaled->y_buffer, 9, 2, 1, 2, 1, 0);
|
||||
aom_extend_frame_borders(scaled);
|
||||
return scaled;
|
||||
} else {
|
||||
return unscaled;
|
||||
#if CONFIG_FRAME_SUPERRES
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
void av1_upscale_normative_and_extend_frame(const YV12_BUFFER_CONFIG *src,
|
||||
YV12_BUFFER_CONFIG *dst, int bd) {
|
||||
#else
|
||||
void av1_upscale_normative_and_extend_frame(const YV12_BUFFER_CONFIG *src,
|
||||
YV12_BUFFER_CONFIG *dst) {
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
// TODO(dkovalev): replace YV12_BUFFER_CONFIG with aom_image_t
|
||||
int i;
|
||||
const uint8_t *const srcs[3] = { src->y_buffer, src->u_buffer,
|
||||
src->v_buffer };
|
||||
const int src_strides[3] = { src->y_stride, src->uv_stride, src->uv_stride };
|
||||
const int src_widths[3] = { src->y_crop_width, src->uv_crop_width,
|
||||
src->uv_crop_width };
|
||||
const int src_heights[3] = { src->y_crop_height, src->uv_crop_height,
|
||||
src->uv_crop_height };
|
||||
uint8_t *const dsts[3] = { dst->y_buffer, dst->u_buffer, dst->v_buffer };
|
||||
const int dst_strides[3] = { dst->y_stride, dst->uv_stride, dst->uv_stride };
|
||||
const int dst_widths[3] = { dst->y_crop_width, dst->uv_crop_width,
|
||||
dst->uv_crop_width };
|
||||
const int dst_heights[3] = { dst->y_crop_height, dst->uv_crop_height,
|
||||
dst->uv_crop_height };
|
||||
|
||||
for (i = 0; i < MAX_MB_PLANE; ++i) {
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
if (src->flags & YV12_FLAG_HIGHBITDEPTH)
|
||||
highbd_upscale_normative_plane(srcs[i], src_heights[i], src_widths[i],
|
||||
src_strides[i], dsts[i], dst_heights[i],
|
||||
dst_widths[i], dst_strides[i], bd);
|
||||
else
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
upscale_normative_plane(srcs[i], src_heights[i], src_widths[i],
|
||||
src_strides[i], dsts[i], dst_heights[i],
|
||||
dst_widths[i], dst_strides[i]);
|
||||
}
|
||||
aom_extend_frame_borders(dst);
|
||||
}
|
||||
#endif // CONFIG_FRAME_SUPERRES
|
||||
|
||||
YV12_BUFFER_CONFIG *av1_scale_if_required(AV1_COMMON *cm,
|
||||
YV12_BUFFER_CONFIG *unscaled,
|
||||
|
|
@ -1054,17 +1168,45 @@ YV12_BUFFER_CONFIG *av1_scale_if_required(AV1_COMMON *cm,
|
|||
}
|
||||
}
|
||||
|
||||
void av1_calculate_scaled_size(int *width, int *height, int num) {
|
||||
if (num != SCALE_DENOMINATOR) {
|
||||
*width = *width * num / SCALE_DENOMINATOR;
|
||||
*height = *height * num / SCALE_DENOMINATOR;
|
||||
// Make width and height even
|
||||
*width += *width & 1;
|
||||
*height += *height & 1;
|
||||
// Calculates scaled dimensions given original dimensions and the scale
|
||||
// denominator. If 'scale_height' is 1, both width and height are scaled;
|
||||
// otherwise, only the width is scaled.
|
||||
static void calculate_scaled_size_helper(int *width, int *height, int denom,
|
||||
int scale_height) {
|
||||
if (denom != SCALE_NUMERATOR) {
|
||||
*width = *width * SCALE_NUMERATOR / denom;
|
||||
*width += *width & 1; // Make it even.
|
||||
if (scale_height) {
|
||||
*height = *height * SCALE_NUMERATOR / denom;
|
||||
*height += *height & 1; // Make it even.
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void av1_calculate_scaled_size(int *width, int *height, int resize_denom) {
|
||||
calculate_scaled_size_helper(width, height, resize_denom, 1);
|
||||
}
|
||||
|
||||
#if CONFIG_FRAME_SUPERRES
|
||||
void av1_calculate_scaled_superres_size(int *width, int *height,
|
||||
int superres_denom) {
|
||||
calculate_scaled_size_helper(width, height, superres_denom,
|
||||
!CONFIG_HORZONLY_FRAME_SUPERRES);
|
||||
}
|
||||
|
||||
void av1_calculate_unscaled_superres_size(int *width, int *height, int denom) {
|
||||
if (denom != SCALE_NUMERATOR) {
|
||||
// Note: av1_calculate_scaled_superres_size() rounds *up* after division
|
||||
// when the resulting dimensions are odd. So here, we round *down*.
|
||||
*width = *width * denom / SCALE_NUMERATOR;
|
||||
#if CONFIG_HORZONLY_FRAME_SUPERRES
|
||||
(void)height;
|
||||
#else
|
||||
*height = *height * denom / SCALE_NUMERATOR;
|
||||
#endif // CONFIG_HORZONLY_FRAME_SUPERRES
|
||||
}
|
||||
}
|
||||
|
||||
// TODO(afergs): Look for in-place upscaling
|
||||
// TODO(afergs): aom_ vs av1_ functions? Which can I use?
|
||||
// Upscale decoded image.
|
||||
|
|
@ -1138,11 +1280,13 @@ void av1_superres_upscale(AV1_COMMON *cm, BufferPool *const pool) {
|
|||
|
||||
// Scale up and back into frame_to_show.
|
||||
assert(frame_to_show->y_crop_width != cm->width);
|
||||
assert(frame_to_show->y_crop_height != cm->height);
|
||||
assert(IMPLIES(!CONFIG_HORZONLY_FRAME_SUPERRES,
|
||||
frame_to_show->y_crop_height != cm->height));
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
av1_resize_and_extend_frame(©_buffer, frame_to_show, (int)cm->bit_depth);
|
||||
av1_upscale_normative_and_extend_frame(©_buffer, frame_to_show,
|
||||
(int)cm->bit_depth);
|
||||
#else
|
||||
av1_resize_and_extend_frame(©_buffer, frame_to_show);
|
||||
av1_upscale_normative_and_extend_frame(©_buffer, frame_to_show);
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
|
||||
// Free the copy buffer
|
||||
|
|
|
|||
28
third_party/aom/av1/common/resize.h
vendored
28
third_party/aom/av1/common/resize.h
vendored
|
|
@ -71,22 +71,40 @@ void av1_resize_and_extend_frame(const YV12_BUFFER_CONFIG *src,
|
|||
YV12_BUFFER_CONFIG *dst);
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
|
||||
YV12_BUFFER_CONFIG *av1_scale_if_required_fast(AV1_COMMON *cm,
|
||||
YV12_BUFFER_CONFIG *unscaled,
|
||||
YV12_BUFFER_CONFIG *scaled);
|
||||
#if CONFIG_FRAME_SUPERRES
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
void av1_upscale_normative_and_extend_frame(const YV12_BUFFER_CONFIG *src,
|
||||
YV12_BUFFER_CONFIG *dst, int bd);
|
||||
#else
|
||||
void av1_upscale_normative_and_extend_frame(const YV12_BUFFER_CONFIG *src,
|
||||
YV12_BUFFER_CONFIG *dst);
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
#endif // CONFIG_FRAME_SUPERRES
|
||||
|
||||
YV12_BUFFER_CONFIG *av1_scale_if_required(AV1_COMMON *cm,
|
||||
YV12_BUFFER_CONFIG *unscaled,
|
||||
YV12_BUFFER_CONFIG *scaled);
|
||||
|
||||
void av1_calculate_scaled_size(int *width, int *height, int num);
|
||||
// Calculates the scaled dimensions from the given original dimensions and the
|
||||
// resize scale denominator.
|
||||
void av1_calculate_scaled_size(int *width, int *height, int resize_denom);
|
||||
|
||||
#if CONFIG_FRAME_SUPERRES
|
||||
// Similar to above, but calculates scaled dimensions after superres from the
|
||||
// given original dimensions and superres scale denominator.
|
||||
void av1_calculate_scaled_superres_size(int *width, int *height,
|
||||
int superres_denom);
|
||||
|
||||
// Inverse of av1_calculate_scaled_superres_size() above: calculates the
|
||||
// original dimensions from the given scaled dimensions and the scale
|
||||
// denominator.
|
||||
void av1_calculate_unscaled_superres_size(int *width, int *height, int denom);
|
||||
|
||||
void av1_superres_upscale(AV1_COMMON *cm, BufferPool *const pool);
|
||||
|
||||
// Returns 1 if a superres upscaled frame is unscaled and 0 otherwise.
|
||||
static INLINE int av1_superres_unscaled(const AV1_COMMON *cm) {
|
||||
return (cm->superres_scale_numerator == SCALE_DENOMINATOR);
|
||||
return (cm->superres_scale_denominator == SCALE_NUMERATOR);
|
||||
}
|
||||
#endif // CONFIG_FRAME_SUPERRES
|
||||
|
||||
|
|
|
|||
825
third_party/aom/av1/common/restoration.c
vendored
825
third_party/aom/av1/common/restoration.c
vendored
File diff suppressed because it is too large
Load diff
191
third_party/aom/av1/common/restoration.h
vendored
191
third_party/aom/av1/common/restoration.h
vendored
|
|
@ -24,18 +24,77 @@ extern "C" {
|
|||
#define CLIP(x, lo, hi) ((x) < (lo) ? (lo) : (x) > (hi) ? (hi) : (x))
|
||||
#define RINT(x) ((x) < 0 ? (int)((x)-0.5) : (int)((x) + 0.5))
|
||||
|
||||
#define RESTORATION_TILESIZE_MAX 256
|
||||
#define RESTORATION_TILEPELS_MAX \
|
||||
(RESTORATION_TILESIZE_MAX * RESTORATION_TILESIZE_MAX * 9 / 4)
|
||||
#define RESTORATION_PROC_UNIT_SIZE 64
|
||||
|
||||
#if CONFIG_STRIPED_LOOP_RESTORATION
|
||||
// Filter tile grid offset upwards compared to the superblock grid
|
||||
#define RESTORATION_TILE_OFFSET 8
|
||||
#endif
|
||||
|
||||
#if CONFIG_STRIPED_LOOP_RESTORATION
|
||||
#define SGRPROJ_BORDER_VERT 2 // Vertical border used for Sgr
|
||||
#else
|
||||
#define SGRPROJ_BORDER_VERT 1 // Vertical border used for Sgr
|
||||
#endif
|
||||
#define SGRPROJ_BORDER_HORZ 2 // Horizontal border used for Sgr
|
||||
|
||||
#if CONFIG_STRIPED_LOOP_RESTORATION
|
||||
#define WIENER_BORDER_VERT 2 // Vertical border used for Wiener
|
||||
#else
|
||||
#define WIENER_BORDER_VERT 1 // Vertical border used for Wiener
|
||||
#endif
|
||||
#define WIENER_HALFWIN 3
|
||||
#define WIENER_BORDER_HORZ (WIENER_HALFWIN) // Horizontal border for Wiener
|
||||
|
||||
// RESTORATION_BORDER_VERT determines line buffer requirement for LR.
|
||||
// Should be set at the max of SGRPROJ_BORDER_VERT and WIENER_BORDER_VERT.
|
||||
// Note the line buffer needed is twice the value of this macro.
|
||||
#if SGRPROJ_BORDER_VERT >= WIENER_BORDER_VERT
|
||||
#define RESTORATION_BORDER_VERT (SGRPROJ_BORDER_VERT)
|
||||
#else
|
||||
#define RESTORATION_BORDER_VERT (WIENER_BORDER_VERT)
|
||||
#endif // SGRPROJ_BORDER_VERT >= WIENER_BORDER_VERT
|
||||
|
||||
#if SGRPROJ_BORDER_HORZ >= WIENER_BORDER_HORZ
|
||||
#define RESTORATION_BORDER_HORZ (SGRPROJ_BORDER_HORZ)
|
||||
#else
|
||||
#define RESTORATION_BORDER_HORZ (WIENER_BORDER_HORZ)
|
||||
#endif // SGRPROJ_BORDER_VERT >= WIENER_BORDER_VERT
|
||||
|
||||
#if CONFIG_STRIPED_LOOP_RESTORATION
|
||||
// Additional pixels to the left and right in above/below buffers
|
||||
// It is RESTORATION_BORDER_HORZ rounded up to get nicer buffer alignment
|
||||
#define RESTORATION_EXTRA_HORZ 4
|
||||
#endif
|
||||
|
||||
// Pad up to 20 more (may be much less is needed)
|
||||
#define RESTORATION_PADDING 20
|
||||
#define RESTORATION_PROC_UNIT_PELS \
|
||||
((RESTORATION_PROC_UNIT_SIZE + RESTORATION_BORDER_HORZ * 2 + \
|
||||
RESTORATION_PADDING) * \
|
||||
(RESTORATION_PROC_UNIT_SIZE + RESTORATION_BORDER_VERT * 2 + \
|
||||
RESTORATION_PADDING))
|
||||
|
||||
#define RESTORATION_TILESIZE_MAX 256
|
||||
#if CONFIG_STRIPED_LOOP_RESTORATION
|
||||
#define RESTORATION_TILEPELS_HORZ_MAX \
|
||||
(RESTORATION_TILESIZE_MAX * 3 / 2 + 2 * RESTORATION_BORDER_HORZ + 16)
|
||||
#define RESTORATION_TILEPELS_VERT_MAX \
|
||||
((RESTORATION_TILESIZE_MAX * 3 / 2 + 2 * RESTORATION_BORDER_VERT + \
|
||||
RESTORATION_TILE_OFFSET))
|
||||
#define RESTORATION_TILEPELS_MAX \
|
||||
(RESTORATION_TILEPELS_HORZ_MAX * RESTORATION_TILEPELS_VERT_MAX)
|
||||
#else
|
||||
#define RESTORATION_TILEPELS_MAX \
|
||||
((RESTORATION_TILESIZE_MAX * 3 / 2 + 2 * RESTORATION_BORDER_HORZ + 16) * \
|
||||
(RESTORATION_TILESIZE_MAX * 3 / 2 + 2 * RESTORATION_BORDER_VERT))
|
||||
#endif
|
||||
|
||||
// Two 32-bit buffers needed for the restored versions from two filters
|
||||
// TODO(debargha, rupert): Refactor to not need the large tilesize to be stored
|
||||
// on the decoder side.
|
||||
#define SGRPROJ_TMPBUF_SIZE (RESTORATION_TILEPELS_MAX * 2 * sizeof(int32_t))
|
||||
|
||||
// 4 32-bit buffers needed for the filter:
|
||||
// 2 for the restored versions of the frame and
|
||||
// 2 for each restoration operation
|
||||
#define SGRPROJ_OUTBUF_SIZE \
|
||||
((RESTORATION_TILESIZE_MAX * 3 / 2) * (RESTORATION_TILESIZE_MAX * 3 / 2 + 16))
|
||||
#define SGRPROJ_TMPBUF_SIZE \
|
||||
(RESTORATION_TILEPELS_MAX * 2 * sizeof(int32_t) + \
|
||||
SGRPROJ_OUTBUF_SIZE * 2 * sizeof(int32_t))
|
||||
#define SGRPROJ_EXTBUF_SIZE (0)
|
||||
#define SGRPROJ_PARAMS_BITS 4
|
||||
#define SGRPROJ_PARAMS (1 << SGRPROJ_PARAMS_BITS)
|
||||
|
|
@ -65,19 +124,22 @@ extern "C" {
|
|||
|
||||
#define SGRPROJ_BITS (SGRPROJ_PRJ_BITS * 2 + SGRPROJ_PARAMS_BITS)
|
||||
|
||||
#define MAX_RADIUS 3 // Only 1, 2, 3 allowed
|
||||
#define MAX_RADIUS 2 // Only 1, 2, 3 allowed
|
||||
#define MAX_EPS 80 // Max value of eps
|
||||
#define MAX_NELEM ((2 * MAX_RADIUS + 1) * (2 * MAX_RADIUS + 1))
|
||||
#define SGRPROJ_MTABLE_BITS 20
|
||||
#define SGRPROJ_RECIP_BITS 12
|
||||
|
||||
#define WIENER_HALFWIN 3
|
||||
#define WIENER_HALFWIN1 (WIENER_HALFWIN + 1)
|
||||
#define WIENER_WIN (2 * WIENER_HALFWIN + 1)
|
||||
#define WIENER_WIN2 ((WIENER_WIN) * (WIENER_WIN))
|
||||
#define WIENER_TMPBUF_SIZE (0)
|
||||
#define WIENER_EXTBUF_SIZE (0)
|
||||
|
||||
// If WIENER_WIN_CHROMA == WIENER_WIN - 2, that implies 5x5 filters are used for
|
||||
// chroma. To use 7x7 for chroma set WIENER_WIN_CHROMA to WIENER_WIN.
|
||||
#define WIENER_WIN_CHROMA (WIENER_WIN - 2)
|
||||
|
||||
#define WIENER_FILT_PREC_BITS 7
|
||||
#define WIENER_FILT_STEP (1 << WIENER_FILT_PREC_BITS)
|
||||
|
||||
|
|
@ -131,10 +193,6 @@ extern "C" {
|
|||
#if WIENER_FILT_PREC_BITS != 7
|
||||
#error "Wiener filter currently only works if WIENER_FILT_PREC_BITS == 7"
|
||||
#endif
|
||||
typedef struct {
|
||||
DECLARE_ALIGNED(16, InterpKernel, vfilter);
|
||||
DECLARE_ALIGNED(16, InterpKernel, hfilter);
|
||||
} WienerInfo;
|
||||
|
||||
typedef struct {
|
||||
#if USE_HIGHPASS_IN_SGRPROJ
|
||||
|
|
@ -148,13 +206,9 @@ typedef struct {
|
|||
int e2;
|
||||
} sgr_params_type;
|
||||
|
||||
typedef struct {
|
||||
int ep;
|
||||
int xqd[2];
|
||||
} SgrprojInfo;
|
||||
|
||||
typedef struct {
|
||||
int restoration_tilesize;
|
||||
int procunit_width, procunit_height;
|
||||
RestorationType frame_restoration_type;
|
||||
RestorationType *restoration_type;
|
||||
// Wiener filter
|
||||
|
|
@ -170,6 +224,20 @@ typedef struct {
|
|||
int tile_width, tile_height;
|
||||
int nhtiles, nvtiles;
|
||||
int32_t *tmpbuf;
|
||||
#if CONFIG_STRIPED_LOOP_RESTORATION
|
||||
int component;
|
||||
int subsampling_y;
|
||||
uint8_t *stripe_boundary_above[MAX_MB_PLANE];
|
||||
uint8_t *stripe_boundary_below[MAX_MB_PLANE];
|
||||
int stripe_boundary_stride[MAX_MB_PLANE];
|
||||
// Temporary buffers to save/restore 2 lines above/below the restoration
|
||||
// stripe
|
||||
// Allow for filter margin to left and right
|
||||
uint16_t
|
||||
tmp_save_above[2][RESTORATION_TILESIZE_MAX + 2 * RESTORATION_EXTRA_HORZ];
|
||||
uint16_t
|
||||
tmp_save_below[2][RESTORATION_TILESIZE_MAX + 2 * RESTORATION_EXTRA_HORZ];
|
||||
#endif
|
||||
} RestorationInternal;
|
||||
|
||||
static INLINE void set_default_sgrproj(SgrprojInfo *sgrproj_info) {
|
||||
|
|
@ -196,6 +264,8 @@ static INLINE int av1_get_rest_ntiles(int width, int height, int tilesize,
|
|||
int tile_width_, tile_height_;
|
||||
tile_width_ = (tilesize < 0) ? width : AOMMIN(tilesize, width);
|
||||
tile_height_ = (tilesize < 0) ? height : AOMMIN(tilesize, height);
|
||||
assert(tile_width_ > 0 && tile_height_ > 0);
|
||||
|
||||
nhtiles_ = (width + (tile_width_ >> 1)) / tile_width_;
|
||||
nvtiles_ = (height + (tile_height_ >> 1)) / tile_height_;
|
||||
if (tile_width) *tile_width = tile_width_;
|
||||
|
|
@ -205,37 +275,33 @@ static INLINE int av1_get_rest_ntiles(int width, int height, int tilesize,
|
|||
return (nhtiles_ * nvtiles_);
|
||||
}
|
||||
|
||||
static INLINE void av1_get_rest_tile_limits(
|
||||
int tile_idx, int subtile_idx, int subtile_bits, int nhtiles, int nvtiles,
|
||||
int tile_width, int tile_height, int im_width, int im_height, int clamp_h,
|
||||
int clamp_v, int *h_start, int *h_end, int *v_start, int *v_end) {
|
||||
typedef struct { int h_start, h_end, v_start, v_end; } RestorationTileLimits;
|
||||
|
||||
static INLINE RestorationTileLimits
|
||||
av1_get_rest_tile_limits(int tile_idx, int nhtiles, int nvtiles, int tile_width,
|
||||
int tile_height, int im_width,
|
||||
#if CONFIG_STRIPED_LOOP_RESTORATION
|
||||
int im_height, int subsampling_y) {
|
||||
#else
|
||||
int im_height) {
|
||||
#endif
|
||||
const int htile_idx = tile_idx % nhtiles;
|
||||
const int vtile_idx = tile_idx / nhtiles;
|
||||
*h_start = htile_idx * tile_width;
|
||||
*v_start = vtile_idx * tile_height;
|
||||
*h_end = (htile_idx < nhtiles - 1) ? *h_start + tile_width : im_width;
|
||||
*v_end = (vtile_idx < nvtiles - 1) ? *v_start + tile_height : im_height;
|
||||
if (subtile_bits) {
|
||||
const int num_subtiles_1d = (1 << subtile_bits);
|
||||
const int subtile_width = (*h_end - *h_start) >> subtile_bits;
|
||||
const int subtile_height = (*v_end - *v_start) >> subtile_bits;
|
||||
const int subtile_idx_h = subtile_idx & (num_subtiles_1d - 1);
|
||||
const int subtile_idx_v = subtile_idx >> subtile_bits;
|
||||
*h_start += subtile_idx_h * subtile_width;
|
||||
*v_start += subtile_idx_v * subtile_height;
|
||||
*h_end = subtile_idx_h == num_subtiles_1d - 1 ? *h_end
|
||||
: *h_start + subtile_width;
|
||||
*v_end = subtile_idx_v == num_subtiles_1d - 1 ? *v_end
|
||||
: *v_start + subtile_height;
|
||||
}
|
||||
if (clamp_h) {
|
||||
*h_start = AOMMAX(*h_start, clamp_h);
|
||||
*h_end = AOMMIN(*h_end, im_width - clamp_h);
|
||||
}
|
||||
if (clamp_v) {
|
||||
*v_start = AOMMAX(*v_start, clamp_v);
|
||||
*v_end = AOMMIN(*v_end, im_height - clamp_v);
|
||||
}
|
||||
RestorationTileLimits limits;
|
||||
limits.h_start = htile_idx * tile_width;
|
||||
limits.v_start = vtile_idx * tile_height;
|
||||
limits.h_end =
|
||||
(htile_idx < nhtiles - 1) ? limits.h_start + tile_width : im_width;
|
||||
limits.v_end =
|
||||
(vtile_idx < nvtiles - 1) ? limits.v_start + tile_height : im_height;
|
||||
#if CONFIG_STRIPED_LOOP_RESTORATION
|
||||
// Offset the tile upwards to align with the restoration processing stripe
|
||||
limits.v_start -= RESTORATION_TILE_OFFSET >> subsampling_y;
|
||||
if (limits.v_start < 0) limits.v_start = 0;
|
||||
if (limits.v_end < im_height)
|
||||
limits.v_end -= RESTORATION_TILE_OFFSET >> subsampling_y;
|
||||
#endif
|
||||
return limits;
|
||||
}
|
||||
|
||||
extern const sgr_params_type sgr_params[SGRPROJ_PARAMS];
|
||||
|
|
@ -248,15 +314,34 @@ int av1_alloc_restoration_struct(struct AV1Common *cm,
|
|||
int height);
|
||||
void av1_free_restoration_struct(RestorationInfo *rst_info);
|
||||
|
||||
void extend_frame(uint8_t *data, int width, int height, int stride);
|
||||
void extend_frame(uint8_t *data, int width, int height, int stride,
|
||||
int border_horz, int border_vert);
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
void extend_frame_highbd(uint16_t *data, int width, int height, int stride);
|
||||
void extend_frame_highbd(uint16_t *data, int width, int height, int stride,
|
||||
int border_horz, int border_vert);
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
void decode_xq(int *xqd, int *xq);
|
||||
void av1_loop_restoration_frame(YV12_BUFFER_CONFIG *frame, struct AV1Common *cm,
|
||||
RestorationInfo *rsi, int components_pattern,
|
||||
int partial_frame, YV12_BUFFER_CONFIG *dst);
|
||||
void av1_loop_restoration_precal();
|
||||
|
||||
// Return 1 iff the block at mi_row, mi_col with size bsize is a
|
||||
// top-level superblock containing the top-left corner of at least one
|
||||
// loop restoration tile.
|
||||
//
|
||||
// If the block is a top-level superblock, the function writes to
|
||||
// *rcol0, *rcol1, *rrow0, *rrow1. The rectangle of indices given by
|
||||
// [*rcol0, *rcol1) x [*rrow0, *rrow1) will point at the set of rtiles
|
||||
// whose top left corners lie in the superblock. Note that the set is
|
||||
// only nonempty if *rcol0 < *rcol1 and *rrow0 < *rrow1.
|
||||
int av1_loop_restoration_corners_in_sb(const struct AV1Common *cm, int plane,
|
||||
int mi_row, int mi_col, BLOCK_SIZE bsize,
|
||||
int *rcol0, int *rcol1, int *rrow0,
|
||||
int *rrow1, int *nhtiles);
|
||||
|
||||
void av1_loop_restoration_save_boundary_lines(YV12_BUFFER_CONFIG *frame,
|
||||
struct AV1Common *cm);
|
||||
#ifdef __cplusplus
|
||||
} // extern "C"
|
||||
#endif
|
||||
|
|
|
|||
2
third_party/aom/av1/common/scale.h
vendored
2
third_party/aom/av1/common/scale.h
vendored
|
|
@ -19,7 +19,7 @@
|
|||
extern "C" {
|
||||
#endif
|
||||
|
||||
#define SCALE_DENOMINATOR 16
|
||||
#define SCALE_NUMERATOR 8
|
||||
|
||||
#define REF_SCALE_SHIFT 14
|
||||
#define REF_NO_SCALE (1 << REF_SCALE_SHIFT)
|
||||
|
|
|
|||
1823
third_party/aom/av1/common/scan.c
vendored
1823
third_party/aom/av1/common/scan.c
vendored
File diff suppressed because it is too large
Load diff
34
third_party/aom/av1/common/scan.h
vendored
34
third_party/aom/av1/common/scan.h
vendored
|
|
@ -30,6 +30,9 @@ extern const SCAN_ORDER av1_intra_scan_orders[TX_SIZES_ALL][TX_TYPES];
|
|||
extern const SCAN_ORDER av1_inter_scan_orders[TX_SIZES_ALL][TX_TYPES];
|
||||
|
||||
#if CONFIG_ADAPT_SCAN
|
||||
#define USE_2X2_PROB 1
|
||||
#define USE_TOPOLOGICAL_SORT 0
|
||||
#define USE_LIMIT_SCAN_DISTANCE 0
|
||||
void av1_update_scan_count_facade(AV1_COMMON *cm, FRAME_COUNTS *counts,
|
||||
TX_SIZE tx_size, TX_TYPE tx_type,
|
||||
const tran_low_t *dqcoeffs, int max_scan);
|
||||
|
|
@ -39,6 +42,7 @@ void av1_update_scan_count_facade(AV1_COMMON *cm, FRAME_COUNTS *counts,
|
|||
// will be scanned first
|
||||
void av1_augment_prob(TX_SIZE tx_size, TX_TYPE tx_type, uint32_t *prob);
|
||||
|
||||
#if USE_TOPOLOGICAL_SORT
|
||||
// apply quick sort on nonzero probabilities to obtain a sort order
|
||||
void av1_update_sort_order(TX_SIZE tx_size, TX_TYPE tx_type,
|
||||
const uint32_t *non_zero_prob, int16_t *sort_order);
|
||||
|
|
@ -48,14 +52,24 @@ void av1_update_sort_order(TX_SIZE tx_size, TX_TYPE tx_type,
|
|||
// scanned before the to-be-scanned coefficient.
|
||||
void av1_update_scan_order(TX_SIZE tx_size, int16_t *sort_order, int16_t *scan,
|
||||
int16_t *iscan);
|
||||
#else // USE_TOPOLOGICAL_SORT
|
||||
void av1_update_scan_order(TX_SIZE tx_size, TX_TYPE tx_type,
|
||||
uint32_t *non_zero_prob, int16_t *scan,
|
||||
int16_t *iscan);
|
||||
#endif // USE_TOPOLOGICAL_SORT
|
||||
|
||||
// For each coeff_idx in scan[], update its above and left neighbors in
|
||||
// neighbors[] accordingly.
|
||||
void av1_update_neighbors(int tx_size, const int16_t *scan,
|
||||
void av1_update_neighbors(TX_SIZE tx_size, const int16_t *scan,
|
||||
const int16_t *iscan, int16_t *neighbors);
|
||||
void av1_init_scan_order(AV1_COMMON *cm);
|
||||
void av1_adapt_scan_order(AV1_COMMON *cm);
|
||||
#endif
|
||||
#if USE_2X2_PROB
|
||||
void av1_down_sample_scan_count(uint32_t *non_zero_count_ds,
|
||||
const uint32_t *non_zero_count,
|
||||
TX_SIZE tx_size);
|
||||
#endif // USE_2X2_PROB
|
||||
#endif // CONFIG_ADAPT_SCAN
|
||||
void av1_deliver_eob_threshold(const AV1_COMMON *cm, MACROBLOCKD *xd);
|
||||
|
||||
static INLINE int get_coef_context(const int16_t *neighbors,
|
||||
|
|
@ -77,6 +91,17 @@ static INLINE const SCAN_ORDER *get_default_scan(TX_SIZE tx_size,
|
|||
#endif // CONFIG_EXT_TX
|
||||
}
|
||||
|
||||
static INLINE int do_adapt_scan(TX_SIZE tx_size, TX_TYPE tx_type) {
|
||||
(void)tx_size;
|
||||
#if CONFIG_EXT_TX
|
||||
if (tx_size_2d[tx_size] >= 1024 && tx_type != DCT_DCT) return 0;
|
||||
return tx_type < IDTX;
|
||||
#else
|
||||
(void)tx_type;
|
||||
return 1;
|
||||
#endif
|
||||
}
|
||||
|
||||
static INLINE const SCAN_ORDER *get_scan(const AV1_COMMON *cm, TX_SIZE tx_size,
|
||||
TX_TYPE tx_type,
|
||||
const MB_MODE_INFO *mbmi) {
|
||||
|
|
@ -84,12 +109,15 @@ static INLINE const SCAN_ORDER *get_scan(const AV1_COMMON *cm, TX_SIZE tx_size,
|
|||
// use the DCT_DCT scan order for MRC_DCT for now
|
||||
if (tx_type == MRC_DCT) tx_type = DCT_DCT;
|
||||
#endif // CONFIG_MRC_TX
|
||||
#if CONFIG_LGT_FROM_PRED
|
||||
if (mbmi->use_lgt) tx_type = DCT_DCT;
|
||||
#endif
|
||||
const int is_inter = is_inter_block(mbmi);
|
||||
#if CONFIG_ADAPT_SCAN
|
||||
(void)mbmi;
|
||||
(void)is_inter;
|
||||
#if CONFIG_EXT_TX
|
||||
if (tx_type >= IDTX)
|
||||
if (!do_adapt_scan(tx_size, tx_type))
|
||||
return get_default_scan(tx_size, tx_type, is_inter);
|
||||
else
|
||||
#endif // CONFIG_EXT_TX
|
||||
|
|
|
|||
11
third_party/aom/av1/common/seg_common.c
vendored
11
third_party/aom/av1/common/seg_common.c
vendored
|
|
@ -16,10 +16,18 @@
|
|||
#include "av1/common/seg_common.h"
|
||||
#include "av1/common/quant_common.h"
|
||||
|
||||
#if CONFIG_LOOPFILTER_LEVEL
|
||||
static const int seg_feature_data_signed[SEG_LVL_MAX] = { 1, 1, 1, 1, 0, 0 };
|
||||
|
||||
static const int seg_feature_data_max[SEG_LVL_MAX] = {
|
||||
MAXQ, MAX_LOOP_FILTER, MAX_LOOP_FILTER, MAX_LOOP_FILTER, 0
|
||||
};
|
||||
#else
|
||||
static const int seg_feature_data_signed[SEG_LVL_MAX] = { 1, 1, 0, 0 };
|
||||
|
||||
static const int seg_feature_data_max[SEG_LVL_MAX] = { MAXQ, MAX_LOOP_FILTER, 3,
|
||||
0 };
|
||||
#endif // CONFIG_LOOPFILTER_LEVEL
|
||||
|
||||
// These functions provide access to new segment level features.
|
||||
// Eventually these function may be "optimized out" but for the moment,
|
||||
|
|
@ -46,10 +54,11 @@ int av1_is_segfeature_signed(SEG_LVL_FEATURES feature_id) {
|
|||
|
||||
void av1_set_segdata(struct segmentation *seg, int segment_id,
|
||||
SEG_LVL_FEATURES feature_id, int seg_data) {
|
||||
assert(seg_data <= seg_feature_data_max[feature_id]);
|
||||
if (seg_data < 0) {
|
||||
assert(seg_feature_data_signed[feature_id]);
|
||||
assert(-seg_data <= seg_feature_data_max[feature_id]);
|
||||
} else {
|
||||
assert(seg_data <= seg_feature_data_max[feature_id]);
|
||||
}
|
||||
|
||||
seg->feature_data[segment_id][feature_id] = seg_data;
|
||||
|
|
|
|||
27
third_party/aom/av1/common/seg_common.h
vendored
27
third_party/aom/av1/common/seg_common.h
vendored
|
|
@ -26,14 +26,37 @@ extern "C" {
|
|||
|
||||
#define PREDICTION_PROBS 3
|
||||
|
||||
#if CONFIG_LOOPFILTER_LEVEL
|
||||
typedef enum {
|
||||
SEG_LVL_ALT_Q, // Use alternate Quantizer ....
|
||||
SEG_LVL_ALT_LF_Y_V, // Use alternate loop filter value on y plane vertical
|
||||
SEG_LVL_ALT_LF_Y_H, // Use alternate loop filter value on y plane horizontal
|
||||
SEG_LVL_ALT_LF_U, // Use alternate loop filter value on u plane
|
||||
SEG_LVL_ALT_LF_V, // Use alternate loop filter value on v plane
|
||||
SEG_LVL_REF_FRAME, // Optional Segment reference frame
|
||||
SEG_LVL_SKIP, // Optional Segment (0,0) + skip mode
|
||||
#if CONFIG_SEGMENT_ZEROMV
|
||||
SEG_LVL_ZEROMV,
|
||||
SEG_LVL_MAX
|
||||
#else
|
||||
SEG_LVL_MAX
|
||||
#endif
|
||||
} SEG_LVL_FEATURES;
|
||||
#else // CONFIG_LOOPFILTER_LEVEL
|
||||
// Segment level features.
|
||||
typedef enum {
|
||||
SEG_LVL_ALT_Q = 0, // Use alternate Quantizer ....
|
||||
SEG_LVL_ALT_LF = 1, // Use alternate loop filter value...
|
||||
SEG_LVL_REF_FRAME = 2, // Optional Segment reference frame
|
||||
SEG_LVL_SKIP = 3, // Optional Segment (0,0) + skip mode
|
||||
SEG_LVL_MAX = 4 // Number of features supported
|
||||
SEG_LVL_SKIP = 3, // Optional Segment (0,0) + skip mode
|
||||
#if CONFIG_SEGMENT_ZEROMV
|
||||
SEG_LVL_ZEROMV = 4,
|
||||
SEG_LVL_MAX = 5
|
||||
#else
|
||||
SEG_LVL_MAX = 4
|
||||
#endif
|
||||
} SEG_LVL_FEATURES;
|
||||
#endif // CONFIG_LOOPFILTER_LEVEL
|
||||
|
||||
struct segmentation {
|
||||
uint8_t enabled;
|
||||
|
|
|
|||
29
third_party/aom/av1/common/thread_common.c
vendored
29
third_party/aom/av1/common/thread_common.c
vendored
|
|
@ -290,6 +290,13 @@ static void loop_filter_rows_mt(YV12_BUFFER_CONFIG *frame, AV1_COMMON *cm,
|
|||
int start, int stop, int y_only,
|
||||
AVxWorker *workers, int nworkers,
|
||||
AV1LfSync *lf_sync) {
|
||||
#if CONFIG_EXT_PARTITION
|
||||
printf(
|
||||
"STOPPING: This code has not been modified to work with the "
|
||||
"extended coding unit size experiment");
|
||||
exit(EXIT_FAILURE);
|
||||
#endif // CONFIG_EXT_PARTITION
|
||||
|
||||
const AVxWorkerInterface *const winterface = aom_get_worker_interface();
|
||||
// Number of superblock rows and cols
|
||||
const int sb_rows = mi_rows_aligned_to_sb(cm) >> cm->mib_size_log2;
|
||||
|
|
@ -299,13 +306,6 @@ static void loop_filter_rows_mt(YV12_BUFFER_CONFIG *frame, AV1_COMMON *cm,
|
|||
const int num_workers = AOMMIN(nworkers, tile_cols);
|
||||
int i;
|
||||
|
||||
#if CONFIG_EXT_PARTITION
|
||||
printf(
|
||||
"STOPPING: This code has not been modified to work with the "
|
||||
"extended coding unit size experiment");
|
||||
exit(EXIT_FAILURE);
|
||||
#endif // CONFIG_EXT_PARTITION
|
||||
|
||||
if (!lf_sync->sync_range || sb_rows != lf_sync->rows ||
|
||||
num_workers > lf_sync->num_workers) {
|
||||
av1_loop_filter_dealloc(lf_sync);
|
||||
|
|
@ -416,8 +416,11 @@ static void loop_filter_rows_mt(YV12_BUFFER_CONFIG *frame, AV1_COMMON *cm,
|
|||
|
||||
void av1_loop_filter_frame_mt(YV12_BUFFER_CONFIG *frame, AV1_COMMON *cm,
|
||||
struct macroblockd_plane planes[MAX_MB_PLANE],
|
||||
int frame_filter_level, int y_only,
|
||||
int partial_frame, AVxWorker *workers,
|
||||
int frame_filter_level,
|
||||
#if CONFIG_LOOPFILTER_LEVEL
|
||||
int frame_filter_level_r,
|
||||
#endif
|
||||
int y_only, int partial_frame, AVxWorker *workers,
|
||||
int num_workers, AV1LfSync *lf_sync) {
|
||||
int start_mi_row, end_mi_row, mi_rows_to_filter;
|
||||
|
||||
|
|
@ -431,8 +434,12 @@ void av1_loop_filter_frame_mt(YV12_BUFFER_CONFIG *frame, AV1_COMMON *cm,
|
|||
mi_rows_to_filter = AOMMAX(cm->mi_rows / 8, 8);
|
||||
}
|
||||
end_mi_row = start_mi_row + mi_rows_to_filter;
|
||||
av1_loop_filter_frame_init(cm, frame_filter_level);
|
||||
|
||||
#if CONFIG_LOOPFILTER_LEVEL
|
||||
av1_loop_filter_frame_init(cm, frame_filter_level, frame_filter_level_r,
|
||||
y_only);
|
||||
#else
|
||||
av1_loop_filter_frame_init(cm, frame_filter_level, frame_filter_level);
|
||||
#endif // CONFIG_LOOPFILTER_LEVEL
|
||||
loop_filter_rows_mt(frame, cm, planes, start_mi_row, end_mi_row, y_only,
|
||||
workers, num_workers, lf_sync);
|
||||
}
|
||||
|
|
|
|||
7
third_party/aom/av1/common/thread_common.h
vendored
7
third_party/aom/av1/common/thread_common.h
vendored
|
|
@ -50,8 +50,11 @@ void av1_loop_filter_dealloc(AV1LfSync *lf_sync);
|
|||
// Multi-threaded loopfilter that uses the tile threads.
|
||||
void av1_loop_filter_frame_mt(YV12_BUFFER_CONFIG *frame, struct AV1Common *cm,
|
||||
struct macroblockd_plane planes[MAX_MB_PLANE],
|
||||
int frame_filter_level, int y_only,
|
||||
int partial_frame, AVxWorker *workers,
|
||||
int frame_filter_level,
|
||||
#if CONFIG_LOOPFILTER_LEVEL
|
||||
int frame_filter_level_r,
|
||||
#endif
|
||||
int y_only, int partial_frame, AVxWorker *workers,
|
||||
int num_workers, AV1LfSync *lf_sync);
|
||||
|
||||
void av1_accumulate_frame_counts(struct FRAME_COUNTS *acc_counts,
|
||||
|
|
|
|||
184
third_party/aom/av1/common/tile_common.c
vendored
184
third_party/aom/av1/common/tile_common.c
vendored
|
|
@ -13,29 +13,18 @@
|
|||
#include "av1/common/onyxc_int.h"
|
||||
#include "aom_dsp/aom_dsp_common.h"
|
||||
|
||||
void av1_tile_set_row(TileInfo *tile, const AV1_COMMON *cm, int row) {
|
||||
tile->mi_row_start = row * cm->tile_height;
|
||||
tile->mi_row_end = AOMMIN(tile->mi_row_start + cm->tile_height, cm->mi_rows);
|
||||
}
|
||||
|
||||
void av1_tile_set_col(TileInfo *tile, const AV1_COMMON *cm, int col) {
|
||||
tile->mi_col_start = col * cm->tile_width;
|
||||
tile->mi_col_end = AOMMIN(tile->mi_col_start + cm->tile_width, cm->mi_cols);
|
||||
}
|
||||
|
||||
#if CONFIG_DEPENDENT_HORZTILES
|
||||
void av1_tile_set_tg_boundary(TileInfo *tile, const AV1_COMMON *const cm,
|
||||
int row, int col) {
|
||||
if (row < cm->tile_rows - 1) {
|
||||
tile->tg_horz_boundary =
|
||||
col >= cm->tile_group_start_col[row][col]
|
||||
? (row == cm->tile_group_start_row[row][col] ? 1 : 0)
|
||||
: (row == cm->tile_group_start_row[row + 1][col] ? 1 : 0);
|
||||
} else {
|
||||
assert(col >= cm->tile_group_start_col[row][col]);
|
||||
tile->tg_horz_boundary =
|
||||
(row == cm->tile_group_start_row[row][col] ? 1 : 0);
|
||||
const int tg_start_row = cm->tile_group_start_row[row][col];
|
||||
const int tg_start_col = cm->tile_group_start_col[row][col];
|
||||
tile->tg_horz_boundary = ((row == tg_start_row && col >= tg_start_col) ||
|
||||
(row == tg_start_row + 1 && col < tg_start_col));
|
||||
#if CONFIG_MAX_TILE
|
||||
if (cm->tile_row_independent[row]) {
|
||||
tile->tg_horz_boundary = 1; // this tile row is independent
|
||||
}
|
||||
#endif
|
||||
}
|
||||
#endif
|
||||
void av1_tile_init(TileInfo *tile, const AV1_COMMON *cm, int row, int col) {
|
||||
|
|
@ -46,6 +35,125 @@ void av1_tile_init(TileInfo *tile, const AV1_COMMON *cm, int row, int col) {
|
|||
#endif
|
||||
}
|
||||
|
||||
#if CONFIG_MAX_TILE
|
||||
|
||||
// Find smallest k>=0 such that (blk_size << k) >= target
|
||||
static int tile_log2(int blk_size, int target) {
|
||||
int k;
|
||||
for (k = 0; (blk_size << k) < target; k++) {
|
||||
}
|
||||
return k;
|
||||
}
|
||||
|
||||
void av1_get_tile_limits(AV1_COMMON *const cm) {
|
||||
int mi_cols = ALIGN_POWER_OF_TWO(cm->mi_cols, MAX_MIB_SIZE_LOG2);
|
||||
int mi_rows = ALIGN_POWER_OF_TWO(cm->mi_rows, MAX_MIB_SIZE_LOG2);
|
||||
int sb_cols = mi_cols >> MAX_MIB_SIZE_LOG2;
|
||||
int sb_rows = mi_rows >> MAX_MIB_SIZE_LOG2;
|
||||
|
||||
cm->min_log2_tile_cols = tile_log2(MAX_TILE_WIDTH_SB, sb_cols);
|
||||
cm->max_log2_tile_cols = tile_log2(1, AOMMIN(sb_cols, MAX_TILE_COLS));
|
||||
cm->max_log2_tile_rows = tile_log2(1, AOMMIN(sb_rows, MAX_TILE_ROWS));
|
||||
cm->min_log2_tiles = tile_log2(MAX_TILE_AREA_SB, sb_cols * sb_rows);
|
||||
cm->min_log2_tiles = AOMMAX(cm->min_log2_tiles, cm->min_log2_tile_cols);
|
||||
// TODO(dominic.symes@arm.com):
|
||||
// Add in levelMinLog2Tiles as a lower limit when levels are defined
|
||||
}
|
||||
|
||||
void av1_calculate_tile_cols(AV1_COMMON *const cm) {
|
||||
int mi_cols = ALIGN_POWER_OF_TWO(cm->mi_cols, MAX_MIB_SIZE_LOG2);
|
||||
int mi_rows = ALIGN_POWER_OF_TWO(cm->mi_rows, MAX_MIB_SIZE_LOG2);
|
||||
int sb_cols = mi_cols >> MAX_MIB_SIZE_LOG2;
|
||||
int sb_rows = mi_rows >> MAX_MIB_SIZE_LOG2;
|
||||
int i;
|
||||
|
||||
if (cm->uniform_tile_spacing_flag) {
|
||||
int start_sb;
|
||||
int size_sb = ALIGN_POWER_OF_TWO(sb_cols, cm->log2_tile_cols);
|
||||
size_sb >>= cm->log2_tile_cols;
|
||||
assert(size_sb > 0);
|
||||
for (i = 0, start_sb = 0; start_sb < sb_cols; i++) {
|
||||
cm->tile_col_start_sb[i] = start_sb;
|
||||
start_sb += size_sb;
|
||||
}
|
||||
cm->tile_cols = i;
|
||||
cm->tile_col_start_sb[i] = sb_cols;
|
||||
cm->min_log2_tile_rows = AOMMAX(cm->min_log2_tiles - cm->log2_tile_cols, 0);
|
||||
cm->max_tile_height_sb = sb_rows >> cm->min_log2_tile_rows;
|
||||
} else {
|
||||
int max_tile_area_sb = (sb_rows * sb_cols);
|
||||
int max_tile_width_sb = 0;
|
||||
cm->log2_tile_cols = tile_log2(1, cm->tile_cols);
|
||||
for (i = 0; i < cm->tile_cols; i++) {
|
||||
int size_sb = cm->tile_col_start_sb[i + 1] - cm->tile_col_start_sb[i];
|
||||
max_tile_width_sb = AOMMAX(max_tile_width_sb, size_sb);
|
||||
}
|
||||
if (cm->min_log2_tiles) {
|
||||
max_tile_area_sb >>= (cm->min_log2_tiles + 1);
|
||||
}
|
||||
cm->max_tile_height_sb = AOMMAX(max_tile_area_sb / max_tile_width_sb, 1);
|
||||
}
|
||||
}
|
||||
|
||||
void av1_calculate_tile_rows(AV1_COMMON *const cm) {
|
||||
int mi_rows = ALIGN_POWER_OF_TWO(cm->mi_rows, MAX_MIB_SIZE_LOG2);
|
||||
int sb_rows = mi_rows >> MAX_MIB_SIZE_LOG2;
|
||||
int start_sb, size_sb, i;
|
||||
|
||||
if (cm->uniform_tile_spacing_flag) {
|
||||
size_sb = ALIGN_POWER_OF_TWO(sb_rows, cm->log2_tile_rows);
|
||||
size_sb >>= cm->log2_tile_rows;
|
||||
assert(size_sb > 0);
|
||||
for (i = 0, start_sb = 0; start_sb < sb_rows; i++) {
|
||||
cm->tile_row_start_sb[i] = start_sb;
|
||||
start_sb += size_sb;
|
||||
}
|
||||
cm->tile_rows = i;
|
||||
cm->tile_row_start_sb[i] = sb_rows;
|
||||
} else {
|
||||
cm->log2_tile_rows = tile_log2(1, cm->tile_rows);
|
||||
}
|
||||
|
||||
#if CONFIG_DEPENDENT_HORZTILES
|
||||
// Record which tile rows must be indpendent for parallelism
|
||||
for (i = 0, start_sb = 0; i < cm->tile_rows; i++) {
|
||||
cm->tile_row_independent[i] = 0;
|
||||
if (cm->tile_row_start_sb[i + 1] - start_sb > cm->max_tile_height_sb) {
|
||||
cm->tile_row_independent[i] = 1;
|
||||
start_sb = cm->tile_row_start_sb[i];
|
||||
}
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
void av1_tile_set_row(TileInfo *tile, const AV1_COMMON *cm, int row) {
|
||||
assert(row < cm->tile_rows);
|
||||
int mi_row_start = cm->tile_row_start_sb[row] << MAX_MIB_SIZE_LOG2;
|
||||
int mi_row_end = cm->tile_row_start_sb[row + 1] << MAX_MIB_SIZE_LOG2;
|
||||
tile->mi_row_start = mi_row_start;
|
||||
tile->mi_row_end = AOMMIN(mi_row_end, cm->mi_rows);
|
||||
}
|
||||
|
||||
void av1_tile_set_col(TileInfo *tile, const AV1_COMMON *cm, int col) {
|
||||
assert(col < cm->tile_cols);
|
||||
int mi_col_start = cm->tile_col_start_sb[col] << MAX_MIB_SIZE_LOG2;
|
||||
int mi_col_end = cm->tile_col_start_sb[col + 1] << MAX_MIB_SIZE_LOG2;
|
||||
tile->mi_col_start = mi_col_start;
|
||||
tile->mi_col_end = AOMMIN(mi_col_end, cm->mi_cols);
|
||||
}
|
||||
|
||||
#else
|
||||
|
||||
void av1_tile_set_row(TileInfo *tile, const AV1_COMMON *cm, int row) {
|
||||
tile->mi_row_start = row * cm->tile_height;
|
||||
tile->mi_row_end = AOMMIN(tile->mi_row_start + cm->tile_height, cm->mi_rows);
|
||||
}
|
||||
|
||||
void av1_tile_set_col(TileInfo *tile, const AV1_COMMON *cm, int col) {
|
||||
tile->mi_col_start = col * cm->tile_width;
|
||||
tile->mi_col_end = AOMMIN(tile->mi_col_start + cm->tile_width, cm->mi_cols);
|
||||
}
|
||||
|
||||
#if CONFIG_EXT_PARTITION
|
||||
#define MIN_TILE_WIDTH_MAX_SB 2
|
||||
#define MAX_TILE_WIDTH_MAX_SB 32
|
||||
|
|
@ -74,6 +182,7 @@ void av1_get_tile_n_bits(int mi_cols, int *min_log2_tile_cols,
|
|||
*max_log2_tile_cols = get_max_log2_tile_cols(max_sb_cols);
|
||||
assert(*min_log2_tile_cols <= *max_log2_tile_cols);
|
||||
}
|
||||
#endif // CONFIG_MAX_TILE
|
||||
|
||||
void av1_setup_frame_boundary_info(const AV1_COMMON *const cm) {
|
||||
MODE_INFO *mi = cm->mi;
|
||||
|
|
@ -103,16 +212,38 @@ void av1_setup_frame_boundary_info(const AV1_COMMON *const cm) {
|
|||
}
|
||||
}
|
||||
|
||||
int get_tile_size(int mi_frame_size, int log2_tile_num, int *ntiles) {
|
||||
// Round the frame up to a whole number of max superblocks
|
||||
mi_frame_size = ALIGN_POWER_OF_TWO(mi_frame_size, MAX_MIB_SIZE_LOG2);
|
||||
|
||||
// Divide by the signalled number of tiles, rounding up to the multiple of
|
||||
// the max superblock size. To do this, shift right (and round up) to get the
|
||||
// tile size in max super-blocks and then shift left again to convert it to
|
||||
// mi units.
|
||||
const int shift = log2_tile_num + MAX_MIB_SIZE_LOG2;
|
||||
const int max_sb_tile_size =
|
||||
ALIGN_POWER_OF_TWO(mi_frame_size, shift) >> shift;
|
||||
const int mi_tile_size = max_sb_tile_size << MAX_MIB_SIZE_LOG2;
|
||||
|
||||
// The actual number of tiles is the ceiling of the frame size in mi units
|
||||
// divided by mi_size. This is at most 1 << log2_tile_num but might be
|
||||
// strictly less if max_sb_tile_size got rounded up significantly.
|
||||
if (ntiles) {
|
||||
*ntiles = (mi_frame_size + mi_tile_size - 1) / mi_tile_size;
|
||||
assert(*ntiles <= (1 << log2_tile_num));
|
||||
}
|
||||
|
||||
return mi_tile_size;
|
||||
}
|
||||
|
||||
#if CONFIG_LOOPFILTERING_ACROSS_TILES
|
||||
void av1_setup_across_tile_boundary_info(const AV1_COMMON *const cm,
|
||||
const TileInfo *const tile_info) {
|
||||
int lpf_across_tiles_enabled = 1;
|
||||
#if CONFIG_LOOPFILTERING_ACROSS_TILES
|
||||
lpf_across_tiles_enabled = cm->loop_filter_across_tiles_enabled;
|
||||
#endif
|
||||
if ((cm->tile_cols * cm->tile_rows > 1) && (!lpf_across_tiles_enabled)) {
|
||||
if (cm->tile_cols * cm->tile_rows > 1) {
|
||||
const int mi_row = tile_info->mi_row_start;
|
||||
const int mi_col = tile_info->mi_col_start;
|
||||
MODE_INFO *const mi_start = cm->mi + mi_row * cm->mi_stride + mi_col;
|
||||
assert(mi_start < cm->mip + cm->mi_alloc_size);
|
||||
MODE_INFO *mi = 0;
|
||||
const int row_diff = tile_info->mi_row_end - tile_info->mi_row_start;
|
||||
const int col_diff = tile_info->mi_col_end - tile_info->mi_col_start;
|
||||
|
|
@ -136,6 +267,10 @@ void av1_setup_across_tile_boundary_info(const AV1_COMMON *const cm,
|
|||
}
|
||||
|
||||
mi = mi_start + (row_diff - 1) * cm->mi_stride;
|
||||
|
||||
// explicit bounds checking
|
||||
assert(mi + col_diff <= cm->mip + cm->mi_alloc_size);
|
||||
|
||||
for (col = 0; col < col_diff; ++col) {
|
||||
mi->mbmi.boundary_info |= TILE_BOTTOM_BOUNDARY;
|
||||
mi += 1;
|
||||
|
|
@ -149,7 +284,6 @@ void av1_setup_across_tile_boundary_info(const AV1_COMMON *const cm,
|
|||
}
|
||||
}
|
||||
|
||||
#if CONFIG_LOOPFILTERING_ACROSS_TILES
|
||||
int av1_disable_loopfilter_on_tile_boundary(const struct AV1Common *cm) {
|
||||
return (!cm->loop_filter_across_tiles_enabled &&
|
||||
(cm->tile_cols * cm->tile_rows > 1));
|
||||
|
|
|
|||
23
third_party/aom/av1/common/tile_common.h
vendored
23
third_party/aom/av1/common/tile_common.h
vendored
|
|
@ -43,13 +43,32 @@ void av1_get_tile_n_bits(int mi_cols, int *min_log2_tile_cols,
|
|||
int *max_log2_tile_cols);
|
||||
|
||||
void av1_setup_frame_boundary_info(const struct AV1Common *const cm);
|
||||
void av1_setup_across_tile_boundary_info(const struct AV1Common *const cm,
|
||||
const TileInfo *const tile_info);
|
||||
|
||||
// Calculate the correct tile size (width or height) for (1 << log2_tile_num)
|
||||
// tiles horizontally or vertically in the frame.
|
||||
int get_tile_size(int mi_frame_size, int log2_tile_num, int *ntiles);
|
||||
|
||||
#if CONFIG_LOOPFILTERING_ACROSS_TILES
|
||||
void av1_setup_across_tile_boundary_info(const struct AV1Common *const cm,
|
||||
const TileInfo *const tile_info);
|
||||
int av1_disable_loopfilter_on_tile_boundary(const struct AV1Common *cm);
|
||||
#endif // CONFIG_LOOPFILTERING_ACROSS_TILES
|
||||
|
||||
#if CONFIG_MAX_TILE
|
||||
|
||||
// Define tile maximum width and area
|
||||
// There is no maximum height since height is limited by area and width limits
|
||||
// The minimum tile width or height is fixed at one superblock
|
||||
#define MAX_TILE_WIDTH (4096) // Max Tile width in pixels
|
||||
#define MAX_TILE_WIDTH_SB (MAX_TILE_WIDTH >> MAX_SB_SIZE_LOG2)
|
||||
#define MAX_TILE_AREA (4096 * 2304) // Maximum tile area in pixels
|
||||
#define MAX_TILE_AREA_SB (MAX_TILE_AREA >> (2 * MAX_SB_SIZE_LOG2))
|
||||
|
||||
void av1_get_tile_limits(struct AV1Common *const cm);
|
||||
void av1_calculate_tile_cols(struct AV1Common *const cm);
|
||||
void av1_calculate_tile_rows(struct AV1Common *const cm);
|
||||
#endif
|
||||
|
||||
#ifdef __cplusplus
|
||||
} // extern "C"
|
||||
#endif
|
||||
|
|
|
|||
5253
third_party/aom/av1/common/token_cdfs.h
vendored
Normal file
5253
third_party/aom/av1/common/token_cdfs.h
vendored
Normal file
File diff suppressed because it is too large
Load diff
176
third_party/aom/av1/common/txb_common.c
vendored
176
third_party/aom/av1/common/txb_common.c
vendored
|
|
@ -10,6 +10,7 @@
|
|||
*/
|
||||
#include "aom/aom_integer.h"
|
||||
#include "av1/common/onyxc_int.h"
|
||||
#include "av1/common/txb_common.h"
|
||||
|
||||
const int16_t av1_coeff_band_4x4[16] = { 0, 1, 2, 3, 4, 5, 6, 7,
|
||||
8, 9, 10, 11, 12, 13, 14, 15 };
|
||||
|
|
@ -95,6 +96,123 @@ const int16_t av1_coeff_band_32x32[1024] = {
|
|||
22, 23, 23, 23, 23, 23, 23, 23, 23, 24, 24, 24, 24, 24, 24, 24, 24,
|
||||
};
|
||||
|
||||
#if LV_MAP_PROB
|
||||
void av1_init_txb_probs(FRAME_CONTEXT *fc) {
|
||||
TX_SIZE tx_size;
|
||||
int plane, ctx, level;
|
||||
|
||||
// Update probability models for transform block skip flag
|
||||
for (tx_size = 0; tx_size < TX_SIZES; ++tx_size) {
|
||||
for (ctx = 0; ctx < TXB_SKIP_CONTEXTS; ++ctx) {
|
||||
fc->txb_skip_cdf[tx_size][ctx][0] =
|
||||
AOM_ICDF(128 * (aom_cdf_prob)fc->txb_skip[tx_size][ctx]);
|
||||
fc->txb_skip_cdf[tx_size][ctx][1] = AOM_ICDF(32768);
|
||||
fc->txb_skip_cdf[tx_size][ctx][2] = 0;
|
||||
}
|
||||
}
|
||||
|
||||
for (plane = 0; plane < PLANE_TYPES; ++plane) {
|
||||
for (ctx = 0; ctx < DC_SIGN_CONTEXTS; ++ctx) {
|
||||
fc->dc_sign_cdf[plane][ctx][0] =
|
||||
AOM_ICDF(128 * (aom_cdf_prob)fc->dc_sign[plane][ctx]);
|
||||
fc->dc_sign_cdf[plane][ctx][1] = AOM_ICDF(32768);
|
||||
fc->dc_sign_cdf[plane][ctx][2] = 0;
|
||||
}
|
||||
}
|
||||
|
||||
// Update probability models for non-zero coefficient map and eob flag.
|
||||
for (tx_size = 0; tx_size < TX_SIZES; ++tx_size) {
|
||||
for (plane = 0; plane < PLANE_TYPES; ++plane) {
|
||||
for (level = 0; level < NUM_BASE_LEVELS; ++level) {
|
||||
for (ctx = 0; ctx < COEFF_BASE_CONTEXTS; ++ctx) {
|
||||
fc->coeff_base_cdf[tx_size][plane][level][ctx][0] = AOM_ICDF(
|
||||
128 * (aom_cdf_prob)fc->coeff_base[tx_size][plane][level][ctx]);
|
||||
fc->coeff_base_cdf[tx_size][plane][level][ctx][1] = AOM_ICDF(32768);
|
||||
fc->coeff_base_cdf[tx_size][plane][level][ctx][2] = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (tx_size = 0; tx_size < TX_SIZES; ++tx_size) {
|
||||
for (plane = 0; plane < PLANE_TYPES; ++plane) {
|
||||
for (ctx = 0; ctx < SIG_COEF_CONTEXTS; ++ctx) {
|
||||
fc->nz_map_cdf[tx_size][plane][ctx][0] =
|
||||
AOM_ICDF(128 * (aom_cdf_prob)fc->nz_map[tx_size][plane][ctx]);
|
||||
fc->nz_map_cdf[tx_size][plane][ctx][1] = AOM_ICDF(32768);
|
||||
fc->nz_map_cdf[tx_size][plane][ctx][2] = 0;
|
||||
}
|
||||
|
||||
for (ctx = 0; ctx < EOB_COEF_CONTEXTS; ++ctx) {
|
||||
fc->eob_flag_cdf[tx_size][plane][ctx][0] =
|
||||
AOM_ICDF(128 * (aom_cdf_prob)fc->eob_flag[tx_size][plane][ctx]);
|
||||
fc->eob_flag_cdf[tx_size][plane][ctx][1] = AOM_ICDF(32768);
|
||||
fc->eob_flag_cdf[tx_size][plane][ctx][2] = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (tx_size = 0; tx_size < TX_SIZES; ++tx_size) {
|
||||
for (plane = 0; plane < PLANE_TYPES; ++plane) {
|
||||
for (ctx = 0; ctx < LEVEL_CONTEXTS; ++ctx) {
|
||||
fc->coeff_lps_cdf[tx_size][plane][ctx][0] =
|
||||
AOM_ICDF(128 * (aom_cdf_prob)fc->coeff_lps[tx_size][plane][ctx]);
|
||||
fc->coeff_lps_cdf[tx_size][plane][ctx][1] = AOM_ICDF(32768);
|
||||
fc->coeff_lps_cdf[tx_size][plane][ctx][2] = 0;
|
||||
}
|
||||
#if BR_NODE
|
||||
for (int br = 0; br < BASE_RANGE_SETS; ++br) {
|
||||
for (ctx = 0; ctx < LEVEL_CONTEXTS; ++ctx) {
|
||||
fc->coeff_br_cdf[tx_size][plane][br][ctx][0] = AOM_ICDF(
|
||||
128 * (aom_cdf_prob)fc->coeff_br[tx_size][plane][br][ctx]);
|
||||
fc->coeff_br_cdf[tx_size][plane][br][ctx][1] = AOM_ICDF(32768);
|
||||
fc->coeff_br_cdf[tx_size][plane][br][ctx][2] = 0;
|
||||
}
|
||||
}
|
||||
#endif // BR_NODE
|
||||
}
|
||||
}
|
||||
#if CONFIG_CTX1D
|
||||
for (tx_size = 0; tx_size < TX_SIZES; ++tx_size) {
|
||||
for (plane = 0; plane < PLANE_TYPES; ++plane) {
|
||||
for (int tx_class = 0; tx_class < TX_CLASSES; ++tx_class) {
|
||||
fc->eob_mode_cdf[tx_size][plane][tx_class][0] = AOM_ICDF(
|
||||
128 * (aom_cdf_prob)fc->eob_mode[tx_size][plane][tx_class]);
|
||||
fc->eob_mode_cdf[tx_size][plane][tx_class][1] = AOM_ICDF(32768);
|
||||
fc->eob_mode_cdf[tx_size][plane][tx_class][2] = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
for (tx_size = 0; tx_size < TX_SIZES; ++tx_size) {
|
||||
for (plane = 0; plane < PLANE_TYPES; ++plane) {
|
||||
for (int tx_class = 0; tx_class < TX_CLASSES; ++tx_class) {
|
||||
for (ctx = 0; ctx < EMPTY_LINE_CONTEXTS; ++ctx) {
|
||||
fc->empty_line_cdf[tx_size][plane][tx_class][ctx][0] = AOM_ICDF(
|
||||
128 *
|
||||
(aom_cdf_prob)fc->empty_line[tx_size][plane][tx_class][ctx]);
|
||||
fc->empty_line_cdf[tx_size][plane][tx_class][ctx][1] =
|
||||
AOM_ICDF(32768);
|
||||
fc->empty_line_cdf[tx_size][plane][tx_class][ctx][2] = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
for (tx_size = 0; tx_size < TX_SIZES; ++tx_size) {
|
||||
for (plane = 0; plane < PLANE_TYPES; ++plane) {
|
||||
for (int tx_class = 0; tx_class < TX_CLASSES; ++tx_class) {
|
||||
for (ctx = 0; ctx < HV_EOB_CONTEXTS; ++ctx) {
|
||||
fc->hv_eob_cdf[tx_size][plane][tx_class][ctx][0] = AOM_ICDF(
|
||||
128 * (aom_cdf_prob)fc->hv_eob[tx_size][plane][tx_class][ctx]);
|
||||
fc->hv_eob_cdf[tx_size][plane][tx_class][ctx][1] = AOM_ICDF(32768);
|
||||
fc->hv_eob_cdf[tx_size][plane][tx_class][ctx][2] = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
#endif // CONFIG_CTX1D
|
||||
}
|
||||
#endif // LV_MAP_PROB
|
||||
|
||||
void av1_adapt_txb_probs(AV1_COMMON *cm, unsigned int count_sat,
|
||||
unsigned int update_factor) {
|
||||
FRAME_CONTEXT *fc = cm->fc;
|
||||
|
|
@ -141,10 +259,64 @@ void av1_adapt_txb_probs(AV1_COMMON *cm, unsigned int count_sat,
|
|||
}
|
||||
|
||||
for (tx_size = 0; tx_size < TX_SIZES; ++tx_size) {
|
||||
for (plane = 0; plane < PLANE_TYPES; ++plane)
|
||||
for (ctx = 0; ctx < LEVEL_CONTEXTS; ++ctx)
|
||||
for (plane = 0; plane < PLANE_TYPES; ++plane) {
|
||||
for (ctx = 0; ctx < LEVEL_CONTEXTS; ++ctx) {
|
||||
fc->coeff_lps[tx_size][plane][ctx] = merge_probs(
|
||||
pre_fc->coeff_lps[tx_size][plane][ctx],
|
||||
counts->coeff_lps[tx_size][plane][ctx], count_sat, update_factor);
|
||||
}
|
||||
#if BR_NODE
|
||||
for (int br = 0; br < BASE_RANGE_SETS; ++br) {
|
||||
for (ctx = 0; ctx < LEVEL_CONTEXTS; ++ctx) {
|
||||
fc->coeff_br[tx_size][plane][br][ctx] =
|
||||
merge_probs(pre_fc->coeff_br[tx_size][plane][br][ctx],
|
||||
counts->coeff_br[tx_size][plane][br][ctx], count_sat,
|
||||
update_factor);
|
||||
}
|
||||
}
|
||||
#endif // BR_NODE
|
||||
}
|
||||
}
|
||||
#if CONFIG_CTX1D
|
||||
for (tx_size = 0; tx_size < TX_SIZES; ++tx_size) {
|
||||
for (plane = 0; plane < PLANE_TYPES; ++plane)
|
||||
for (int tx_class = 0; tx_class < TX_CLASSES; ++tx_class)
|
||||
fc->eob_mode[tx_size][plane][tx_class] =
|
||||
merge_probs(pre_fc->eob_mode[tx_size][plane][tx_class],
|
||||
counts->eob_mode[tx_size][plane][tx_class], count_sat,
|
||||
update_factor);
|
||||
}
|
||||
for (tx_size = 0; tx_size < TX_SIZES; ++tx_size) {
|
||||
for (plane = 0; plane < PLANE_TYPES; ++plane)
|
||||
for (int tx_class = 0; tx_class < TX_CLASSES; ++tx_class)
|
||||
for (ctx = 0; ctx < EMPTY_LINE_CONTEXTS; ++ctx)
|
||||
fc->empty_line[tx_size][plane][tx_class][ctx] =
|
||||
merge_probs(pre_fc->empty_line[tx_size][plane][tx_class][ctx],
|
||||
counts->empty_line[tx_size][plane][tx_class][ctx],
|
||||
count_sat, update_factor);
|
||||
}
|
||||
for (tx_size = 0; tx_size < TX_SIZES; ++tx_size) {
|
||||
for (plane = 0; plane < PLANE_TYPES; ++plane)
|
||||
for (int tx_class = 0; tx_class < TX_CLASSES; ++tx_class)
|
||||
for (ctx = 0; ctx < HV_EOB_CONTEXTS; ++ctx)
|
||||
fc->hv_eob[tx_size][plane][tx_class][ctx] =
|
||||
merge_probs(pre_fc->hv_eob[tx_size][plane][tx_class][ctx],
|
||||
counts->hv_eob[tx_size][plane][tx_class][ctx],
|
||||
count_sat, update_factor);
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
void av1_init_lv_map(AV1_COMMON *cm) {
|
||||
LV_MAP_CTX_TABLE *coeff_ctx_table = &cm->coeff_ctx_table;
|
||||
for (int row = 0; row < 2; ++row) {
|
||||
for (int col = 0; col < 2; ++col) {
|
||||
for (int sig_mag = 0; sig_mag < 2; ++sig_mag) {
|
||||
for (int count = 0; count < BASE_CONTEXT_POSITION_NUM + 1; ++count) {
|
||||
coeff_ctx_table->base_ctx_table[row][col][sig_mag][count] =
|
||||
get_base_ctx_from_count_mag(row, col, count, sig_mag);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
|
|
|||
351
third_party/aom/av1/common/txb_common.h
vendored
351
third_party/aom/av1/common/txb_common.h
vendored
|
|
@ -11,6 +11,10 @@
|
|||
|
||||
#ifndef AV1_COMMON_TXB_COMMON_H_
|
||||
#define AV1_COMMON_TXB_COMMON_H_
|
||||
|
||||
#define REDUCE_CONTEXT_DEPENDENCY 0
|
||||
#define MIN_SCAN_IDX_REDUCE_CONTEXT_DEPENDENCY 0
|
||||
|
||||
extern const int16_t av1_coeff_band_4x4[16];
|
||||
|
||||
extern const int16_t av1_coeff_band_8x8[64];
|
||||
|
|
@ -28,7 +32,6 @@ static INLINE TX_SIZE get_txsize_context(TX_SIZE tx_size) {
|
|||
return txsize_sqr_up_map[tx_size];
|
||||
}
|
||||
|
||||
#define BASE_CONTEXT_POSITION_NUM 12
|
||||
static int base_ref_offset[BASE_CONTEXT_POSITION_NUM][2] = {
|
||||
/* clang-format off*/
|
||||
{ -2, 0 }, { -1, -1 }, { -1, 0 }, { -1, 1 }, { 0, -2 }, { 0, -1 }, { 0, 1 },
|
||||
|
|
@ -36,23 +39,24 @@ static int base_ref_offset[BASE_CONTEXT_POSITION_NUM][2] = {
|
|||
/* clang-format on*/
|
||||
};
|
||||
|
||||
static INLINE int get_level_count(const tran_low_t *tcoeffs, int stride,
|
||||
static INLINE int get_level_count(const tran_low_t *tcoeffs, int bwl,
|
||||
int height, int row, int col, int level,
|
||||
int (*nb_offset)[2], int nb_num) {
|
||||
int count = 0;
|
||||
for (int idx = 0; idx < nb_num; ++idx) {
|
||||
const int ref_row = row + nb_offset[idx][0];
|
||||
const int ref_col = col + nb_offset[idx][1];
|
||||
const int pos = ref_row * stride + ref_col;
|
||||
if (ref_row < 0 || ref_col < 0 || ref_row >= height || ref_col >= stride)
|
||||
if (ref_row < 0 || ref_col < 0 || ref_row >= height ||
|
||||
ref_col >= (1 << bwl))
|
||||
continue;
|
||||
const int pos = (ref_row << bwl) + ref_col;
|
||||
tran_low_t abs_coeff = abs(tcoeffs[pos]);
|
||||
count += abs_coeff > level;
|
||||
}
|
||||
return count;
|
||||
}
|
||||
|
||||
static INLINE void get_mag(int *mag, const tran_low_t *tcoeffs, int stride,
|
||||
static INLINE void get_mag(int *mag, const tran_low_t *tcoeffs, int bwl,
|
||||
int height, int row, int col, int (*nb_offset)[2],
|
||||
int nb_num) {
|
||||
mag[0] = 0;
|
||||
|
|
@ -60,9 +64,10 @@ static INLINE void get_mag(int *mag, const tran_low_t *tcoeffs, int stride,
|
|||
for (int idx = 0; idx < nb_num; ++idx) {
|
||||
const int ref_row = row + nb_offset[idx][0];
|
||||
const int ref_col = col + nb_offset[idx][1];
|
||||
const int pos = ref_row * stride + ref_col;
|
||||
if (ref_row < 0 || ref_col < 0 || ref_row >= height || ref_col >= stride)
|
||||
if (ref_row < 0 || ref_col < 0 || ref_row >= height ||
|
||||
ref_col >= (1 << bwl))
|
||||
continue;
|
||||
const int pos = (ref_row << bwl) + ref_col;
|
||||
tran_low_t abs_coeff = abs(tcoeffs[pos]);
|
||||
if (nb_offset[idx][0] >= 0 && nb_offset[idx][1] >= 0) {
|
||||
if (abs_coeff > mag[0]) {
|
||||
|
|
@ -74,18 +79,50 @@ static INLINE void get_mag(int *mag, const tran_low_t *tcoeffs, int stride,
|
|||
}
|
||||
}
|
||||
}
|
||||
|
||||
static INLINE void get_base_count_mag(int *mag, int *count,
|
||||
const tran_low_t *tcoeffs, int bwl,
|
||||
int height, int row, int col) {
|
||||
mag[0] = 0;
|
||||
mag[1] = 0;
|
||||
for (int i = 0; i < NUM_BASE_LEVELS; ++i) count[i] = 0;
|
||||
for (int idx = 0; idx < BASE_CONTEXT_POSITION_NUM; ++idx) {
|
||||
const int ref_row = row + base_ref_offset[idx][0];
|
||||
const int ref_col = col + base_ref_offset[idx][1];
|
||||
if (ref_row < 0 || ref_col < 0 || ref_row >= height ||
|
||||
ref_col >= (1 << bwl))
|
||||
continue;
|
||||
const int pos = (ref_row << bwl) + ref_col;
|
||||
tran_low_t abs_coeff = abs(tcoeffs[pos]);
|
||||
// count
|
||||
for (int i = 0; i < NUM_BASE_LEVELS; ++i) {
|
||||
count[i] += abs_coeff > i;
|
||||
}
|
||||
// mag
|
||||
if (base_ref_offset[idx][0] >= 0 && base_ref_offset[idx][1] >= 0) {
|
||||
if (abs_coeff > mag[0]) {
|
||||
mag[0] = abs_coeff;
|
||||
mag[1] = 1;
|
||||
} else if (abs_coeff == mag[0]) {
|
||||
++mag[1];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static INLINE int get_level_count_mag(int *mag, const tran_low_t *tcoeffs,
|
||||
int stride, int height, int row, int col,
|
||||
int bwl, int height, int row, int col,
|
||||
int level, int (*nb_offset)[2],
|
||||
int nb_num) {
|
||||
const int stride = 1 << bwl;
|
||||
int count = 0;
|
||||
*mag = 0;
|
||||
for (int idx = 0; idx < nb_num; ++idx) {
|
||||
const int ref_row = row + nb_offset[idx][0];
|
||||
const int ref_col = col + nb_offset[idx][1];
|
||||
const int pos = ref_row * stride + ref_col;
|
||||
if (ref_row < 0 || ref_col < 0 || ref_row >= height || ref_col >= stride)
|
||||
continue;
|
||||
const int pos = (ref_row << bwl) + ref_col;
|
||||
tran_low_t abs_coeff = abs(tcoeffs[pos]);
|
||||
count += abs_coeff > level;
|
||||
if (nb_offset[idx][0] >= 0 && nb_offset[idx][1] >= 0)
|
||||
|
|
@ -95,19 +132,21 @@ static INLINE int get_level_count_mag(int *mag, const tran_low_t *tcoeffs,
|
|||
}
|
||||
|
||||
static INLINE int get_base_ctx_from_count_mag(int row, int col, int count,
|
||||
int mag, int level) {
|
||||
int sig_mag) {
|
||||
const int ctx = (count + 1) >> 1;
|
||||
const int sig_mag = mag > level;
|
||||
int ctx_idx = -1;
|
||||
if (row == 0 && col == 0) {
|
||||
ctx_idx = (ctx << 1) + sig_mag;
|
||||
assert(ctx_idx < 8);
|
||||
// TODO(angiebird): turn this on once the optimization is finalized
|
||||
// assert(ctx_idx < 8);
|
||||
} else if (row == 0) {
|
||||
ctx_idx = 8 + (ctx << 1) + sig_mag;
|
||||
assert(ctx_idx < 18);
|
||||
// TODO(angiebird): turn this on once the optimization is finalized
|
||||
// assert(ctx_idx < 18);
|
||||
} else if (col == 0) {
|
||||
ctx_idx = 8 + 10 + (ctx << 1) + sig_mag;
|
||||
assert(ctx_idx < 28);
|
||||
// TODO(angiebird): turn this on once the optimization is finalized
|
||||
// assert(ctx_idx < 28);
|
||||
} else {
|
||||
ctx_idx = 8 + 10 + 10 + (ctx << 1) + sig_mag;
|
||||
assert(ctx_idx < COEFF_BASE_CONTEXTS);
|
||||
|
|
@ -119,15 +158,14 @@ static INLINE int get_base_ctx(const tran_low_t *tcoeffs,
|
|||
int c, // raster order
|
||||
const int bwl, const int height,
|
||||
const int level) {
|
||||
const int stride = 1 << bwl;
|
||||
const int row = c >> bwl;
|
||||
const int col = c - (row << bwl);
|
||||
const int level_minus_1 = level - 1;
|
||||
int mag;
|
||||
int count = get_level_count_mag(&mag, tcoeffs, stride, height, row, col,
|
||||
level_minus_1, base_ref_offset,
|
||||
BASE_CONTEXT_POSITION_NUM);
|
||||
int ctx_idx = get_base_ctx_from_count_mag(row, col, count, mag, level);
|
||||
int count =
|
||||
get_level_count_mag(&mag, tcoeffs, bwl, height, row, col, level_minus_1,
|
||||
base_ref_offset, BASE_CONTEXT_POSITION_NUM);
|
||||
int ctx_idx = get_base_ctx_from_count_mag(row, col, count, mag > level);
|
||||
return ctx_idx;
|
||||
}
|
||||
|
||||
|
|
@ -139,13 +177,52 @@ static int br_ref_offset[BR_CONTEXT_POSITION_NUM][2] = {
|
|||
/* clang-format on*/
|
||||
};
|
||||
|
||||
static int br_level_map[9] = {
|
||||
static const int br_level_map[9] = {
|
||||
0, 0, 1, 1, 2, 2, 3, 3, 3,
|
||||
};
|
||||
|
||||
static const int coeff_to_br_index[COEFF_BASE_RANGE] = {
|
||||
0, 0, 1, 1, 1, 1, 2, 2, 2, 2, 2, 2, 2, 2,
|
||||
};
|
||||
|
||||
static const int br_index_to_coeff[BASE_RANGE_SETS] = {
|
||||
0, 2, 6,
|
||||
};
|
||||
|
||||
static const int br_extra_bits[BASE_RANGE_SETS] = {
|
||||
1, 2, 3,
|
||||
};
|
||||
|
||||
#define BR_MAG_OFFSET 1
|
||||
// TODO(angiebird): optimize this function by using a table to map from
|
||||
// count/mag to ctx
|
||||
|
||||
static INLINE int get_br_count_mag(int *mag, const tran_low_t *tcoeffs, int bwl,
|
||||
int height, int row, int col, int level) {
|
||||
mag[0] = 0;
|
||||
mag[1] = 0;
|
||||
int count = 0;
|
||||
for (int idx = 0; idx < BR_CONTEXT_POSITION_NUM; ++idx) {
|
||||
const int ref_row = row + br_ref_offset[idx][0];
|
||||
const int ref_col = col + br_ref_offset[idx][1];
|
||||
if (ref_row < 0 || ref_col < 0 || ref_row >= height ||
|
||||
ref_col >= (1 << bwl))
|
||||
continue;
|
||||
const int pos = (ref_row << bwl) + ref_col;
|
||||
tran_low_t abs_coeff = abs(tcoeffs[pos]);
|
||||
count += abs_coeff > level;
|
||||
if (br_ref_offset[idx][0] >= 0 && br_ref_offset[idx][1] >= 0) {
|
||||
if (abs_coeff > mag[0]) {
|
||||
mag[0] = abs_coeff;
|
||||
mag[1] = 1;
|
||||
} else if (abs_coeff == mag[0]) {
|
||||
++mag[1];
|
||||
}
|
||||
}
|
||||
}
|
||||
return count;
|
||||
}
|
||||
|
||||
static INLINE int get_br_ctx_from_count_mag(int row, int col, int count,
|
||||
int mag) {
|
||||
int offset = 0;
|
||||
|
|
@ -153,7 +230,7 @@ static INLINE int get_br_ctx_from_count_mag(int row, int col, int count,
|
|||
offset = 0;
|
||||
else if (mag <= 3)
|
||||
offset = 1;
|
||||
else if (mag <= 6)
|
||||
else if (mag <= 5)
|
||||
offset = 2;
|
||||
else
|
||||
offset = 3;
|
||||
|
|
@ -177,111 +254,171 @@ static INLINE int get_br_ctx_from_count_mag(int row, int col, int count,
|
|||
static INLINE int get_br_ctx(const tran_low_t *tcoeffs,
|
||||
const int c, // raster order
|
||||
const int bwl, const int height) {
|
||||
const int stride = 1 << bwl;
|
||||
const int row = c >> bwl;
|
||||
const int col = c - (row << bwl);
|
||||
const int level_minus_1 = NUM_BASE_LEVELS;
|
||||
int mag;
|
||||
const int count = get_level_count_mag(&mag, tcoeffs, stride, height, row, col,
|
||||
level_minus_1, br_ref_offset,
|
||||
BR_CONTEXT_POSITION_NUM);
|
||||
const int count =
|
||||
get_level_count_mag(&mag, tcoeffs, bwl, height, row, col, level_minus_1,
|
||||
br_ref_offset, BR_CONTEXT_POSITION_NUM);
|
||||
const int ctx = get_br_ctx_from_count_mag(row, col, count, mag);
|
||||
return ctx;
|
||||
}
|
||||
|
||||
#define SIG_REF_OFFSET_NUM 11
|
||||
#define SIG_REF_OFFSET_NUM 7
|
||||
static int sig_ref_offset[SIG_REF_OFFSET_NUM][2] = {
|
||||
{ -2, -1 }, { -2, 0 }, { -2, 1 }, { -1, -2 }, { -1, -1 }, { -1, 0 },
|
||||
{ -1, 1 }, { 0, -2 }, { 0, -1 }, { 1, -2 }, { 1, -1 },
|
||||
{ -2, -1 }, { -2, 0 }, { -1, -2 }, { -1, -1 },
|
||||
{ -1, 0 }, { 0, -2 }, { 0, -1 },
|
||||
};
|
||||
|
||||
static INLINE int get_nz_count(const tran_low_t *tcoeffs, int stride,
|
||||
int height, int row, int col,
|
||||
const int16_t *iscan) {
|
||||
#if REDUCE_CONTEXT_DEPENDENCY
|
||||
static INLINE int get_nz_count(const tran_low_t *tcoeffs, int bwl, int height,
|
||||
int row, int col, int prev_row, int prev_col) {
|
||||
int count = 0;
|
||||
const int pos = row * stride + col;
|
||||
for (int idx = 0; idx < SIG_REF_OFFSET_NUM; ++idx) {
|
||||
const int ref_row = row + sig_ref_offset[idx][0];
|
||||
const int ref_col = col + sig_ref_offset[idx][1];
|
||||
if (ref_row < 0 || ref_col < 0 || ref_row >= height || ref_col >= stride)
|
||||
if (ref_row < 0 || ref_col < 0 || ref_row >= height ||
|
||||
ref_col >= (1 << bwl) || (prev_row == ref_row && prev_col == ref_col))
|
||||
continue;
|
||||
const int nb_pos = ref_row * stride + ref_col;
|
||||
if (iscan[nb_pos] < iscan[pos]) count += (tcoeffs[nb_pos] != 0);
|
||||
const int nb_pos = (ref_row << bwl) + ref_col;
|
||||
count += (tcoeffs[nb_pos] != 0);
|
||||
}
|
||||
return count;
|
||||
}
|
||||
#else
|
||||
static INLINE int get_nz_count(const tran_low_t *tcoeffs, int bwl, int height,
|
||||
int row, int col) {
|
||||
int count = 0;
|
||||
for (int idx = 0; idx < SIG_REF_OFFSET_NUM; ++idx) {
|
||||
const int ref_row = row + sig_ref_offset[idx][0];
|
||||
const int ref_col = col + sig_ref_offset[idx][1];
|
||||
if (ref_row < 0 || ref_col < 0 || ref_row >= height ||
|
||||
ref_col >= (1 << bwl))
|
||||
continue;
|
||||
const int nb_pos = (ref_row << bwl) + ref_col;
|
||||
count += (tcoeffs[nb_pos] != 0);
|
||||
}
|
||||
return count;
|
||||
}
|
||||
#endif
|
||||
|
||||
static INLINE TX_CLASS get_tx_class(TX_TYPE tx_type) {
|
||||
switch (tx_type) {
|
||||
#if CONFIG_EXT_TX
|
||||
case V_DCT:
|
||||
case V_ADST:
|
||||
case V_FLIPADST: return TX_CLASS_VERT;
|
||||
case H_DCT:
|
||||
case H_ADST:
|
||||
case H_FLIPADST: return TX_CLASS_HORIZ;
|
||||
#endif
|
||||
default: return TX_CLASS_2D;
|
||||
}
|
||||
}
|
||||
|
||||
// TODO(angiebird): optimize this function by generate a table that maps from
|
||||
// count to ctx
|
||||
static INLINE int get_nz_map_ctx_from_count(int count,
|
||||
const tran_low_t *tcoeffs,
|
||||
int coeff_idx, // raster order
|
||||
int bwl, const int16_t *iscan) {
|
||||
int bwl, TX_TYPE tx_type) {
|
||||
(void)tx_type;
|
||||
const int row = coeff_idx >> bwl;
|
||||
const int col = coeff_idx - (row << bwl);
|
||||
int ctx = 0;
|
||||
#if CONFIG_EXT_TX
|
||||
int tx_class = get_tx_class(tx_type);
|
||||
int offset;
|
||||
if (tx_class == TX_CLASS_2D)
|
||||
offset = 0;
|
||||
else if (tx_class == TX_CLASS_VERT)
|
||||
offset = SIG_COEF_CONTEXTS_2D;
|
||||
else
|
||||
offset = SIG_COEF_CONTEXTS_2D + SIG_COEF_CONTEXTS_1D;
|
||||
#else
|
||||
int offset = 0;
|
||||
#endif
|
||||
|
||||
if (row == 0 && col == 0) return 0;
|
||||
if (row == 0 && col == 0) return offset + 0;
|
||||
|
||||
if (row == 0 && col == 1) return 1 + (tcoeffs[0] != 0);
|
||||
if (row == 0 && col == 1) return offset + 1 + count;
|
||||
|
||||
if (row == 1 && col == 0) return 3 + (tcoeffs[0] != 0);
|
||||
if (row == 1 && col == 0) return offset + 3 + count;
|
||||
|
||||
if (row == 1 && col == 1) {
|
||||
int pos;
|
||||
ctx = (tcoeffs[0] != 0);
|
||||
|
||||
if (iscan[1] < iscan[coeff_idx]) ctx += (tcoeffs[1] != 0);
|
||||
pos = 1 << bwl;
|
||||
if (iscan[pos] < iscan[coeff_idx]) ctx += (tcoeffs[pos] != 0);
|
||||
|
||||
ctx = (ctx + 1) >> 1;
|
||||
ctx = (count + 1) >> 1;
|
||||
|
||||
assert(5 + ctx <= 7);
|
||||
|
||||
return 5 + ctx;
|
||||
return offset + 5 + ctx;
|
||||
}
|
||||
|
||||
if (row == 0) {
|
||||
ctx = (count + 1) >> 1;
|
||||
|
||||
assert(ctx < 3);
|
||||
return 8 + ctx;
|
||||
assert(ctx < 2);
|
||||
return offset + 8 + ctx;
|
||||
}
|
||||
|
||||
if (col == 0) {
|
||||
ctx = (count + 1) >> 1;
|
||||
|
||||
assert(ctx < 3);
|
||||
return 11 + ctx;
|
||||
assert(ctx < 2);
|
||||
return offset + 10 + ctx;
|
||||
}
|
||||
|
||||
ctx = count >> 1;
|
||||
|
||||
assert(14 + ctx < 20);
|
||||
assert(12 + ctx < 16);
|
||||
|
||||
return 14 + ctx;
|
||||
return offset + 12 + ctx;
|
||||
}
|
||||
|
||||
static INLINE int get_nz_map_ctx(const tran_low_t *tcoeffs,
|
||||
const int coeff_idx, // raster order
|
||||
const int bwl, const int height,
|
||||
const int16_t *iscan) {
|
||||
int stride = 1 << bwl;
|
||||
static INLINE int get_nz_map_ctx(const tran_low_t *tcoeffs, const int scan_idx,
|
||||
const int16_t *scan, const int bwl,
|
||||
const int height, TX_TYPE tx_type) {
|
||||
const int coeff_idx = scan[scan_idx];
|
||||
const int row = coeff_idx >> bwl;
|
||||
const int col = coeff_idx - (row << bwl);
|
||||
int count = get_nz_count(tcoeffs, stride, height, row, col, iscan);
|
||||
return get_nz_map_ctx_from_count(count, tcoeffs, coeff_idx, bwl, iscan);
|
||||
#if REDUCE_CONTEXT_DEPENDENCY
|
||||
int prev_coeff_idx;
|
||||
int prev_row;
|
||||
int prev_col;
|
||||
if (scan_idx > MIN_SCAN_IDX_REDUCE_CONTEXT_DEPENDENCY) {
|
||||
prev_coeff_idx = scan[scan_idx - 1]; // raster order
|
||||
prev_row = prev_coeff_idx >> bwl;
|
||||
prev_col = prev_coeff_idx - (prev_row << bwl);
|
||||
} else {
|
||||
prev_coeff_idx = -1;
|
||||
prev_row = -1;
|
||||
prev_col = -1;
|
||||
}
|
||||
int count = get_nz_count(tcoeffs, bwl, height, row, col, prev_row, prev_col);
|
||||
#else
|
||||
int count = get_nz_count(tcoeffs, bwl, height, row, col);
|
||||
#endif
|
||||
return get_nz_map_ctx_from_count(count, coeff_idx, bwl, tx_type);
|
||||
}
|
||||
|
||||
static INLINE int get_eob_ctx(const tran_low_t *tcoeffs,
|
||||
const int coeff_idx, // raster order
|
||||
const TX_SIZE txs_ctx) {
|
||||
const TX_SIZE txs_ctx, TX_TYPE tx_type) {
|
||||
(void)tcoeffs;
|
||||
if (txs_ctx == TX_4X4) return av1_coeff_band_4x4[coeff_idx];
|
||||
if (txs_ctx == TX_8X8) return av1_coeff_band_8x8[coeff_idx];
|
||||
if (txs_ctx == TX_16X16) return av1_coeff_band_16x16[coeff_idx];
|
||||
if (txs_ctx == TX_32X32) return av1_coeff_band_32x32[coeff_idx];
|
||||
int offset = 0;
|
||||
#if CONFIG_CTX1D
|
||||
TX_CLASS tx_class = get_tx_class(tx_type);
|
||||
if (tx_class == TX_CLASS_VERT)
|
||||
offset = EOB_COEF_CONTEXTS_2D;
|
||||
else if (tx_class == TX_CLASS_HORIZ)
|
||||
offset = EOB_COEF_CONTEXTS_2D + EOB_COEF_CONTEXTS_1D;
|
||||
#else
|
||||
(void)tx_type;
|
||||
#endif
|
||||
|
||||
if (txs_ctx == TX_4X4) return offset + av1_coeff_band_4x4[coeff_idx];
|
||||
if (txs_ctx == TX_8X8) return offset + av1_coeff_band_8x8[coeff_idx];
|
||||
if (txs_ctx == TX_16X16) return offset + av1_coeff_band_16x16[coeff_idx];
|
||||
if (txs_ctx == TX_32X32) return offset + av1_coeff_band_32x32[coeff_idx];
|
||||
|
||||
assert(0);
|
||||
return 0;
|
||||
|
|
@ -369,6 +506,86 @@ static INLINE void get_txb_ctx(BLOCK_SIZE plane_bsize, TX_SIZE tx_size,
|
|||
}
|
||||
}
|
||||
|
||||
#if LV_MAP_PROB
|
||||
void av1_init_txb_probs(FRAME_CONTEXT *fc);
|
||||
#endif // LV_MAP_PROB
|
||||
|
||||
void av1_adapt_txb_probs(AV1_COMMON *cm, unsigned int count_sat,
|
||||
unsigned int update_factor);
|
||||
|
||||
void av1_init_lv_map(AV1_COMMON *cm);
|
||||
|
||||
#if CONFIG_CTX1D
|
||||
static INLINE void get_eob_vert(int16_t *eob_ls, const tran_low_t *tcoeff,
|
||||
int w, int h) {
|
||||
for (int c = 0; c < w; ++c) {
|
||||
eob_ls[c] = 0;
|
||||
for (int r = h - 1; r >= 0; --r) {
|
||||
int coeff_idx = r * w + c;
|
||||
if (tcoeff[coeff_idx] != 0) {
|
||||
eob_ls[c] = r + 1;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static INLINE void get_eob_horiz(int16_t *eob_ls, const tran_low_t *tcoeff,
|
||||
int w, int h) {
|
||||
for (int r = 0; r < h; ++r) {
|
||||
eob_ls[r] = 0;
|
||||
for (int c = w - 1; c >= 0; --c) {
|
||||
int coeff_idx = r * w + c;
|
||||
if (tcoeff[coeff_idx] != 0) {
|
||||
eob_ls[r] = c + 1;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static INLINE int get_empty_line_ctx(int line_idx, int16_t *eob_ls) {
|
||||
if (line_idx > 0) {
|
||||
int prev_eob = eob_ls[line_idx - 1];
|
||||
if (prev_eob == 0) {
|
||||
return 1;
|
||||
} else if (prev_eob < 3) {
|
||||
return 2;
|
||||
} else if (prev_eob < 6) {
|
||||
return 3;
|
||||
} else {
|
||||
return 4;
|
||||
}
|
||||
} else {
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
|
||||
#define MAX_POS_CTX 8
|
||||
static int pos_ctx[MAX_HVTX_SIZE] = {
|
||||
0, 1, 2, 2, 3, 3, 3, 3, 4, 4, 4, 4, 5, 5, 5, 5,
|
||||
6, 6, 6, 6, 6, 6, 6, 6, 7, 7, 7, 7, 7, 7, 7, 7,
|
||||
};
|
||||
static INLINE int get_hv_eob_ctx(int line_idx, int pos, int16_t *eob_ls) {
|
||||
if (line_idx > 0) {
|
||||
int prev_eob = eob_ls[line_idx - 1];
|
||||
int diff = pos + 1 - prev_eob;
|
||||
int abs_diff = abs(diff);
|
||||
int ctx_idx = pos_ctx[abs_diff];
|
||||
assert(ctx_idx < MAX_POS_CTX);
|
||||
if (diff < 0) {
|
||||
ctx_idx += MAX_POS_CTX;
|
||||
assert(ctx_idx >= MAX_POS_CTX);
|
||||
assert(ctx_idx < 2 * MAX_POS_CTX);
|
||||
}
|
||||
return ctx_idx;
|
||||
} else {
|
||||
int ctx_idx = MAX_POS_CTX + MAX_POS_CTX + pos_ctx[pos];
|
||||
assert(ctx_idx < HV_EOB_CONTEXTS);
|
||||
assert(HV_EOB_CONTEXTS == MAX_POS_CTX * 3);
|
||||
return ctx_idx;
|
||||
}
|
||||
}
|
||||
#endif // CONFIG_CTX1D
|
||||
|
||||
#endif // AV1_COMMON_TXB_COMMON_H_
|
||||
|
|
|
|||
460
third_party/aom/av1/common/warped_motion.c
vendored
460
third_party/aom/av1/common/warped_motion.c
vendored
|
|
@ -912,8 +912,8 @@ static void highbd_warp_plane_old(const WarpedMotionParams *const wm,
|
|||
in[0] = j;
|
||||
in[1] = i;
|
||||
projectpoints(wm->wmmat, in, out, 1, 2, 2, subsampling_x, subsampling_y);
|
||||
out[0] = ROUND_POWER_OF_TWO_SIGNED(out[0] * x_scale, 4);
|
||||
out[1] = ROUND_POWER_OF_TWO_SIGNED(out[1] * y_scale, 4);
|
||||
out[0] = ROUND_POWER_OF_TWO_SIGNED(out[0] * x_scale, SCALE_SUBPEL_BITS);
|
||||
out[1] = ROUND_POWER_OF_TWO_SIGNED(out[1] * y_scale, SCALE_SUBPEL_BITS);
|
||||
if (conv_params->do_average)
|
||||
pred[(j - p_col) + (i - p_row) * p_stride] = ROUND_POWER_OF_TWO(
|
||||
pred[(j - p_col) + (i - p_row) * p_stride] +
|
||||
|
|
@ -939,136 +939,51 @@ void av1_highbd_warp_affine_c(const int32_t *mat, const uint16_t *ref,
|
|||
int16_t beta, int16_t gamma, int16_t delta) {
|
||||
int32_t tmp[15 * 8];
|
||||
int i, j, k, l, m;
|
||||
|
||||
for (i = p_row; i < p_row + p_height; i += 8) {
|
||||
for (j = p_col; j < p_col + p_width; j += 8) {
|
||||
int32_t x4, y4, ix4, sx4, iy4, sy4;
|
||||
if (subsampling_x)
|
||||
x4 = (mat[2] * 4 * (j + 4) + mat[3] * 4 * (i + 4) + mat[0] * 2 +
|
||||
(mat[2] + mat[3] - (1 << WARPEDMODEL_PREC_BITS))) /
|
||||
4;
|
||||
else
|
||||
x4 = mat[2] * (j + 4) + mat[3] * (i + 4) + mat[0];
|
||||
|
||||
if (subsampling_y)
|
||||
y4 = (mat[4] * 4 * (j + 4) + mat[5] * 4 * (i + 4) + mat[1] * 2 +
|
||||
(mat[4] + mat[5] - (1 << WARPEDMODEL_PREC_BITS))) /
|
||||
4;
|
||||
else
|
||||
y4 = mat[4] * (j + 4) + mat[5] * (i + 4) + mat[1];
|
||||
|
||||
ix4 = x4 >> WARPEDMODEL_PREC_BITS;
|
||||
sx4 = x4 & ((1 << WARPEDMODEL_PREC_BITS) - 1);
|
||||
iy4 = y4 >> WARPEDMODEL_PREC_BITS;
|
||||
sy4 = y4 & ((1 << WARPEDMODEL_PREC_BITS) - 1);
|
||||
|
||||
sx4 += alpha * (-4) + beta * (-4);
|
||||
sy4 += gamma * (-4) + delta * (-4);
|
||||
|
||||
sx4 &= ~((1 << WARP_PARAM_REDUCE_BITS) - 1);
|
||||
sy4 &= ~((1 << WARP_PARAM_REDUCE_BITS) - 1);
|
||||
|
||||
// Horizontal filter
|
||||
for (k = -7; k < 8; ++k) {
|
||||
int iy = iy4 + k;
|
||||
if (iy < 0)
|
||||
iy = 0;
|
||||
else if (iy > height - 1)
|
||||
iy = height - 1;
|
||||
|
||||
int sx = sx4 + beta * (k + 4);
|
||||
for (l = -4; l < 4; ++l) {
|
||||
int ix = ix4 + l - 3;
|
||||
const int offs = ROUND_POWER_OF_TWO(sx, WARPEDDIFF_PREC_BITS) +
|
||||
WARPEDPIXEL_PREC_SHIFTS;
|
||||
assert(offs >= 0 && offs <= WARPEDPIXEL_PREC_SHIFTS * 3);
|
||||
const int16_t *coeffs = warped_filter[offs];
|
||||
|
||||
int32_t sum = 1 << (bd + WARPEDPIXEL_FILTER_BITS - 1);
|
||||
for (m = 0; m < 8; ++m) {
|
||||
int sample_x = ix + m;
|
||||
if (sample_x < 0)
|
||||
sample_x = 0;
|
||||
else if (sample_x > width - 1)
|
||||
sample_x = width - 1;
|
||||
sum += ref[iy * stride + sample_x] * coeffs[m];
|
||||
}
|
||||
sum = ROUND_POWER_OF_TWO(sum, HORSHEAR_REDUCE_PREC_BITS);
|
||||
assert(0 <= sum &&
|
||||
sum < (1 << (bd + WARPEDPIXEL_FILTER_BITS + 1 -
|
||||
HORSHEAR_REDUCE_PREC_BITS)));
|
||||
tmp[(k + 7) * 8 + (l + 4)] = sum;
|
||||
sx += alpha;
|
||||
}
|
||||
}
|
||||
|
||||
// Vertical filter
|
||||
for (k = -4; k < AOMMIN(4, p_row + p_height - i - 4); ++k) {
|
||||
int sy = sy4 + delta * (k + 4);
|
||||
for (l = -4; l < 4; ++l) {
|
||||
uint16_t *p =
|
||||
&pred[(i - p_row + k + 4) * p_stride + (j - p_col + l + 4)];
|
||||
const int offs = ROUND_POWER_OF_TWO(sy, WARPEDDIFF_PREC_BITS) +
|
||||
WARPEDPIXEL_PREC_SHIFTS;
|
||||
assert(offs >= 0 && offs <= WARPEDPIXEL_PREC_SHIFTS * 3);
|
||||
const int16_t *coeffs = warped_filter[offs];
|
||||
|
||||
int32_t sum = 1 << (bd + 2 * WARPEDPIXEL_FILTER_BITS -
|
||||
HORSHEAR_REDUCE_PREC_BITS);
|
||||
for (m = 0; m < 8; ++m) {
|
||||
sum += tmp[(k + m + 4) * 8 + (l + 4)] * coeffs[m];
|
||||
}
|
||||
sum = ROUND_POWER_OF_TWO(sum, VERSHEAR_REDUCE_PREC_BITS);
|
||||
assert(0 <= sum && sum < (1 << (bd + 2)));
|
||||
uint16_t px =
|
||||
clip_pixel_highbd(sum - (1 << (bd - 1)) - (1 << bd), bd);
|
||||
if (conv_params->do_average)
|
||||
*p = ROUND_POWER_OF_TWO(*p + px, 1);
|
||||
else
|
||||
*p = px;
|
||||
sy += gamma;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#if CONFIG_CONVOLVE_ROUND
|
||||
void av1_highbd_warp_affine_post_round_c(
|
||||
const int32_t *mat, const uint16_t *ref, int width, int height, int stride,
|
||||
uint16_t *pred, int p_col, int p_row, int p_width, int p_height,
|
||||
int p_stride, int subsampling_x, int subsampling_y, int bd,
|
||||
ConvolveParams *conv_params, int16_t alpha, int16_t beta, int16_t gamma,
|
||||
int16_t delta) {
|
||||
(void)pred;
|
||||
(void)p_stride;
|
||||
int32_t tmp[15 * 8];
|
||||
int i, j, k, l, m;
|
||||
const int offset_bits_horiz = bd + FILTER_BITS - 1;
|
||||
const int offset_bits_vert = bd + 2 * FILTER_BITS - conv_params->round_0;
|
||||
const int use_conv_params = conv_params->round == CONVOLVE_OPT_NO_ROUND;
|
||||
const int reduce_bits_horiz =
|
||||
use_conv_params ? conv_params->round_0 : HORSHEAR_REDUCE_PREC_BITS;
|
||||
const int max_bits_horiz =
|
||||
use_conv_params
|
||||
? bd + FILTER_BITS + 1 - conv_params->round_0
|
||||
: bd + WARPEDPIXEL_FILTER_BITS + 1 - HORSHEAR_REDUCE_PREC_BITS;
|
||||
const int offset_bits_horiz =
|
||||
use_conv_params ? bd + FILTER_BITS - 1 : bd + WARPEDPIXEL_FILTER_BITS - 1;
|
||||
const int offset_bits_vert =
|
||||
use_conv_params
|
||||
? bd + 2 * FILTER_BITS - conv_params->round_0
|
||||
: bd + 2 * WARPEDPIXEL_FILTER_BITS - HORSHEAR_REDUCE_PREC_BITS;
|
||||
if (use_conv_params) {
|
||||
conv_params->do_post_rounding = 1;
|
||||
}
|
||||
assert(FILTER_BITS == WARPEDPIXEL_FILTER_BITS);
|
||||
#else
|
||||
const int reduce_bits_horiz = HORSHEAR_REDUCE_PREC_BITS;
|
||||
const int max_bits_horiz =
|
||||
bd + WARPEDPIXEL_FILTER_BITS + 1 - HORSHEAR_REDUCE_PREC_BITS;
|
||||
const int offset_bits_horiz = bd + WARPEDPIXEL_FILTER_BITS - 1;
|
||||
const int offset_bits_vert =
|
||||
bd + 2 * WARPEDPIXEL_FILTER_BITS - HORSHEAR_REDUCE_PREC_BITS;
|
||||
#endif
|
||||
(void)max_bits_horiz;
|
||||
|
||||
for (i = p_row; i < p_row + p_height; i += 8) {
|
||||
for (j = p_col; j < p_col + p_width; j += 8) {
|
||||
int32_t x4, y4, ix4, sx4, iy4, sy4;
|
||||
if (subsampling_x)
|
||||
x4 = (mat[2] * 4 * (j + 4) + mat[3] * 4 * (i + 4) + mat[0] * 2 +
|
||||
(mat[2] + mat[3] - (1 << WARPEDMODEL_PREC_BITS))) /
|
||||
4;
|
||||
else
|
||||
x4 = mat[2] * (j + 4) + mat[3] * (i + 4) + mat[0];
|
||||
// Calculate the center of this 8x8 block,
|
||||
// project to luma coordinates (if in a subsampled chroma plane),
|
||||
// apply the affine transformation,
|
||||
// then convert back to the original coordinates (if necessary)
|
||||
const int32_t src_x = (j + 4) << subsampling_x;
|
||||
const int32_t src_y = (i + 4) << subsampling_y;
|
||||
const int32_t dst_x = mat[2] * src_x + mat[3] * src_y + mat[0];
|
||||
const int32_t dst_y = mat[4] * src_x + mat[5] * src_y + mat[1];
|
||||
const int32_t x4 = dst_x >> subsampling_x;
|
||||
const int32_t y4 = dst_y >> subsampling_y;
|
||||
|
||||
if (subsampling_y)
|
||||
y4 = (mat[4] * 4 * (j + 4) + mat[5] * 4 * (i + 4) + mat[1] * 2 +
|
||||
(mat[4] + mat[5] - (1 << WARPEDMODEL_PREC_BITS))) /
|
||||
4;
|
||||
else
|
||||
y4 = mat[4] * (j + 4) + mat[5] * (i + 4) + mat[1];
|
||||
|
||||
ix4 = x4 >> WARPEDMODEL_PREC_BITS;
|
||||
sx4 = x4 & ((1 << WARPEDMODEL_PREC_BITS) - 1);
|
||||
iy4 = y4 >> WARPEDMODEL_PREC_BITS;
|
||||
sy4 = y4 & ((1 << WARPEDMODEL_PREC_BITS) - 1);
|
||||
int32_t ix4 = x4 >> WARPEDMODEL_PREC_BITS;
|
||||
int32_t sx4 = x4 & ((1 << WARPEDMODEL_PREC_BITS) - 1);
|
||||
int32_t iy4 = y4 >> WARPEDMODEL_PREC_BITS;
|
||||
int32_t sy4 = y4 & ((1 << WARPEDMODEL_PREC_BITS) - 1);
|
||||
|
||||
sx4 += alpha * (-4) + beta * (-4);
|
||||
sy4 += gamma * (-4) + delta * (-4);
|
||||
|
|
@ -1101,9 +1016,8 @@ void av1_highbd_warp_affine_post_round_c(
|
|||
sample_x = width - 1;
|
||||
sum += ref[iy * stride + sample_x] * coeffs[m];
|
||||
}
|
||||
sum = ROUND_POWER_OF_TWO(sum, conv_params->round_0);
|
||||
assert(0 <= sum &&
|
||||
sum < (1 << (bd + FILTER_BITS + 1 - conv_params->round_0)));
|
||||
sum = ROUND_POWER_OF_TWO(sum, reduce_bits_horiz);
|
||||
assert(0 <= sum && sum < (1 << max_bits_horiz));
|
||||
tmp[(k + 7) * 8 + (l + 4)] = sum;
|
||||
sx += alpha;
|
||||
}
|
||||
|
|
@ -1112,7 +1026,7 @@ void av1_highbd_warp_affine_post_round_c(
|
|||
// Vertical filter
|
||||
for (k = -4; k < AOMMIN(4, p_row + p_height - i - 4); ++k) {
|
||||
int sy = sy4 + delta * (k + 4);
|
||||
for (l = -4; l < 4; ++l) {
|
||||
for (l = -4; l < AOMMIN(4, p_col + p_width - j - 4); ++l) {
|
||||
const int offs = ROUND_POWER_OF_TWO(sy, WARPEDDIFF_PREC_BITS) +
|
||||
WARPEDPIXEL_PREC_SHIFTS;
|
||||
assert(offs >= 0 && offs <= WARPEDPIXEL_PREC_SHIFTS * 3);
|
||||
|
|
@ -1122,22 +1036,41 @@ void av1_highbd_warp_affine_post_round_c(
|
|||
for (m = 0; m < 8; ++m) {
|
||||
sum += tmp[(k + m + 4) * 8 + (l + 4)] * coeffs[m];
|
||||
}
|
||||
|
||||
sum = ROUND_POWER_OF_TWO(sum, conv_params->round_1) -
|
||||
(1 << (offset_bits_horiz + FILTER_BITS - conv_params->round_0 -
|
||||
conv_params->round_1)) -
|
||||
(1 << (offset_bits_vert - conv_params->round_1));
|
||||
CONV_BUF_TYPE *p =
|
||||
&conv_params->dst[(i - p_row + k + 4) * conv_params->dst_stride +
|
||||
(j - p_col + l + 4)];
|
||||
*p += sum;
|
||||
#if CONFIG_CONVOLVE_ROUND
|
||||
if (use_conv_params) {
|
||||
CONV_BUF_TYPE *p =
|
||||
&conv_params
|
||||
->dst[(i - p_row + k + 4) * conv_params->dst_stride +
|
||||
(j - p_col + l + 4)];
|
||||
sum = ROUND_POWER_OF_TWO(sum, conv_params->round_1) -
|
||||
(1 << (offset_bits_horiz + FILTER_BITS -
|
||||
conv_params->round_0 - conv_params->round_1)) -
|
||||
(1 << (offset_bits_vert - conv_params->round_1));
|
||||
if (conv_params->do_average)
|
||||
*p += sum;
|
||||
else
|
||||
*p = sum;
|
||||
} else {
|
||||
#else
|
||||
{
|
||||
#endif
|
||||
uint16_t *p =
|
||||
&pred[(i - p_row + k + 4) * p_stride + (j - p_col + l + 4)];
|
||||
sum = ROUND_POWER_OF_TWO(sum, VERSHEAR_REDUCE_PREC_BITS);
|
||||
assert(0 <= sum && sum < (1 << (bd + 2)));
|
||||
uint16_t px =
|
||||
clip_pixel_highbd(sum - (1 << (bd - 1)) - (1 << bd), bd);
|
||||
if (conv_params->do_average)
|
||||
*p = ROUND_POWER_OF_TWO(*p + px, 1);
|
||||
else
|
||||
*p = px;
|
||||
}
|
||||
sy += gamma;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
static void highbd_warp_plane(WarpedMotionParams *wm, const uint8_t *const ref8,
|
||||
int width, int height, int stride,
|
||||
|
|
@ -1160,25 +1093,10 @@ static void highbd_warp_plane(WarpedMotionParams *wm, const uint8_t *const ref8,
|
|||
|
||||
const uint16_t *const ref = CONVERT_TO_SHORTPTR(ref8);
|
||||
uint16_t *pred = CONVERT_TO_SHORTPTR(pred8);
|
||||
#if CONFIG_CONVOLVE_ROUND
|
||||
if (conv_params->round == CONVOLVE_OPT_NO_ROUND) {
|
||||
conv_params->do_post_rounding = 1;
|
||||
av1_highbd_warp_affine_post_round(
|
||||
mat, ref, width, height, stride, pred, p_col, p_row, p_width,
|
||||
p_height, p_stride, subsampling_x, subsampling_y, bd, conv_params,
|
||||
alpha, beta, gamma, delta);
|
||||
} else {
|
||||
av1_highbd_warp_affine(mat, ref, width, height, stride, pred, p_col,
|
||||
p_row, p_width, p_height, p_stride, subsampling_x,
|
||||
subsampling_y, bd, conv_params, alpha, beta, gamma,
|
||||
delta);
|
||||
}
|
||||
#else
|
||||
av1_highbd_warp_affine(mat, ref, width, height, stride, pred, p_col, p_row,
|
||||
p_width, p_height, p_stride, subsampling_x,
|
||||
subsampling_y, bd, conv_params, alpha, beta, gamma,
|
||||
delta);
|
||||
#endif
|
||||
} else {
|
||||
highbd_warp_plane_old(wm, ref8, width, height, stride, pred8, p_col, p_row,
|
||||
p_width, p_height, p_stride, subsampling_x,
|
||||
|
|
@ -1251,8 +1169,8 @@ static void warp_plane_old(const WarpedMotionParams *const wm,
|
|||
in[0] = j;
|
||||
in[1] = i;
|
||||
projectpoints(wm->wmmat, in, out, 1, 2, 2, subsampling_x, subsampling_y);
|
||||
out[0] = ROUND_POWER_OF_TWO_SIGNED(out[0] * x_scale, 4);
|
||||
out[1] = ROUND_POWER_OF_TWO_SIGNED(out[1] * y_scale, 4);
|
||||
out[0] = ROUND_POWER_OF_TWO_SIGNED(out[0] * x_scale, SCALE_SUBPEL_BITS);
|
||||
out[1] = ROUND_POWER_OF_TWO_SIGNED(out[1] * y_scale, SCALE_SUBPEL_BITS);
|
||||
if (conv_params->do_average)
|
||||
pred[(j - p_col) + (i - p_row) * p_stride] = ROUND_POWER_OF_TWO(
|
||||
pred[(j - p_col) + (i - p_row) * p_stride] +
|
||||
|
|
@ -1359,143 +1277,51 @@ void av1_warp_affine_c(const int32_t *mat, const uint8_t *ref, int width,
|
|||
int32_t tmp[15 * 8];
|
||||
int i, j, k, l, m;
|
||||
const int bd = 8;
|
||||
|
||||
for (i = p_row; i < p_row + p_height; i += 8) {
|
||||
for (j = p_col; j < p_col + p_width; j += 8) {
|
||||
int32_t x4, y4, ix4, sx4, iy4, sy4;
|
||||
if (subsampling_x)
|
||||
x4 = (mat[2] * 4 * (j + 4) + mat[3] * 4 * (i + 4) + mat[0] * 2 +
|
||||
(mat[2] + mat[3] - (1 << WARPEDMODEL_PREC_BITS))) /
|
||||
4;
|
||||
else
|
||||
x4 = mat[2] * (j + 4) + mat[3] * (i + 4) + mat[0];
|
||||
|
||||
if (subsampling_y)
|
||||
y4 = (mat[4] * 4 * (j + 4) + mat[5] * 4 * (i + 4) + mat[1] * 2 +
|
||||
(mat[4] + mat[5] - (1 << WARPEDMODEL_PREC_BITS))) /
|
||||
4;
|
||||
else
|
||||
y4 = mat[4] * (j + 4) + mat[5] * (i + 4) + mat[1];
|
||||
|
||||
ix4 = x4 >> WARPEDMODEL_PREC_BITS;
|
||||
sx4 = x4 & ((1 << WARPEDMODEL_PREC_BITS) - 1);
|
||||
iy4 = y4 >> WARPEDMODEL_PREC_BITS;
|
||||
sy4 = y4 & ((1 << WARPEDMODEL_PREC_BITS) - 1);
|
||||
|
||||
sx4 += alpha * (-4) + beta * (-4);
|
||||
sy4 += gamma * (-4) + delta * (-4);
|
||||
|
||||
sx4 &= ~((1 << WARP_PARAM_REDUCE_BITS) - 1);
|
||||
sy4 &= ~((1 << WARP_PARAM_REDUCE_BITS) - 1);
|
||||
|
||||
// Horizontal filter
|
||||
for (k = -7; k < 8; ++k) {
|
||||
// Clamp to top/bottom edge of the frame
|
||||
int iy = iy4 + k;
|
||||
if (iy < 0)
|
||||
iy = 0;
|
||||
else if (iy > height - 1)
|
||||
iy = height - 1;
|
||||
|
||||
int sx = sx4 + beta * (k + 4);
|
||||
|
||||
for (l = -4; l < 4; ++l) {
|
||||
int ix = ix4 + l - 3;
|
||||
// At this point, sx = sx4 + alpha * l + beta * k
|
||||
const int offs = ROUND_POWER_OF_TWO(sx, WARPEDDIFF_PREC_BITS) +
|
||||
WARPEDPIXEL_PREC_SHIFTS;
|
||||
assert(offs >= 0 && offs <= WARPEDPIXEL_PREC_SHIFTS * 3);
|
||||
const int16_t *coeffs = warped_filter[offs];
|
||||
|
||||
int32_t sum = 1 << (bd + WARPEDPIXEL_FILTER_BITS - 1);
|
||||
for (m = 0; m < 8; ++m) {
|
||||
// Clamp to left/right edge of the frame
|
||||
int sample_x = ix + m;
|
||||
if (sample_x < 0)
|
||||
sample_x = 0;
|
||||
else if (sample_x > width - 1)
|
||||
sample_x = width - 1;
|
||||
|
||||
sum += ref[iy * stride + sample_x] * coeffs[m];
|
||||
}
|
||||
sum = ROUND_POWER_OF_TWO(sum, HORSHEAR_REDUCE_PREC_BITS);
|
||||
assert(0 <= sum &&
|
||||
sum < (1 << (bd + WARPEDPIXEL_FILTER_BITS + 1 -
|
||||
HORSHEAR_REDUCE_PREC_BITS)));
|
||||
tmp[(k + 7) * 8 + (l + 4)] = sum;
|
||||
sx += alpha;
|
||||
}
|
||||
}
|
||||
|
||||
// Vertical filter
|
||||
for (k = -4; k < AOMMIN(4, p_row + p_height - i - 4); ++k) {
|
||||
int sy = sy4 + delta * (k + 4);
|
||||
for (l = -4; l < AOMMIN(4, p_col + p_width - j - 4); ++l) {
|
||||
uint8_t *p =
|
||||
&pred[(i - p_row + k + 4) * p_stride + (j - p_col + l + 4)];
|
||||
// At this point, sy = sy4 + gamma * l + delta * k
|
||||
const int offs = ROUND_POWER_OF_TWO(sy, WARPEDDIFF_PREC_BITS) +
|
||||
WARPEDPIXEL_PREC_SHIFTS;
|
||||
assert(offs >= 0 && offs <= WARPEDPIXEL_PREC_SHIFTS * 3);
|
||||
const int16_t *coeffs = warped_filter[offs];
|
||||
|
||||
int32_t sum = 1 << (bd + 2 * WARPEDPIXEL_FILTER_BITS -
|
||||
HORSHEAR_REDUCE_PREC_BITS);
|
||||
for (m = 0; m < 8; ++m) {
|
||||
sum += tmp[(k + m + 4) * 8 + (l + 4)] * coeffs[m];
|
||||
}
|
||||
sum = ROUND_POWER_OF_TWO(sum, VERSHEAR_REDUCE_PREC_BITS);
|
||||
assert(0 <= sum && sum < (1 << (bd + 2)));
|
||||
uint8_t px = clip_pixel(sum - (1 << (bd - 1)) - (1 << bd));
|
||||
if (conv_params->do_average)
|
||||
*p = ROUND_POWER_OF_TWO(*p + px, 1);
|
||||
else
|
||||
*p = px;
|
||||
sy += gamma;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#if CONFIG_CONVOLVE_ROUND
|
||||
void av1_warp_affine_post_round_c(const int32_t *mat, const uint8_t *ref,
|
||||
int width, int height, int stride,
|
||||
uint8_t *pred, int p_col, int p_row,
|
||||
int p_width, int p_height, int p_stride,
|
||||
int subsampling_x, int subsampling_y,
|
||||
ConvolveParams *conv_params, int16_t alpha,
|
||||
int16_t beta, int16_t gamma, int16_t delta) {
|
||||
(void)pred;
|
||||
(void)p_stride;
|
||||
int32_t tmp[15 * 8];
|
||||
int i, j, k, l, m;
|
||||
const int bd = 8;
|
||||
const int offset_bits_horiz = bd + FILTER_BITS - 1;
|
||||
const int offset_bits_vert = bd + 2 * FILTER_BITS - conv_params->round_0;
|
||||
const int use_conv_params = conv_params->round == CONVOLVE_OPT_NO_ROUND;
|
||||
const int reduce_bits_horiz =
|
||||
use_conv_params ? conv_params->round_0 : HORSHEAR_REDUCE_PREC_BITS;
|
||||
const int max_bits_horiz =
|
||||
use_conv_params
|
||||
? bd + FILTER_BITS + 1 - conv_params->round_0
|
||||
: bd + WARPEDPIXEL_FILTER_BITS + 1 - HORSHEAR_REDUCE_PREC_BITS;
|
||||
const int offset_bits_horiz =
|
||||
use_conv_params ? bd + FILTER_BITS - 1 : bd + WARPEDPIXEL_FILTER_BITS - 1;
|
||||
const int offset_bits_vert =
|
||||
use_conv_params
|
||||
? bd + 2 * FILTER_BITS - conv_params->round_0
|
||||
: bd + 2 * WARPEDPIXEL_FILTER_BITS - HORSHEAR_REDUCE_PREC_BITS;
|
||||
if (use_conv_params) {
|
||||
conv_params->do_post_rounding = 1;
|
||||
}
|
||||
assert(FILTER_BITS == WARPEDPIXEL_FILTER_BITS);
|
||||
#else
|
||||
const int reduce_bits_horiz = HORSHEAR_REDUCE_PREC_BITS;
|
||||
const int max_bits_horiz =
|
||||
bd + WARPEDPIXEL_FILTER_BITS + 1 - HORSHEAR_REDUCE_PREC_BITS;
|
||||
const int offset_bits_horiz = bd + WARPEDPIXEL_FILTER_BITS - 1;
|
||||
const int offset_bits_vert =
|
||||
bd + 2 * WARPEDPIXEL_FILTER_BITS - HORSHEAR_REDUCE_PREC_BITS;
|
||||
#endif
|
||||
(void)max_bits_horiz;
|
||||
|
||||
for (i = p_row; i < p_row + p_height; i += 8) {
|
||||
for (j = p_col; j < p_col + p_width; j += 8) {
|
||||
int32_t x4, y4, ix4, sx4, iy4, sy4;
|
||||
if (subsampling_x)
|
||||
x4 = (mat[2] * 4 * (j + 4) + mat[3] * 4 * (i + 4) + mat[0] * 2 +
|
||||
(mat[2] + mat[3] - (1 << WARPEDMODEL_PREC_BITS))) /
|
||||
4;
|
||||
else
|
||||
x4 = mat[2] * (j + 4) + mat[3] * (i + 4) + mat[0];
|
||||
// Calculate the center of this 8x8 block,
|
||||
// project to luma coordinates (if in a subsampled chroma plane),
|
||||
// apply the affine transformation,
|
||||
// then convert back to the original coordinates (if necessary)
|
||||
const int32_t src_x = (j + 4) << subsampling_x;
|
||||
const int32_t src_y = (i + 4) << subsampling_y;
|
||||
const int32_t dst_x = mat[2] * src_x + mat[3] * src_y + mat[0];
|
||||
const int32_t dst_y = mat[4] * src_x + mat[5] * src_y + mat[1];
|
||||
const int32_t x4 = dst_x >> subsampling_x;
|
||||
const int32_t y4 = dst_y >> subsampling_y;
|
||||
|
||||
if (subsampling_y)
|
||||
y4 = (mat[4] * 4 * (j + 4) + mat[5] * 4 * (i + 4) + mat[1] * 2 +
|
||||
(mat[4] + mat[5] - (1 << WARPEDMODEL_PREC_BITS))) /
|
||||
4;
|
||||
else
|
||||
y4 = mat[4] * (j + 4) + mat[5] * (i + 4) + mat[1];
|
||||
|
||||
ix4 = x4 >> WARPEDMODEL_PREC_BITS;
|
||||
sx4 = x4 & ((1 << WARPEDMODEL_PREC_BITS) - 1);
|
||||
iy4 = y4 >> WARPEDMODEL_PREC_BITS;
|
||||
sy4 = y4 & ((1 << WARPEDMODEL_PREC_BITS) - 1);
|
||||
int32_t ix4 = x4 >> WARPEDMODEL_PREC_BITS;
|
||||
int32_t sx4 = x4 & ((1 << WARPEDMODEL_PREC_BITS) - 1);
|
||||
int32_t iy4 = y4 >> WARPEDMODEL_PREC_BITS;
|
||||
int32_t sy4 = y4 & ((1 << WARPEDMODEL_PREC_BITS) - 1);
|
||||
|
||||
sx4 += alpha * (-4) + beta * (-4);
|
||||
sy4 += gamma * (-4) + delta * (-4);
|
||||
|
|
@ -1533,9 +1359,8 @@ void av1_warp_affine_post_round_c(const int32_t *mat, const uint8_t *ref,
|
|||
|
||||
sum += ref[iy * stride + sample_x] * coeffs[m];
|
||||
}
|
||||
sum = ROUND_POWER_OF_TWO(sum, conv_params->round_0);
|
||||
assert(0 <= sum &&
|
||||
sum < (1 << (bd + FILTER_BITS + 1 - conv_params->round_0)));
|
||||
sum = ROUND_POWER_OF_TWO(sum, reduce_bits_horiz);
|
||||
assert(0 <= sum && sum < (1 << max_bits_horiz));
|
||||
tmp[(k + 7) * 8 + (l + 4)] = sum;
|
||||
sx += alpha;
|
||||
}
|
||||
|
|
@ -1552,26 +1377,43 @@ void av1_warp_affine_post_round_c(const int32_t *mat, const uint8_t *ref,
|
|||
const int16_t *coeffs = warped_filter[offs];
|
||||
|
||||
int32_t sum = 1 << offset_bits_vert;
|
||||
|
||||
for (m = 0; m < 8; ++m) {
|
||||
sum += tmp[(k + m + 4) * 8 + (l + 4)] * coeffs[m];
|
||||
}
|
||||
|
||||
sum = ROUND_POWER_OF_TWO(sum, conv_params->round_1) -
|
||||
(1 << (offset_bits_horiz + FILTER_BITS - conv_params->round_0 -
|
||||
conv_params->round_1)) -
|
||||
(1 << (offset_bits_vert - conv_params->round_1));
|
||||
CONV_BUF_TYPE *p =
|
||||
&conv_params->dst[(i - p_row + k + 4) * conv_params->dst_stride +
|
||||
(j - p_col + l + 4)];
|
||||
*p += sum;
|
||||
#if CONFIG_CONVOLVE_ROUND
|
||||
if (use_conv_params) {
|
||||
CONV_BUF_TYPE *p =
|
||||
&conv_params
|
||||
->dst[(i - p_row + k + 4) * conv_params->dst_stride +
|
||||
(j - p_col + l + 4)];
|
||||
sum = ROUND_POWER_OF_TWO(sum, conv_params->round_1) -
|
||||
(1 << (offset_bits_horiz + FILTER_BITS -
|
||||
conv_params->round_0 - conv_params->round_1)) -
|
||||
(1 << (offset_bits_vert - conv_params->round_1));
|
||||
if (conv_params->do_average)
|
||||
*p += sum;
|
||||
else
|
||||
*p = sum;
|
||||
} else {
|
||||
#else
|
||||
{
|
||||
#endif
|
||||
uint8_t *p =
|
||||
&pred[(i - p_row + k + 4) * p_stride + (j - p_col + l + 4)];
|
||||
sum = ROUND_POWER_OF_TWO(sum, VERSHEAR_REDUCE_PREC_BITS);
|
||||
assert(0 <= sum && sum < (1 << (bd + 2)));
|
||||
uint8_t px = clip_pixel(sum - (1 << (bd - 1)) - (1 << bd));
|
||||
if (conv_params->do_average)
|
||||
*p = ROUND_POWER_OF_TWO(*p + px, 1);
|
||||
else
|
||||
*p = px;
|
||||
}
|
||||
sy += gamma;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
#endif // CONFIG_CONVOLVE_ROUND
|
||||
|
||||
static void warp_plane(WarpedMotionParams *wm, const uint8_t *const ref,
|
||||
int width, int height, int stride, uint8_t *pred,
|
||||
|
|
@ -1590,23 +1432,9 @@ static void warp_plane(WarpedMotionParams *wm, const uint8_t *const ref,
|
|||
const int16_t gamma = wm->gamma;
|
||||
const int16_t delta = wm->delta;
|
||||
|
||||
#if CONFIG_CONVOLVE_ROUND
|
||||
if (conv_params->round == CONVOLVE_OPT_NO_ROUND) {
|
||||
conv_params->do_post_rounding = 1;
|
||||
av1_warp_affine_post_round(mat, ref, width, height, stride, pred, p_col,
|
||||
p_row, p_width, p_height, p_stride,
|
||||
subsampling_x, subsampling_y, conv_params,
|
||||
alpha, beta, gamma, delta);
|
||||
} else {
|
||||
av1_warp_affine(mat, ref, width, height, stride, pred, p_col, p_row,
|
||||
p_width, p_height, p_stride, subsampling_x, subsampling_y,
|
||||
conv_params, alpha, beta, gamma, delta);
|
||||
}
|
||||
#else
|
||||
av1_warp_affine(mat, ref, width, height, stride, pred, p_col, p_row,
|
||||
p_width, p_height, p_stride, subsampling_x, subsampling_y,
|
||||
conv_params, alpha, beta, gamma, delta);
|
||||
#endif
|
||||
} else {
|
||||
warp_plane_old(wm, ref, width, height, stride, pred, p_col, p_row, p_width,
|
||||
p_height, p_stride, subsampling_x, subsampling_y, x_scale,
|
||||
|
|
|
|||
5
third_party/aom/av1/common/warped_motion.h
vendored
5
third_party/aom/av1/common/warped_motion.h
vendored
|
|
@ -30,10 +30,9 @@
|
|||
#define LEAST_SQUARES_SAMPLES_MAX (1 << LEAST_SQUARES_SAMPLES_MAX_BITS)
|
||||
|
||||
#if WARPED_MOTION_SORT_SAMPLES
|
||||
// #define SAMPLES_ARRAY_SIZE (LEAST_SQUARES_SAMPLES_MAX * 2)
|
||||
// Search half bsize on the top and half bsize on the left, 1 upper-left block,
|
||||
// Search 1 row on the top and 1 column on the left, 1 upper-left block,
|
||||
// 1 upper-right block.
|
||||
#define SAMPLES_ARRAY_SIZE ((MAX_MIB_SIZE * MAX_MIB_SIZE + 2) * 2)
|
||||
#define SAMPLES_ARRAY_SIZE ((MAX_MIB_SIZE * 2 + 2) * 2)
|
||||
#else
|
||||
#define SAMPLES_ARRAY_SIZE (LEAST_SQUARES_SAMPLES_MAX * 2)
|
||||
#endif // WARPED_MOTION_SORT_SAMPLES
|
||||
|
|
|
|||
645
third_party/aom/av1/common/x86/av1_convolve_scale_sse4.c
vendored
Normal file
645
third_party/aom/av1/common/x86/av1_convolve_scale_sse4.c
vendored
Normal file
|
|
@ -0,0 +1,645 @@
|
|||
/*
|
||||
* Copyright (c) 2017, Alliance for Open Media. All rights reserved
|
||||
*
|
||||
* This source code is subject to the terms of the BSD 2 Clause License and
|
||||
* the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
|
||||
* was not distributed with this source code in the LICENSE file, you can
|
||||
* obtain it at www.aomedia.org/license/software. If the Alliance for Open
|
||||
* Media Patent License 1.0 was not distributed with this source code in the
|
||||
* PATENTS file, you can obtain it at www.aomedia.org/license/patent.
|
||||
*/
|
||||
|
||||
#include <assert.h>
|
||||
#include <smmintrin.h>
|
||||
|
||||
#include "./aom_dsp_rtcd.h"
|
||||
#include "aom_dsp/aom_convolve.h"
|
||||
#include "aom_dsp/aom_dsp_common.h"
|
||||
#include "aom_dsp/aom_filter.h"
|
||||
#include "av1/common/convolve.h"
|
||||
|
||||
// Make a mask for coefficients of 10/12 tap filters. The coefficients are
|
||||
// packed "89ab89ab". If it's a 12-tap filter, we want all 1's; if it's a
|
||||
// 10-tap filter, we want "11001100" to just match the 8,9 terms.
|
||||
static __m128i make_1012_mask(int ntaps) {
|
||||
uint32_t low = 0xffffffff;
|
||||
uint32_t high = (ntaps == 12) ? low : 0;
|
||||
return _mm_set_epi32(high, low, high, low);
|
||||
}
|
||||
|
||||
// Zero-extend the given input operand to an entire __m128i register.
|
||||
//
|
||||
// Note that there's almost an intrinsic to do this but 32-bit Visual Studio
|
||||
// doesn't have _mm_set_epi64x so we have to do it by hand.
|
||||
static __m128i extend_32_to_128(uint32_t x) {
|
||||
return _mm_set_epi32(0, 0, 0, x);
|
||||
}
|
||||
|
||||
// Load an SSE register from p and bitwise AND with a.
|
||||
static __m128i load_and_128i(const void *p, __m128i a) {
|
||||
const __m128d ad = _mm_castsi128_pd(a);
|
||||
const __m128d bd = _mm_load1_pd((const double *)p);
|
||||
return _mm_castpd_si128(_mm_and_pd(ad, bd));
|
||||
}
|
||||
|
||||
// The horizontal filter for av1_convolve_2d_scale_sse4_1. This is the more
|
||||
// general version, supporting 10 and 12 tap filters. For 8-tap filters, use
|
||||
// hfilter8.
|
||||
static void hfilter(const uint8_t *src, int src_stride, int32_t *dst, int w,
|
||||
int h, int subpel_x_qn, int x_step_qn,
|
||||
const InterpFilterParams *filter_params, unsigned round) {
|
||||
const int bd = 8;
|
||||
const int ntaps = filter_params->taps;
|
||||
assert(ntaps == 10 || ntaps == 12);
|
||||
|
||||
src -= ntaps / 2 - 1;
|
||||
|
||||
// Construct a mask with which we'll AND filter coefficients 89ab89ab to zero
|
||||
// out the unneeded entries.
|
||||
const __m128i hicoeff_mask = make_1012_mask(ntaps);
|
||||
|
||||
int32_t round_add32 = (1 << round) / 2 + (1 << (bd + FILTER_BITS - 1));
|
||||
const __m128i round_add = _mm_set1_epi32(round_add32);
|
||||
const __m128i round_shift = extend_32_to_128(round);
|
||||
|
||||
int x_qn = subpel_x_qn;
|
||||
for (int x = 0; x < w; ++x, x_qn += x_step_qn) {
|
||||
const uint8_t *const src_col = src + (x_qn >> SCALE_SUBPEL_BITS);
|
||||
const int filter_idx = (x_qn & SCALE_SUBPEL_MASK) >> SCALE_EXTRA_BITS;
|
||||
assert(filter_idx < SUBPEL_SHIFTS);
|
||||
const int16_t *filter =
|
||||
av1_get_interp_filter_subpel_kernel(*filter_params, filter_idx);
|
||||
|
||||
// The "lo" coefficients are coefficients 0..7. For a 12-tap filter, the
|
||||
// "hi" coefficients are arranged as 89ab89ab. For a 10-tap filter, they
|
||||
// are masked out with hicoeff_mask.
|
||||
const __m128i coefflo = _mm_loadu_si128((__m128i *)filter);
|
||||
const __m128i coeffhi = load_and_128i(filter + 8, hicoeff_mask);
|
||||
const __m128i zero = _mm_castps_si128(_mm_setzero_ps());
|
||||
|
||||
int y;
|
||||
for (y = 0; y <= h - 4; y += 4) {
|
||||
const uint8_t *const src0 = src_col + y * src_stride;
|
||||
const uint8_t *const src1 = src0 + 1 * src_stride;
|
||||
const uint8_t *const src2 = src0 + 2 * src_stride;
|
||||
const uint8_t *const src3 = src0 + 3 * src_stride;
|
||||
|
||||
// Load up source data. This is 8-bit input data, so each load gets 16
|
||||
// pixels (we need at most 12)
|
||||
const __m128i data08 = _mm_loadu_si128((__m128i *)src0);
|
||||
const __m128i data18 = _mm_loadu_si128((__m128i *)src1);
|
||||
const __m128i data28 = _mm_loadu_si128((__m128i *)src2);
|
||||
const __m128i data38 = _mm_loadu_si128((__m128i *)src3);
|
||||
|
||||
// Now zero-extend up to 16-bit precision by interleaving with zeros. For
|
||||
// the "high" pixels (8 to 11), interleave first (so that the expansion
|
||||
// to 16-bits operates on an entire register).
|
||||
const __m128i data0lo = _mm_unpacklo_epi8(data08, zero);
|
||||
const __m128i data1lo = _mm_unpacklo_epi8(data18, zero);
|
||||
const __m128i data2lo = _mm_unpacklo_epi8(data28, zero);
|
||||
const __m128i data3lo = _mm_unpacklo_epi8(data38, zero);
|
||||
const __m128i data01hi8 = _mm_unpackhi_epi32(data08, data18);
|
||||
const __m128i data23hi8 = _mm_unpackhi_epi32(data28, data38);
|
||||
const __m128i data01hi = _mm_unpacklo_epi8(data01hi8, zero);
|
||||
const __m128i data23hi = _mm_unpacklo_epi8(data23hi8, zero);
|
||||
|
||||
// Multiply by coefficients
|
||||
const __m128i conv0lo = _mm_madd_epi16(data0lo, coefflo);
|
||||
const __m128i conv1lo = _mm_madd_epi16(data1lo, coefflo);
|
||||
const __m128i conv2lo = _mm_madd_epi16(data2lo, coefflo);
|
||||
const __m128i conv3lo = _mm_madd_epi16(data3lo, coefflo);
|
||||
const __m128i conv01hi = _mm_madd_epi16(data01hi, coeffhi);
|
||||
const __m128i conv23hi = _mm_madd_epi16(data23hi, coeffhi);
|
||||
|
||||
// Reduce horizontally and add
|
||||
const __m128i conv01lo = _mm_hadd_epi32(conv0lo, conv1lo);
|
||||
const __m128i conv23lo = _mm_hadd_epi32(conv2lo, conv3lo);
|
||||
const __m128i convlo = _mm_hadd_epi32(conv01lo, conv23lo);
|
||||
const __m128i convhi = _mm_hadd_epi32(conv01hi, conv23hi);
|
||||
const __m128i conv = _mm_add_epi32(convlo, convhi);
|
||||
|
||||
// Divide down by (1 << round), rounding to nearest.
|
||||
const __m128i shifted =
|
||||
_mm_sra_epi32(_mm_add_epi32(conv, round_add), round_shift);
|
||||
|
||||
// Write transposed to the output
|
||||
_mm_storeu_si128((__m128i *)(dst + y + x * h), shifted);
|
||||
}
|
||||
for (; y < h; ++y) {
|
||||
const uint8_t *const src_row = src_col + y * src_stride;
|
||||
|
||||
int32_t sum = (1 << (bd + FILTER_BITS - 1));
|
||||
for (int k = 0; k < ntaps; ++k) {
|
||||
sum += filter[k] * src_row[k];
|
||||
}
|
||||
|
||||
dst[y + x * h] = ROUND_POWER_OF_TWO(sum, round);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// A specialised version of hfilter, the horizontal filter for
|
||||
// av1_convolve_2d_scale_sse4_1. This version only supports 8 tap filters.
|
||||
static void hfilter8(const uint8_t *src, int src_stride, int32_t *dst, int w,
|
||||
int h, int subpel_x_qn, int x_step_qn,
|
||||
const InterpFilterParams *filter_params, unsigned round) {
|
||||
const int bd = 8;
|
||||
const int ntaps = 8;
|
||||
|
||||
src -= ntaps / 2 - 1;
|
||||
|
||||
int32_t round_add32 = (1 << round) / 2 + (1 << (bd + FILTER_BITS - 1));
|
||||
const __m128i round_add = _mm_set1_epi32(round_add32);
|
||||
const __m128i round_shift = extend_32_to_128(round);
|
||||
|
||||
int x_qn = subpel_x_qn;
|
||||
for (int x = 0; x < w; ++x, x_qn += x_step_qn) {
|
||||
const uint8_t *const src_col = src + (x_qn >> SCALE_SUBPEL_BITS);
|
||||
const int filter_idx = (x_qn & SCALE_SUBPEL_MASK) >> SCALE_EXTRA_BITS;
|
||||
assert(filter_idx < SUBPEL_SHIFTS);
|
||||
const int16_t *filter =
|
||||
av1_get_interp_filter_subpel_kernel(*filter_params, filter_idx);
|
||||
|
||||
// Load the filter coefficients
|
||||
const __m128i coefflo = _mm_loadu_si128((__m128i *)filter);
|
||||
const __m128i zero = _mm_castps_si128(_mm_setzero_ps());
|
||||
|
||||
int y;
|
||||
for (y = 0; y <= h - 4; y += 4) {
|
||||
const uint8_t *const src0 = src_col + y * src_stride;
|
||||
const uint8_t *const src1 = src0 + 1 * src_stride;
|
||||
const uint8_t *const src2 = src0 + 2 * src_stride;
|
||||
const uint8_t *const src3 = src0 + 3 * src_stride;
|
||||
|
||||
// Load up source data. This is 8-bit input data; each load is just
|
||||
// loading the lower half of the register and gets 8 pixels
|
||||
const __m128i data08 = _mm_loadl_epi64((__m128i *)src0);
|
||||
const __m128i data18 = _mm_loadl_epi64((__m128i *)src1);
|
||||
const __m128i data28 = _mm_loadl_epi64((__m128i *)src2);
|
||||
const __m128i data38 = _mm_loadl_epi64((__m128i *)src3);
|
||||
|
||||
// Now zero-extend up to 16-bit precision by interleaving with
|
||||
// zeros. Drop the upper half of each register (which just had zeros)
|
||||
const __m128i data0lo = _mm_unpacklo_epi8(data08, zero);
|
||||
const __m128i data1lo = _mm_unpacklo_epi8(data18, zero);
|
||||
const __m128i data2lo = _mm_unpacklo_epi8(data28, zero);
|
||||
const __m128i data3lo = _mm_unpacklo_epi8(data38, zero);
|
||||
|
||||
// Multiply by coefficients
|
||||
const __m128i conv0lo = _mm_madd_epi16(data0lo, coefflo);
|
||||
const __m128i conv1lo = _mm_madd_epi16(data1lo, coefflo);
|
||||
const __m128i conv2lo = _mm_madd_epi16(data2lo, coefflo);
|
||||
const __m128i conv3lo = _mm_madd_epi16(data3lo, coefflo);
|
||||
|
||||
// Reduce horizontally and add
|
||||
const __m128i conv01lo = _mm_hadd_epi32(conv0lo, conv1lo);
|
||||
const __m128i conv23lo = _mm_hadd_epi32(conv2lo, conv3lo);
|
||||
const __m128i conv = _mm_hadd_epi32(conv01lo, conv23lo);
|
||||
|
||||
// Divide down by (1 << round), rounding to nearest.
|
||||
const __m128i shifted =
|
||||
_mm_sra_epi32(_mm_add_epi32(conv, round_add), round_shift);
|
||||
|
||||
// Write transposed to the output
|
||||
_mm_storeu_si128((__m128i *)(dst + y + x * h), shifted);
|
||||
}
|
||||
for (; y < h; ++y) {
|
||||
const uint8_t *const src_row = src_col + y * src_stride;
|
||||
|
||||
int32_t sum = (1 << (bd + FILTER_BITS - 1));
|
||||
for (int k = 0; k < ntaps; ++k) {
|
||||
sum += filter[k] * src_row[k];
|
||||
}
|
||||
|
||||
dst[y + x * h] = ROUND_POWER_OF_TWO(sum, round);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Do a 12-tap convolution with the given coefficients, loading data from src.
|
||||
static __m128i convolve_32(const int32_t *src, __m128i coeff03, __m128i coeff47,
|
||||
__m128i coeff8d) {
|
||||
const __m128i data03 = _mm_loadu_si128((__m128i *)src);
|
||||
const __m128i data47 = _mm_loadu_si128((__m128i *)(src + 4));
|
||||
const __m128i data8d = _mm_loadu_si128((__m128i *)(src + 8));
|
||||
const __m128i conv03 = _mm_mullo_epi32(data03, coeff03);
|
||||
const __m128i conv47 = _mm_mullo_epi32(data47, coeff47);
|
||||
const __m128i conv8d = _mm_mullo_epi32(data8d, coeff8d);
|
||||
return _mm_add_epi32(_mm_add_epi32(conv03, conv47), conv8d);
|
||||
}
|
||||
|
||||
// Do an 8-tap convolution with the given coefficients, loading data from src.
|
||||
static __m128i convolve_32_8(const int32_t *src, __m128i coeff03,
|
||||
__m128i coeff47) {
|
||||
const __m128i data03 = _mm_loadu_si128((__m128i *)src);
|
||||
const __m128i data47 = _mm_loadu_si128((__m128i *)(src + 4));
|
||||
const __m128i conv03 = _mm_mullo_epi32(data03, coeff03);
|
||||
const __m128i conv47 = _mm_mullo_epi32(data47, coeff47);
|
||||
return _mm_add_epi32(conv03, conv47);
|
||||
}
|
||||
|
||||
// The vertical filter for av1_convolve_2d_scale_sse4_1. This is the more
|
||||
// general version, supporting 10 and 12 tap filters. For 8-tap filters, use
|
||||
// vfilter8.
|
||||
static void vfilter(const int32_t *src, int src_stride, int32_t *dst,
|
||||
int dst_stride, int w, int h, int subpel_y_qn,
|
||||
int y_step_qn, const InterpFilterParams *filter_params,
|
||||
const ConvolveParams *conv_params, int bd) {
|
||||
const int offset_bits = bd + 2 * FILTER_BITS - conv_params->round_0;
|
||||
const int ntaps = filter_params->taps;
|
||||
|
||||
// Construct a mask with which we'll AND filter coefficients 89ab to zero out
|
||||
// the unneeded entries. The upper bits of this mask are unused.
|
||||
const __m128i hicoeff_mask = make_1012_mask(ntaps);
|
||||
|
||||
int32_t round_add32 = (1 << conv_params->round_1) / 2 + (1 << offset_bits);
|
||||
const __m128i round_add = _mm_set1_epi32(round_add32);
|
||||
const __m128i round_shift = extend_32_to_128(conv_params->round_1);
|
||||
|
||||
const int32_t sub32 = ((1 << (offset_bits - conv_params->round_1)) +
|
||||
(1 << (offset_bits - conv_params->round_1 - 1)));
|
||||
const __m128i sub = _mm_set1_epi32(sub32);
|
||||
|
||||
int y_qn = subpel_y_qn;
|
||||
for (int y = 0; y < h; ++y, y_qn += y_step_qn) {
|
||||
const int32_t *src_y = src + (y_qn >> SCALE_SUBPEL_BITS);
|
||||
const int filter_idx = (y_qn & SCALE_SUBPEL_MASK) >> SCALE_EXTRA_BITS;
|
||||
assert(filter_idx < SUBPEL_SHIFTS);
|
||||
const int16_t *filter =
|
||||
av1_get_interp_filter_subpel_kernel(*filter_params, filter_idx);
|
||||
|
||||
// Load up coefficients for the filter and sign-extend to 32-bit precision
|
||||
// (to do so, calculate sign bits and then interleave)
|
||||
const __m128i zero = _mm_castps_si128(_mm_setzero_ps());
|
||||
const __m128i coeff0716 = _mm_loadu_si128((__m128i *)filter);
|
||||
const __m128i coeffhi16 = load_and_128i(filter + 8, hicoeff_mask);
|
||||
const __m128i csign0716 = _mm_cmplt_epi16(coeff0716, zero);
|
||||
const __m128i csignhi16 = _mm_cmplt_epi16(coeffhi16, zero);
|
||||
const __m128i coeff03 = _mm_unpacklo_epi16(coeff0716, csign0716);
|
||||
const __m128i coeff47 = _mm_unpackhi_epi16(coeff0716, csign0716);
|
||||
const __m128i coeff8d = _mm_unpacklo_epi16(coeffhi16, csignhi16);
|
||||
|
||||
int x;
|
||||
for (x = 0; x <= w - 4; x += 4) {
|
||||
const int32_t *const src0 = src_y + x * src_stride;
|
||||
const int32_t *const src1 = src0 + 1 * src_stride;
|
||||
const int32_t *const src2 = src0 + 2 * src_stride;
|
||||
const int32_t *const src3 = src0 + 3 * src_stride;
|
||||
|
||||
// Load the source data for the three rows, adding the three registers of
|
||||
// convolved products to one as we go (conv0..conv3) to avoid the
|
||||
// register pressure getting too high.
|
||||
const __m128i conv0 = convolve_32(src0, coeff03, coeff47, coeff8d);
|
||||
const __m128i conv1 = convolve_32(src1, coeff03, coeff47, coeff8d);
|
||||
const __m128i conv2 = convolve_32(src2, coeff03, coeff47, coeff8d);
|
||||
const __m128i conv3 = convolve_32(src3, coeff03, coeff47, coeff8d);
|
||||
|
||||
// Now reduce horizontally to get one lane for each result
|
||||
const __m128i conv01 = _mm_hadd_epi32(conv0, conv1);
|
||||
const __m128i conv23 = _mm_hadd_epi32(conv2, conv3);
|
||||
const __m128i conv = _mm_hadd_epi32(conv01, conv23);
|
||||
|
||||
// Divide down by (1 << round_1), rounding to nearest and subtract sub32.
|
||||
const __m128i shifted =
|
||||
_mm_sra_epi32(_mm_add_epi32(conv, round_add), round_shift);
|
||||
const __m128i subbed = _mm_sub_epi32(shifted, sub);
|
||||
|
||||
int32_t *dst_x = dst + y * dst_stride + x;
|
||||
const __m128i result =
|
||||
(conv_params->do_average)
|
||||
? _mm_add_epi32(subbed, _mm_loadu_si128((__m128i *)dst_x))
|
||||
: subbed;
|
||||
|
||||
_mm_storeu_si128((__m128i *)dst_x, result);
|
||||
}
|
||||
for (; x < w; ++x) {
|
||||
const int32_t *src_x = src_y + x * src_stride;
|
||||
CONV_BUF_TYPE sum = 1 << offset_bits;
|
||||
for (int k = 0; k < ntaps; ++k) sum += filter[k] * src_x[k];
|
||||
CONV_BUF_TYPE res = ROUND_POWER_OF_TWO(sum, conv_params->round_1) - sub32;
|
||||
if (conv_params->do_average)
|
||||
dst[y * dst_stride + x] += res;
|
||||
else
|
||||
dst[y * dst_stride + x] = res;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// A specialised version of vfilter, the vertical filter for
|
||||
// av1_convolve_2d_scale_sse4_1. This version only supports 8 tap filters.
|
||||
static void vfilter8(const int32_t *src, int src_stride, int32_t *dst,
|
||||
int dst_stride, int w, int h, int subpel_y_qn,
|
||||
int y_step_qn, const InterpFilterParams *filter_params,
|
||||
const ConvolveParams *conv_params, int bd) {
|
||||
const int offset_bits = bd + 2 * FILTER_BITS - conv_params->round_0;
|
||||
const int ntaps = 8;
|
||||
|
||||
int32_t round_add32 = (1 << conv_params->round_1) / 2 + (1 << offset_bits);
|
||||
const __m128i round_add = _mm_set1_epi32(round_add32);
|
||||
const __m128i round_shift = extend_32_to_128(conv_params->round_1);
|
||||
|
||||
const int32_t sub32 = ((1 << (offset_bits - conv_params->round_1)) +
|
||||
(1 << (offset_bits - conv_params->round_1 - 1)));
|
||||
const __m128i sub = _mm_set1_epi32(sub32);
|
||||
|
||||
int y_qn = subpel_y_qn;
|
||||
for (int y = 0; y < h; ++y, y_qn += y_step_qn) {
|
||||
const int32_t *src_y = src + (y_qn >> SCALE_SUBPEL_BITS);
|
||||
const int filter_idx = (y_qn & SCALE_SUBPEL_MASK) >> SCALE_EXTRA_BITS;
|
||||
assert(filter_idx < SUBPEL_SHIFTS);
|
||||
const int16_t *filter =
|
||||
av1_get_interp_filter_subpel_kernel(*filter_params, filter_idx);
|
||||
|
||||
// Load up coefficients for the filter and sign-extend to 32-bit precision
|
||||
// (to do so, calculate sign bits and then interleave)
|
||||
const __m128i zero = _mm_castps_si128(_mm_setzero_ps());
|
||||
const __m128i coeff0716 = _mm_loadu_si128((__m128i *)filter);
|
||||
const __m128i csign0716 = _mm_cmplt_epi16(coeff0716, zero);
|
||||
const __m128i coeff03 = _mm_unpacklo_epi16(coeff0716, csign0716);
|
||||
const __m128i coeff47 = _mm_unpackhi_epi16(coeff0716, csign0716);
|
||||
|
||||
int x;
|
||||
for (x = 0; x <= w - 4; x += 4) {
|
||||
const int32_t *const src0 = src_y + x * src_stride;
|
||||
const int32_t *const src1 = src0 + 1 * src_stride;
|
||||
const int32_t *const src2 = src0 + 2 * src_stride;
|
||||
const int32_t *const src3 = src0 + 3 * src_stride;
|
||||
|
||||
// Load the source data for the three rows, adding the three registers of
|
||||
// convolved products to one as we go (conv0..conv3) to avoid the
|
||||
// register pressure getting too high.
|
||||
const __m128i conv0 = convolve_32_8(src0, coeff03, coeff47);
|
||||
const __m128i conv1 = convolve_32_8(src1, coeff03, coeff47);
|
||||
const __m128i conv2 = convolve_32_8(src2, coeff03, coeff47);
|
||||
const __m128i conv3 = convolve_32_8(src3, coeff03, coeff47);
|
||||
|
||||
// Now reduce horizontally to get one lane for each result
|
||||
const __m128i conv01 = _mm_hadd_epi32(conv0, conv1);
|
||||
const __m128i conv23 = _mm_hadd_epi32(conv2, conv3);
|
||||
const __m128i conv = _mm_hadd_epi32(conv01, conv23);
|
||||
|
||||
// Divide down by (1 << round_1), rounding to nearest and subtract sub32.
|
||||
const __m128i shifted =
|
||||
_mm_sra_epi32(_mm_add_epi32(conv, round_add), round_shift);
|
||||
const __m128i subbed = _mm_sub_epi32(shifted, sub);
|
||||
|
||||
int32_t *dst_x = dst + y * dst_stride + x;
|
||||
const __m128i result =
|
||||
(conv_params->do_average)
|
||||
? _mm_add_epi32(subbed, _mm_loadu_si128((__m128i *)dst_x))
|
||||
: subbed;
|
||||
|
||||
_mm_storeu_si128((__m128i *)dst_x, result);
|
||||
}
|
||||
for (; x < w; ++x) {
|
||||
const int32_t *src_x = src_y + x * src_stride;
|
||||
CONV_BUF_TYPE sum = 1 << offset_bits;
|
||||
for (int k = 0; k < ntaps; ++k) sum += filter[k] * src_x[k];
|
||||
CONV_BUF_TYPE res = ROUND_POWER_OF_TWO(sum, conv_params->round_1) - sub32;
|
||||
if (conv_params->do_average)
|
||||
dst[y * dst_stride + x] += res;
|
||||
else
|
||||
dst[y * dst_stride + x] = res;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void av1_convolve_2d_scale_sse4_1(const uint8_t *src, int src_stride,
|
||||
CONV_BUF_TYPE *dst, int dst_stride, int w,
|
||||
int h, InterpFilterParams *filter_params_x,
|
||||
InterpFilterParams *filter_params_y,
|
||||
const int subpel_x_qn, const int x_step_qn,
|
||||
const int subpel_y_qn, const int y_step_qn,
|
||||
ConvolveParams *conv_params) {
|
||||
int32_t tmp[(2 * MAX_SB_SIZE + MAX_FILTER_TAP) * MAX_SB_SIZE];
|
||||
int im_h = (((h - 1) * y_step_qn + subpel_y_qn) >> SCALE_SUBPEL_BITS) +
|
||||
filter_params_y->taps;
|
||||
|
||||
const int xtaps = filter_params_x->taps;
|
||||
const int ytaps = filter_params_y->taps;
|
||||
|
||||
const int fo_vert = ytaps / 2 - 1;
|
||||
|
||||
// horizontal filter
|
||||
if (xtaps == 8)
|
||||
hfilter8(src - fo_vert * src_stride, src_stride, tmp, w, im_h, subpel_x_qn,
|
||||
x_step_qn, filter_params_x, conv_params->round_0);
|
||||
else
|
||||
hfilter(src - fo_vert * src_stride, src_stride, tmp, w, im_h, subpel_x_qn,
|
||||
x_step_qn, filter_params_x, conv_params->round_0);
|
||||
|
||||
// vertical filter (input is transposed)
|
||||
if (ytaps == 8)
|
||||
vfilter8(tmp, im_h, dst, dst_stride, w, h, subpel_y_qn, y_step_qn,
|
||||
filter_params_y, conv_params, 8);
|
||||
else
|
||||
vfilter(tmp, im_h, dst, dst_stride, w, h, subpel_y_qn, y_step_qn,
|
||||
filter_params_y, conv_params, 8);
|
||||
}
|
||||
|
||||
#if CONFIG_HIGHBITDEPTH
|
||||
// An wrapper to generate the SHUFPD instruction with __m128i types (just
|
||||
// writing _mm_shuffle_pd at the callsites gets a bit ugly because of the
|
||||
// casts)
|
||||
static __m128i mm_shuffle0_si128(__m128i a, __m128i b) {
|
||||
__m128d ad = _mm_castsi128_pd(a);
|
||||
__m128d bd = _mm_castsi128_pd(b);
|
||||
return _mm_castpd_si128(_mm_shuffle_pd(ad, bd, 0));
|
||||
}
|
||||
|
||||
// The horizontal filter for av1_highbd_convolve_2d_scale_sse4_1. This
|
||||
// is the more general version, supporting 10 and 12 tap filters. For
|
||||
// 8-tap filters, use hfilter8.
|
||||
static void highbd_hfilter(const uint16_t *src, int src_stride, int32_t *dst,
|
||||
int w, int h, int subpel_x_qn, int x_step_qn,
|
||||
const InterpFilterParams *filter_params,
|
||||
unsigned round, int bd) {
|
||||
const int ntaps = filter_params->taps;
|
||||
assert(ntaps == 10 || ntaps == 12);
|
||||
|
||||
src -= ntaps / 2 - 1;
|
||||
|
||||
// Construct a mask with which we'll AND filter coefficients 89ab89ab to zero
|
||||
// out the unneeded entries.
|
||||
const __m128i hicoeff_mask = make_1012_mask(ntaps);
|
||||
|
||||
int32_t round_add32 = (1 << round) / 2 + (1 << (bd + FILTER_BITS - 1));
|
||||
const __m128i round_add = _mm_set1_epi32(round_add32);
|
||||
const __m128i round_shift = extend_32_to_128(round);
|
||||
|
||||
int x_qn = subpel_x_qn;
|
||||
for (int x = 0; x < w; ++x, x_qn += x_step_qn) {
|
||||
const uint16_t *const src_col = src + (x_qn >> SCALE_SUBPEL_BITS);
|
||||
const int filter_idx = (x_qn & SCALE_SUBPEL_MASK) >> SCALE_EXTRA_BITS;
|
||||
assert(filter_idx < SUBPEL_SHIFTS);
|
||||
const int16_t *filter =
|
||||
av1_get_interp_filter_subpel_kernel(*filter_params, filter_idx);
|
||||
|
||||
// The "lo" coefficients are coefficients 0..7. For a 12-tap filter, the
|
||||
// "hi" coefficients are arranged as 89ab89ab. For a 10-tap filter, they
|
||||
// are masked out with hicoeff_mask.
|
||||
const __m128i coefflo = _mm_loadu_si128((__m128i *)filter);
|
||||
const __m128i coeffhi = load_and_128i(filter + 8, hicoeff_mask);
|
||||
|
||||
int y;
|
||||
for (y = 0; y <= h - 4; y += 4) {
|
||||
const uint16_t *const src0 = src_col + y * src_stride;
|
||||
const uint16_t *const src1 = src0 + 1 * src_stride;
|
||||
const uint16_t *const src2 = src0 + 2 * src_stride;
|
||||
const uint16_t *const src3 = src0 + 3 * src_stride;
|
||||
|
||||
// Load up source data. This is 16-bit input data, so each load gets 8
|
||||
// pixels (we need at most 12)
|
||||
const __m128i data0lo = _mm_loadu_si128((__m128i *)src0);
|
||||
const __m128i data1lo = _mm_loadu_si128((__m128i *)src1);
|
||||
const __m128i data2lo = _mm_loadu_si128((__m128i *)src2);
|
||||
const __m128i data3lo = _mm_loadu_si128((__m128i *)src3);
|
||||
const __m128i data0hi = _mm_loadu_si128((__m128i *)(src0 + 8));
|
||||
const __m128i data1hi = _mm_loadu_si128((__m128i *)(src1 + 8));
|
||||
const __m128i data2hi = _mm_loadu_si128((__m128i *)(src2 + 8));
|
||||
const __m128i data3hi = _mm_loadu_si128((__m128i *)(src3 + 8));
|
||||
|
||||
// The "hi" data has rubbish in the top half so interleave pairs together
|
||||
// to minimise the calculation we need to do.
|
||||
const __m128i data01hi = mm_shuffle0_si128(data0hi, data1hi);
|
||||
const __m128i data23hi = mm_shuffle0_si128(data2hi, data3hi);
|
||||
|
||||
// Multiply by coefficients
|
||||
const __m128i conv0lo = _mm_madd_epi16(data0lo, coefflo);
|
||||
const __m128i conv1lo = _mm_madd_epi16(data1lo, coefflo);
|
||||
const __m128i conv2lo = _mm_madd_epi16(data2lo, coefflo);
|
||||
const __m128i conv3lo = _mm_madd_epi16(data3lo, coefflo);
|
||||
const __m128i conv01hi = _mm_madd_epi16(data01hi, coeffhi);
|
||||
const __m128i conv23hi = _mm_madd_epi16(data23hi, coeffhi);
|
||||
|
||||
// Reduce horizontally and add
|
||||
const __m128i conv01lo = _mm_hadd_epi32(conv0lo, conv1lo);
|
||||
const __m128i conv23lo = _mm_hadd_epi32(conv2lo, conv3lo);
|
||||
const __m128i convlo = _mm_hadd_epi32(conv01lo, conv23lo);
|
||||
const __m128i convhi = _mm_hadd_epi32(conv01hi, conv23hi);
|
||||
const __m128i conv = _mm_add_epi32(convlo, convhi);
|
||||
|
||||
// Divide down by (1 << round), rounding to nearest.
|
||||
const __m128i shifted =
|
||||
_mm_sra_epi32(_mm_add_epi32(conv, round_add), round_shift);
|
||||
|
||||
// Write transposed to the output
|
||||
_mm_storeu_si128((__m128i *)(dst + y + x * h), shifted);
|
||||
}
|
||||
for (; y < h; ++y) {
|
||||
const uint16_t *const src_row = src_col + y * src_stride;
|
||||
|
||||
int32_t sum = (1 << (bd + FILTER_BITS - 1));
|
||||
for (int k = 0; k < ntaps; ++k) {
|
||||
sum += filter[k] * src_row[k];
|
||||
}
|
||||
|
||||
dst[y + x * h] = ROUND_POWER_OF_TWO(sum, round);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// A specialised version of hfilter, the horizontal filter for
|
||||
// av1_highbd_convolve_2d_scale_sse4_1. This version only supports 8 tap
|
||||
// filters.
|
||||
static void highbd_hfilter8(const uint16_t *src, int src_stride, int32_t *dst,
|
||||
int w, int h, int subpel_x_qn, int x_step_qn,
|
||||
const InterpFilterParams *filter_params,
|
||||
unsigned round, int bd) {
|
||||
const int ntaps = 8;
|
||||
|
||||
src -= ntaps / 2 - 1;
|
||||
|
||||
int32_t round_add32 = (1 << round) / 2 + (1 << (bd + FILTER_BITS - 1));
|
||||
const __m128i round_add = _mm_set1_epi32(round_add32);
|
||||
const __m128i round_shift = extend_32_to_128(round);
|
||||
|
||||
int x_qn = subpel_x_qn;
|
||||
for (int x = 0; x < w; ++x, x_qn += x_step_qn) {
|
||||
const uint16_t *const src_col = src + (x_qn >> SCALE_SUBPEL_BITS);
|
||||
const int filter_idx = (x_qn & SCALE_SUBPEL_MASK) >> SCALE_EXTRA_BITS;
|
||||
assert(filter_idx < SUBPEL_SHIFTS);
|
||||
const int16_t *filter =
|
||||
av1_get_interp_filter_subpel_kernel(*filter_params, filter_idx);
|
||||
|
||||
// Load the filter coefficients
|
||||
const __m128i coefflo = _mm_loadu_si128((__m128i *)filter);
|
||||
|
||||
int y;
|
||||
for (y = 0; y <= h - 4; y += 4) {
|
||||
const uint16_t *const src0 = src_col + y * src_stride;
|
||||
const uint16_t *const src1 = src0 + 1 * src_stride;
|
||||
const uint16_t *const src2 = src0 + 2 * src_stride;
|
||||
const uint16_t *const src3 = src0 + 3 * src_stride;
|
||||
|
||||
// Load up source data. This is 16-bit input data, so each load gets the 8
|
||||
// pixels we need.
|
||||
const __m128i data0lo = _mm_loadu_si128((__m128i *)src0);
|
||||
const __m128i data1lo = _mm_loadu_si128((__m128i *)src1);
|
||||
const __m128i data2lo = _mm_loadu_si128((__m128i *)src2);
|
||||
const __m128i data3lo = _mm_loadu_si128((__m128i *)src3);
|
||||
|
||||
// Multiply by coefficients
|
||||
const __m128i conv0lo = _mm_madd_epi16(data0lo, coefflo);
|
||||
const __m128i conv1lo = _mm_madd_epi16(data1lo, coefflo);
|
||||
const __m128i conv2lo = _mm_madd_epi16(data2lo, coefflo);
|
||||
const __m128i conv3lo = _mm_madd_epi16(data3lo, coefflo);
|
||||
|
||||
// Reduce horizontally and add
|
||||
const __m128i conv01lo = _mm_hadd_epi32(conv0lo, conv1lo);
|
||||
const __m128i conv23lo = _mm_hadd_epi32(conv2lo, conv3lo);
|
||||
const __m128i conv = _mm_hadd_epi32(conv01lo, conv23lo);
|
||||
|
||||
// Divide down by (1 << round), rounding to nearest.
|
||||
const __m128i shifted =
|
||||
_mm_sra_epi32(_mm_add_epi32(conv, round_add), round_shift);
|
||||
|
||||
// Write transposed to the output
|
||||
_mm_storeu_si128((__m128i *)(dst + y + x * h), shifted);
|
||||
}
|
||||
for (; y < h; ++y) {
|
||||
const uint16_t *const src_row = src_col + y * src_stride;
|
||||
|
||||
int32_t sum = (1 << (bd + FILTER_BITS - 1));
|
||||
for (int k = 0; k < ntaps; ++k) {
|
||||
sum += filter[k] * src_row[k];
|
||||
}
|
||||
|
||||
dst[y + x * h] = ROUND_POWER_OF_TWO(sum, round);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void av1_highbd_convolve_2d_scale_sse4_1(
|
||||
const uint16_t *src, int src_stride, CONV_BUF_TYPE *dst, int dst_stride,
|
||||
int w, int h, InterpFilterParams *filter_params_x,
|
||||
InterpFilterParams *filter_params_y, const int subpel_x_qn,
|
||||
const int x_step_qn, const int subpel_y_qn, const int y_step_qn,
|
||||
ConvolveParams *conv_params, int bd) {
|
||||
int32_t tmp[(2 * MAX_SB_SIZE + MAX_FILTER_TAP) * MAX_SB_SIZE];
|
||||
int im_h = (((h - 1) * y_step_qn + subpel_y_qn) >> SCALE_SUBPEL_BITS) +
|
||||
filter_params_y->taps;
|
||||
|
||||
const int xtaps = filter_params_x->taps;
|
||||
const int ytaps = filter_params_y->taps;
|
||||
const int fo_vert = ytaps / 2 - 1;
|
||||
|
||||
// horizontal filter
|
||||
if (xtaps == 8)
|
||||
highbd_hfilter8(src - fo_vert * src_stride, src_stride, tmp, w, im_h,
|
||||
subpel_x_qn, x_step_qn, filter_params_x,
|
||||
conv_params->round_0, bd);
|
||||
else
|
||||
highbd_hfilter(src - fo_vert * src_stride, src_stride, tmp, w, im_h,
|
||||
subpel_x_qn, x_step_qn, filter_params_x,
|
||||
conv_params->round_0, bd);
|
||||
|
||||
// vertical filter (input is transposed)
|
||||
if (ytaps == 8)
|
||||
vfilter8(tmp, im_h, dst, dst_stride, w, h, subpel_y_qn, y_step_qn,
|
||||
filter_params_y, conv_params, bd);
|
||||
else
|
||||
vfilter(tmp, im_h, dst, dst_stride, w, h, subpel_y_qn, y_step_qn,
|
||||
filter_params_y, conv_params, bd);
|
||||
}
|
||||
#endif // CONFIG_HIGHBITDEPTH
|
||||
Some files were not shown because too many files have changed in this diff Show more
Loading…
Add table
Add a link
Reference in a new issue